[{"data":1,"prerenderedAt":969},["ShallowReactive",2],{"navigation":3,"\u002Fblog\u002Fin-place-restart-multi-standby":334,"\u002Fblog\u002Fin-place-restart-multi-standby-surround":958},[4,43,91,154,181,224,262,293],{"title":5,"path":6,"stem":7,"children":8,"icon":42},"Getting Started","\u002Fdocs\u002Fgetting-started","1.docs\u002F1.getting-started\u002F1.index",[9,12,17,22,27,32,37],{"title":10,"path":6,"stem":7,"icon":11},"Getting started","i-lucide-flag",{"title":13,"path":14,"stem":15,"icon":16},"Modules overview","\u002Fdocs\u002Fgetting-started\u002Fmodules-overview","1.docs\u002F1.getting-started\u002F2.modules-overview","i-lucide-boxes",{"title":18,"path":19,"stem":20,"icon":21},"Installation","\u002Fdocs\u002Fgetting-started\u002Finstallation","1.docs\u002F1.getting-started\u002F3.installation","i-lucide-download",{"title":23,"path":24,"stem":25,"icon":26},"License configuration","\u002Fdocs\u002Fgetting-started\u002Flicense-configuration","1.docs\u002F1.getting-started\u002F4.license-configuration","i-lucide-key-round",{"title":28,"path":29,"stem":30,"icon":31},"Your first app","\u002Fdocs\u002Fgetting-started\u002Ffirst-app","1.docs\u002F1.getting-started\u002F5.first-app","i-lucide-square-play",{"title":33,"path":34,"stem":35,"icon":36},"Project setup","\u002Fdocs\u002Fgetting-started\u002Fproject-setup","1.docs\u002F1.getting-started\u002F6.project-setup","i-lucide-package",{"title":38,"path":39,"stem":40,"icon":41},"AI assistants","\u002Fdocs\u002Fgetting-started\u002Fai-assistants","1.docs\u002F1.getting-started\u002F7.ai-assistants","i-lucide-bot",false,{"title":44,"path":45,"stem":46,"children":47,"icon":50},"Core Concepts","\u002Fdocs\u002Fconcepts","1.docs\u002F2.concepts\u002F1.index",[48,51,56,61,66,71,76,81,86],{"title":49,"path":45,"stem":46,"icon":50},"Core concepts","i-lucide-book-open",{"title":52,"path":53,"stem":54,"icon":55},"Architecture","\u002Fdocs\u002Fconcepts\u002Farchitecture","1.docs\u002F2.concepts\u002F2.architecture","i-lucide-layers",{"title":57,"path":58,"stem":59,"icon":60},"Exactly-once semantics","\u002Fdocs\u002Fconcepts\u002Fexactly-once","1.docs\u002F2.concepts\u002F3.exactly-once","i-lucide-shield-check",{"title":62,"path":63,"stem":64,"icon":65},"Lanes and parallelism","\u002Fdocs\u002Fconcepts\u002Flanes-and-parallelism","1.docs\u002F2.concepts\u002F4.lanes-and-parallelism","i-lucide-split",{"title":67,"path":68,"stem":69,"icon":70},"State and thread-safety","\u002Fdocs\u002Fconcepts\u002Fstate-and-thread-safety","1.docs\u002F2.concepts\u002F5.state-and-thread-safety","i-lucide-lock",{"title":72,"path":73,"stem":74,"icon":75},"Event time and watermarks","\u002Fdocs\u002Fconcepts\u002Fevent-time-and-watermarks","1.docs\u002F2.concepts\u002F6.event-time-and-watermarks","i-lucide-clock",{"title":77,"path":78,"stem":79,"icon":80},"The configuration model","\u002Fdocs\u002Fconcepts\u002Fconfiguration-model","1.docs\u002F2.concepts\u002F7.configuration-model","i-lucide-sliders-horizontal",{"title":82,"path":83,"stem":84,"icon":85},"The error-handling model","\u002Fdocs\u002Fconcepts\u002Ferror-handling-model","1.docs\u002F2.concepts\u002F8.error-handling-model","i-lucide-triangle-alert",{"title":87,"path":88,"stem":89,"icon":90},"How StoatFlow differs from Kafka Streams","\u002Fdocs\u002Fconcepts\u002Fhow-stoatflow-differs-from-ks","1.docs\u002F2.concepts\u002F9.how-stoatflow-differs-from-ks","i-lucide-arrow-left-right",{"title":92,"path":93,"stem":94,"children":95,"icon":98},"Building Topologies","\u002Fdocs\u002Fbuilding","1.docs\u002F3.building\u002F1.index",[96,99,104,109,114,119,124,129,134,139,144,149],{"title":97,"path":93,"stem":94,"icon":98},"Building topologies","i-lucide-blocks",{"title":100,"path":101,"stem":102,"icon":103},"Error handling and DLQ","\u002Fdocs\u002Fbuilding\u002Ferror-handling-dlq","1.docs\u002F3.building\u002F10.error-handling-dlq","i-lucide-circle-x",{"title":105,"path":106,"stem":107,"icon":108},"State stores","\u002Fdocs\u002Fbuilding\u002Fstate-stores","1.docs\u002F3.building\u002F11.state-stores","i-lucide-database",{"title":110,"path":111,"stem":112,"icon":113},"Testing topologies","\u002Fdocs\u002Fbuilding\u002Ftesting","1.docs\u002F3.building\u002F12.testing","i-lucide-flask-conical",{"title":115,"path":116,"stem":117,"icon":118},"Sources and sinks","\u002Fdocs\u002Fbuilding\u002Fstreams-builder","1.docs\u002F3.building\u002F2.streams-builder","i-lucide-import",{"title":120,"path":121,"stem":122,"icon":123},"KStream and KTable operations","\u002Fdocs\u002Fbuilding\u002Fkstream-ktable","1.docs\u002F3.building\u002F3.kstream-ktable","i-lucide-waypoints",{"title":125,"path":126,"stem":127,"icon":128},"Aggregations","\u002Fdocs\u002Fbuilding\u002Faggregations","1.docs\u002F3.building\u002F4.aggregations","i-lucide-sigma",{"title":130,"path":131,"stem":132,"icon":133},"Windowing","\u002Fdocs\u002Fbuilding\u002Fwindowing","1.docs\u002F3.building\u002F5.windowing","i-lucide-calendar-clock",{"title":135,"path":136,"stem":137,"icon":138},"Joins","\u002Fdocs\u002Fbuilding\u002Fjoins","1.docs\u002F3.building\u002F6.joins","i-lucide-git-merge",{"title":140,"path":141,"stem":142,"icon":143},"The Processor API","\u002Fdocs\u002Fbuilding\u002Fprocessor-api","1.docs\u002F3.building\u002F7.processor-api","i-lucide-cpu",{"title":145,"path":146,"stem":147,"icon":148},"Scheduled sources","\u002Fdocs\u002Fbuilding\u002Fscheduled-sources","1.docs\u002F3.building\u002F8.scheduled-sources","i-lucide-timer",{"title":150,"path":151,"stem":152,"icon":153},"Serdes and Avro","\u002Fdocs\u002Fbuilding\u002Fserdes","1.docs\u002F3.building\u002F9.serdes","i-lucide-binary",{"title":155,"path":156,"stem":157,"children":158,"icon":80},"Configuration","\u002Fdocs\u002Fconfiguration","1.docs\u002F4.configuration\u002F1.index",[159,161,166,171,176],{"title":160,"path":156,"stem":157,"icon":80},"How configuration works",{"title":162,"path":163,"stem":164,"icon":165},"Engine configuration (:core)","\u002Fdocs\u002Fconfiguration\u002Fcore-config","1.docs\u002F4.configuration\u002F2.core-config","i-lucide-settings-2",{"title":167,"path":168,"stem":169,"icon":170},"Runtime configuration (:runtime)","\u002Fdocs\u002Fconfiguration\u002Fruntime-config","1.docs\u002F4.configuration\u002F3.runtime-config","i-lucide-server-cog",{"title":172,"path":173,"stem":174,"icon":175},"Defaults, adaptivity, and presets","\u002Fdocs\u002Fconfiguration\u002Fdefaults-and-presets","1.docs\u002F4.configuration\u002F4.defaults-and-presets","i-lucide-gauge",{"title":177,"path":178,"stem":179,"icon":180},"Kafka client configuration","\u002Fdocs\u002Fconfiguration\u002Fkafka-client-config","1.docs\u002F4.configuration\u002F5.kafka-client-config","i-lucide-plug",{"title":182,"path":183,"stem":184,"children":185,"icon":188},"Running in Production","\u002Fdocs\u002Fruntime","1.docs\u002F5.runtime\u002F1.index",[186,189,194,199,204,209,214,219],{"title":187,"path":183,"stem":184,"icon":188},"The runtime","i-lucide-server",{"title":190,"path":191,"stem":192,"icon":193},"The REST API","\u002Fdocs\u002Fruntime\u002Frest-api","1.docs\u002F5.runtime\u002F2.rest-api","i-lucide-globe",{"title":195,"path":196,"stem":197,"icon":198},"Health checks","\u002Fdocs\u002Fruntime\u002Fhealth-checks","1.docs\u002F5.runtime\u002F3.health-checks","i-lucide-heart-pulse",{"title":200,"path":201,"stem":202,"icon":203},"Metrics","\u002Fdocs\u002Fruntime\u002Fmetrics","1.docs\u002F5.runtime\u002F4.metrics","i-lucide-activity",{"title":205,"path":206,"stem":207,"icon":208},"Pause and resume","\u002Fdocs\u002Fruntime\u002Fpause-unpause","1.docs\u002F5.runtime\u002F5.pause-unpause","i-lucide-pause",{"title":210,"path":211,"stem":212,"icon":213},"Plugins and lifecycle hooks","\u002Fdocs\u002Fruntime\u002Fplugins","1.docs\u002F5.runtime\u002F6.plugins","i-lucide-puzzle",{"title":215,"path":216,"stem":217,"icon":218},"Docker images","\u002Fdocs\u002Fruntime\u002Fdocker","1.docs\u002F5.runtime\u002F7.docker","i-lucide-container",{"title":220,"path":221,"stem":222,"icon":223},"GraalVM native image","\u002Fdocs\u002Fruntime\u002Fnative-image","1.docs\u002F5.runtime\u002F8.native-image","i-lucide-zap",{"title":225,"path":226,"stem":227,"children":228,"icon":231},"Deploying & Operating","\u002Fdocs\u002Foperating","1.docs\u002F6.operating\u002F1.index",[229,232,237,242,247,252,257],{"title":230,"path":226,"stem":227,"icon":231},"Deploying and operating","i-lucide-life-buoy",{"title":233,"path":234,"stem":235,"icon":236},"Running on Kubernetes","\u002Fdocs\u002Foperating\u002Fkubernetes","1.docs\u002F6.operating\u002F2.kubernetes","i-lucide-ship",{"title":238,"path":239,"stem":240,"icon":241},"High availability","\u002Fdocs\u002Foperating\u002Fhigh-availability","1.docs\u002F6.operating\u002F3.high-availability","i-lucide-copy",{"title":243,"path":244,"stem":245,"icon":246},"Liveness and readiness probes","\u002Fdocs\u002Foperating\u002Fprobes","1.docs\u002F6.operating\u002F4.probes","i-lucide-stethoscope",{"title":248,"path":249,"stem":250,"icon":251},"Observability","\u002Fdocs\u002Foperating\u002Fobservability","1.docs\u002F6.operating\u002F5.observability","i-lucide-telescope",{"title":253,"path":254,"stem":255,"icon":256},"Tuning under load","\u002Fdocs\u002Foperating\u002Ftuning","1.docs\u002F6.operating\u002F6.tuning","i-lucide-sliders",{"title":258,"path":259,"stem":260,"icon":261},"Production checklist","\u002Fdocs\u002Foperating\u002Fproduction-checklist","1.docs\u002F6.operating\u002F7.production-checklist","i-lucide-clipboard-check",{"title":263,"path":264,"stem":265,"children":266,"icon":90},"Migrating from Kafka Streams","\u002Fdocs\u002Fmigration","1.docs\u002F7.migration\u002F1.index",[267,268,273,278,283,288],{"title":263,"path":264,"stem":265,"icon":90},{"title":269,"path":270,"stem":271,"icon":272},"Automated port","\u002Fdocs\u002Fmigration\u002Fautomated-port","1.docs\u002F7.migration\u002F2.automated-port","i-lucide-wand-sparkles",{"title":274,"path":275,"stem":276,"icon":277},"Migration without carrying state","\u002Fdocs\u002Fmigration\u002Fwithout-data-migration","1.docs\u002F7.migration\u002F3.without-data-migration","i-lucide-sparkles",{"title":279,"path":280,"stem":281,"icon":282},"Migration carrying state","\u002Fdocs\u002Fmigration\u002Fwith-data-migration","1.docs\u002F7.migration\u002F4.with-data-migration","i-lucide-database-backup",{"title":284,"path":285,"stem":286,"icon":287},"The migration tool","\u002Fdocs\u002Fmigration\u002Fmigration-tool","1.docs\u002F7.migration\u002F5.migration-tool","i-lucide-truck",{"title":289,"path":290,"stem":291,"icon":292},"Reusing your Kafka Streams dashboards","\u002Fdocs\u002Fmigration\u002Freusing-kafka-streams-dashboards","1.docs\u002F7.migration\u002F6.reusing-kafka-streams-dashboards","i-lucide-line-chart",{"title":294,"path":295,"stem":296,"children":297,"icon":299},"Reference","\u002Fdocs\u002Freference","1.docs\u002F8.reference\u002F1.index",[298,300,305,310,315,320,325,329],{"title":294,"path":295,"stem":296,"icon":299},"i-lucide-list",{"title":301,"path":302,"stem":303,"icon":304},"Configuration reference","\u002Fdocs\u002Freference\u002Fconfiguration-reference","1.docs\u002F8.reference\u002F2.configuration-reference","i-lucide-table",{"title":306,"path":307,"stem":308,"icon":309},"REST API reference","\u002Fdocs\u002Freference\u002Frest-api-reference","1.docs\u002F8.reference\u002F3.rest-api-reference","i-lucide-network",{"title":311,"path":312,"stem":313,"icon":314},"Gradle plugin reference","\u002Fdocs\u002Freference\u002Fgradle-plugin-reference","1.docs\u002F8.reference\u002F4.gradle-plugin-reference","i-lucide-box",{"title":316,"path":317,"stem":318,"icon":319},"Maven reference","\u002Fdocs\u002Freference\u002Fmaven-reference","1.docs\u002F8.reference\u002F5.maven-reference","i-simple-icons-apachemaven",{"title":321,"path":322,"stem":323,"icon":324},"Kafka Streams compatibility matrix","\u002Fdocs\u002Freference\u002Fks-compatibility-matrix","1.docs\u002F8.reference\u002F6.ks-compatibility-matrix","i-lucide-table-2",{"title":326,"path":327,"stem":328,"icon":175},"Metrics reference","\u002Fdocs\u002Freference\u002Fmetrics-reference","1.docs\u002F8.reference\u002F7.metrics-reference",{"title":330,"path":331,"stem":332,"icon":333},"Glossary","\u002Fdocs\u002Freference\u002Fglossary","1.docs\u002F8.reference\u002F8.glossary","i-lucide-book-a",{"id":335,"title":336,"authors":337,"badge":343,"body":345,"date":947,"description":948,"draft":42,"extension":949,"image":950,"meta":952,"navigation":953,"path":954,"seo":955,"stem":956,"__hash__":957},"posts\u002F3.blog\u002F7.in-place-restart-multi-standby.md","In-place engine restart: the primitive behind multi-standby HA",[338],{"name":339,"to":340,"avatar":341},"Hartmut Armbruster","https:\u002F\u002Fwww.linkedin.com\u002Fin\u002Fhartmut-co-uk\u002F",{"src":342},"\u002Fassets\u002Fhartmut_armbruster_monochromatic.jpg",{"label":344},"Engineering",{"type":346,"value":347,"toc":935},"minimark",[348,367,374,443,448,492,495,506,512,515,522,537,547,550,557,576,595,598,604,622,638,655,666,669,684,690,821,828,831,834,837,893,896,914],[349,350,351,352,357,358,362,363,366],"p",{},"The ",[353,354,356],"a",{"href":355},"\u002Fblog\u002Fha-failover-testing","failover-testing"," post ended on a piece of work that was still exploratory, and a constraint it told you to live with. The work: tear down and rebuild the processing engine ",[359,360,361],"em",{},"in the same process",", without exiting the JVM, and resume from the last committed transactional offsets — an in-place restart. The constraint, stated plainly in its tradeoffs: ",[359,364,365],{},"two is the supported topology; stay at two",". Both have moved. The in-place restart shipped, and a real leader election shipped on top of it — so scaling a StoatFlow HA cluster past two standbys, which used to crash-loop, now elects exactly one successor and holds.",[349,368,369,370,373],{},"Two changes, and they belong together. This is what they are, why the pair needed them, and what a live three-pod cluster does with them — measured on StoatFlow ",[371,372],"stoatflow-version",{},", an exactly-once word-count deployment on Kubernetes.",[375,376,377,383],"blockquote",{},[349,378,379],{},[380,381,382],"strong",{},"TL;DR",[384,385,386,401,407,417,428,434],"ul",{},[387,388,389,392,393,396,397,400],"li",{},[380,390,391],{},"What:"," An ",[380,394,395],{},"in-place engine restart"," — rebuild the engine inside the live JVM, no process exit — and a ",[380,398,399],{},"leader election"," built on it that makes one active plus K warm standbys a supported topology.",[387,402,403,406],{},[380,404,405],{},"Why:"," Every role change used to wrap engine teardown in a process restart, so a graceful demote bounced the pod and paid a cold store open. And scaling the pair past two crash-looped: competing standbys promoted at once and the exactly-once fence resolved the losers by killing them.",[387,408,409,412,413,416],{},[380,410,411],{},"How:"," The restart swaps the engine instance in place and resumes from the last committed offsets. Promotion now goes through a token claimed ",[359,414,415],{},"before"," the fence, as the sole authority to promote; a standby that loses the claim stands down as a return value, not a crash.",[387,418,419,422,423,427],{},[380,420,421],{},"The guarantee:"," Exactly-once is unchanged — the same ",[424,425,426],"code",{},"transactional.id"," epoch fence that protects a cross-pod handoff protects an in-place one, and it stays the backstop behind the election.",[387,429,430,433],{},[380,431,432],{},"The catch:"," Still exactly one active. K standbys are redundancy, not scale — and each one is another changelog reader. To go faster, scale up, not out.",[387,435,436,439,440,442],{},[380,437,438],{},"Docs:"," ",[353,441,238],{"href":239},".",[444,445,447],"h2",{"id":446},"in-this-post","In this post",[384,449,450,456,462,468,474,480,486],{},[387,451,452],{},[353,453,455],{"href":454},"#the-process-round-trip-we-set-out-to-remove","The process round-trip we set out to remove",[387,457,458],{},[353,459,461],{"href":460},"#restarting-in-place","Restarting in place",[387,463,464],{},[353,465,467],{"href":466},"#from-a-pair-to-a-cluster","From a pair to a cluster",[387,469,470],{},[353,471,473],{"href":472},"#a-real-election","A real election",[387,475,476],{},[353,477,479],{"href":478},"#what-the-cluster-does-now","What the cluster does now",[387,481,482],{},[353,483,485],{"href":484},"#tradeoffs-and-limits","Tradeoffs and limits",[387,487,488],{},[353,489,491],{"href":490},"#running-it","Running it",[444,493,455],{"id":494},"the-process-round-trip-we-set-out-to-remove",[349,496,497,498,501,502,505],{},"Look at the four ways a StoatFlow active can lose its role — a graceful ",[424,499,500],{},"\u002Fha\u002Fswitch",", a SIGTERM from a rolling deploy, a JVM crash, a lost node — and one round-trip recurs underneath all of them: the active ",[359,503,504],{},"exits the process and lets Kubernetes restart it",". A graceful demote self-terminated and came back as a fresh pod. A crash was a container restart. Even an in-place recovery was the orchestrator rebuilding the whole container. The engine teardown and re-initialisation always happened, but always wrapped in a process restart.",[349,507,508,509,511],{},"That wrapper is expensive in exactly the case you most want to be cheap. A ",[424,510,500],{}," is a planned, graceful handoff — it should cost about one commit. But if the demoted active has to exit and be rescheduled, it pays the full price of a cold start on the way back: a new container, a fresh JVM, and a RocksDB store that opens and replays the changelog before the pod is a warm standby again. On large state that is minutes, for an operation whose useful work took milliseconds. The failover-testing measurements named this as the last remaining cost on the fast paths — the promotion itself was already sub-second; what was left was the process round-trip around it.",[444,513,461],{"id":514},"restarting-in-place",[349,516,517,518,521],{},"The in-place engine restart removes the wrapper. Instead of exiting, the instance swaps its engine: it tears down the old ",[424,519,520],{},"StreamProcessingEngine"," and builds a fresh one inside the same live process, then resumes from the last committed transactional offsets. The JVM never exits, the container is never rescheduled, and — the part that matters for recovery time — the state stores are never closed and reopened. The store registry is reused across the swap, so a fresh engine attaches to the already-open, already-warm RocksDB rather than cold-starting it.",[349,523,524,525,528,529,532,533,536],{},"Two dispositions cover the reasons an engine gets rebuilt. A ",[380,526,527],{},"graceful"," restart drains the lanes, broadcasts a final commit barrier, and commits the in-flight transaction before tearing down — the new engine resumes from a clean, committed boundary with nothing to reprocess. An ",[380,530,531],{},"abort"," restart throws the in-flight epoch away: it aborts the open transaction, tears down, and the fresh engine reprocesses the uncommitted tail. That is an in-process crash recovery, and it is what a transient processing fault gets — the faithful analogue of Kafka Streams' ",[424,534,535],{},"REPLACE_THREAD",", which recycles the failed stream thread and keeps the application up. StoatFlow has no per-partition thread to recycle, so until now a fault that Kafka Streams would have shrugged off took the whole instance down. Now it rebuilds the engine and stays running, behind a fault-only budget that escalates to a real shutdown if restarts start looping.",[349,538,539,540,542,543,546],{},"Exactly-once holds across the swap the same way it holds across a pod restart. The fresh engine's transactional producer is fenced by its ",[424,541,426],{}," epoch — the identical mechanism that fences a crashed pod's producer when a standby takes over — so an aborted epoch's writes can never reach a ",[424,544,545],{},"read_committed"," consumer, whether the engine that wrote them died with the JVM or was torn down inside it. The restart is scheduler-agnostic: the same primitive serves a transient fault, a hot-standby demote, and — later — dynamic lane re-tuning. Build it once, cleanly, and each of those gets faster.",[444,548,467],{"id":549},"from-a-pair-to-a-cluster",[349,551,552,553,556],{},"Hot standby shipped as a pair: one active, one warm standby. One spare is enough to survive a lost node, but operators wanted more — a second spare so a rolling deploy never drops to zero redundancy, and headroom to lose a node ",[359,554,555],{},"during"," a deploy. The obvious move is to raise the replica count. It didn't work.",[349,558,559,560,563,564,567,568,571,572,575],{},"Scaling the pair to three replicas crash-looped. Two standbys would decide to promote at the same moment; the exactly-once fence did its job and let only one commit — but it resolved the losers by fencing their producers, which surfaced as ",[424,561,562],{},"ProducerFencedException",", flipped them to ",[424,565,566],{},"ERROR",", and let Kubernetes restart them straight back into the same race. The failover-testing post measured the pair and, in its tradeoffs, said exactly this: ",[359,569,570],{},"stay at two",". On the live cluster the failure was not subtle — three pods logged ",[380,573,574],{},"52, 53, and 53 restarts in twelve hours"," before the fix.",[349,577,578,579,582,583,586,587,594],{},"The root cause is worth stating precisely, because it explains the shape of the fix. StoatFlow had a strong ",[359,580,581],{},"fencing"," layer and a weak ",[359,584,585],{},"election"," layer. Fencing — the transactional producer epoch — guarantees at most one instance ever commits, and it is airtight. But it is a safety mechanism, not a selection one: when two standbys promote together, the fence decides the winner by ",[359,588,589,590,593],{},"who called ",[424,591,592],{},"initTransactions()"," last",", which is a function of timing, not of which standby is furthest caught up. And its way of saying no to a loser is to kill it. With one standby there is never a second promoter, so the gap never showed. With two, the election collapsed onto the fence — and the fence elects by accident and resolves by casualty.",[444,596,473],{"id":597},"a-real-election",[349,599,600,601,603],{},"The fix is to add the election layer the pair never needed: pick the successor deliberately, and ",[359,602,415],{}," anyone touches the fence.",[349,605,606,607,610,611,614,615,617,618,621],{},"Promotion now goes through a ",[380,608,609],{},"promotion token"," — a single record on the same compacted coordination topic the pair already uses, claimed by a compare-and-set. Winning the claim is the sole authority to promote: a standby reads the current holder, and if its claim goes through it proceeds to fence-and-restore; if it loses, it ",[380,612,613],{},"stands down"," — it becomes a standby again and waits, as an ordinary return value, with no fenced producer, no ",[424,616,566],{},", no restart. That single reordering — claim the token, ",[359,619,620],{},"then"," fence — is what turns the crash-loop into a clean stand-down. The loser never reaches the fence, so the fence never has to kill it.",[349,623,624,625,629,630,633,634,637],{},"Who claims? The token is contended by the election winner, and the winner is chosen the way the ",[353,626,628],{"href":627},"\u002Fblog\u002Fhot-standby-high-availability","hot-standby post"," said it would be — lag-aware, so the freshest standby wins. Candidates are ranked by replication lag; the most caught-up wins, and an exact tie breaks on a stable hash of the pod identity, so the choice always converges on one even with no coordinator to ask. Lag narrows the field to who ",[359,631,632],{},"should"," take over; the hash guarantees the field narrows to ",[359,635,636],{},"one","; the token compare-and-set serialises the claim, so that even if two pods pick differently under a brief split view, only one wins and the other stands down.",[349,639,640,641,644,645,647,648,650,651,654],{},"The load-bearing rule is when a claimant may take the token from a holder that already has it. The answer: only if the holder is ",[359,642,643],{},"unprotected"," — either stale, its heartbeats gone quiet past the staleness window so it has likely crashed, or gracefully draining, on its way out and saying so. Against a live, healthy holder, a claimant stands down. That one rule does double duty. It lets a genuinely dead active's token be reclaimed quickly, and it lets a ",[359,646,527],{}," demote hand off immediately: the demoting active publishes that it is draining, which marks its token takeable, so the designated successor claims it at once instead of waiting out the staleness clock. This is exactly where the in-place restart plugs in — a graceful ",[424,649,500],{}," now demotes the old active ",[359,652,653],{},"in place",": it drains, drops the token, rebuilds its engine as a standby in the same pod, and rejoins warm. No exit, no reschedule, no cold store open.",[349,656,657,658,661,662,665],{},"The fence does not go away. It is still there, still airtight, still the thing that makes split-brain impossible under exactly-once. The token and the fence divide the work cleanly: the token is an ",[380,659,660],{},"availability"," mechanism — it bounds how often two pods ever contend to promote — and the fence is the ",[380,663,664],{},"correctness"," mechanism — it bounds what happens if they do. Correctness never depends on the token being perfect. If the election ever misfires, the fence still guarantees one committer; the token just means the loser learns it lost by reading a record instead of by being executed.",[444,667,479],{"id":668},"what-the-cluster-does-now",[349,670,671,672,675,676,679,680,683],{},"The proof is the same three-replica deployment, on the same cluster, with only the build changed. On the old image it logged 52, 53, and 53 restarts across the three pods in twelve hours. On the fixed image it logs ",[380,673,674],{},"zero"," — steady state is one ",[424,677,678],{},"ACTIVE"," and two ",[424,681,682],{},"READY_STANDBY",", a single token holder, and a token epoch that ticks up once per election rather than climbing without bound.",[349,685,686,687,689],{},"Every way an active can lose its role was run against the three-pod cluster, under load, exactly-once, with the three pods co-located on one worker. In each, the election picked one successor and no pod ever reached ",[424,688,566],{}," or CrashLoopBackOff.",[691,692,693,712],"table",{},[694,695,696],"thead",{},[697,698,699,703,706,709],"tr",{},[700,701,702],"th",{},"Scenario",[700,704,705],{},"Mechanism",[700,707,708],{},"Outcome",[700,710,711],{},"Pods restarted",[713,714,715,734,754,770,786,804],"tbody",{},[697,716,717,723,728,731],{},[718,719,720],"td",{},[380,721,722],{},"Graceful switch",[718,724,725,727],{},[424,726,500],{}," → in-place role swap",[718,729,730],{},"freshest standby elected; old active demotes in place",[718,732,733],{},"none",[697,735,736,741,744,751],{},[718,737,738],{},[380,739,740],{},"SIGTERM \u002F rolling deploy",[718,742,743],{},"shutdown-hook handoff",[718,745,746,747,750],{},"a standby elected on the ",[424,748,749],{},"DRAINING"," signal",[718,752,753],{},"old active recreated by the StatefulSet; standbys none",[697,755,756,761,764,767],{},[718,757,758],{},[380,759,760],{},"JVM crash",[718,762,763],{},"SIGKILL the active's JVM",[718,765,766],{},"same pod restarts in place and re-promotes; standbys untouched",[718,768,769],{},"the crashed pod only",[697,771,772,777,780,783],{},[718,773,774],{},[380,775,776],{},"Node loss",[718,778,779],{},"cordon + SIGKILL + force-delete",[718,781,782],{},"a standby elected via staleness detection",[718,784,785],{},"crashed pod recreated; standbys none",[697,787,788,793,798,801],{},[718,789,790],{},[380,791,792],{},"Rolling upgrade",[718,794,795],{},[424,796,797],{},"rollout restart",[718,799,800],{},"one pod not-ready at a time, active rolled last",[718,802,803],{},"one at a time, redundancy preserved",[697,805,806,811,816,819],{},[718,807,808],{},[380,809,810],{},"Repeated switch ×3",[718,812,813,815],{},[424,814,500],{}," in a loop",[718,817,818],{},"one successor each time; never deadlocked in all-standby",[718,820,733],{},[349,822,823,824,827],{},"Two honesty notes on the numbers behind that table. The promotion critical path — from deciding to promote to serving — measured between roughly 0.75 and 1.8 seconds across these runs, but this is a CPU-constrained box: three engines and a load generator sharing one eight-core worker, heavier than the two-worker pair the ",[353,825,826],{"href":355},"earlier measurements"," ran on. Don't read these as faster than those — it is a different, tighter test. And the wall-clock on the ungraceful paths is dominated by things that are not the election: a crashed pod carrying twelve hours of state spends most of its recovery in RocksDB restore, and a lost node spends most of its stop-the-world in the ~7-second staleness detection window. The election itself is the fast part; the numbers around it are state size and detection, as they were for the pair.",[349,829,830],{},"The rolling upgrade is where the extra standby earns its place. Readiness is gated not only on being caught up but on redundancy: a caught-up standby reports ready to roll only if rolling it would still leave the active plus at least one other caught-up standby. So a three-pod roll takes one pod down at a time, highest ordinal first, the active last — and never drops below an active and a warm spare while it runs. The pair had no spare to preserve during a roll; a cluster does, and the roll refuses to spend it.",[444,832,485],{"id":833},"tradeoffs-and-limits",[349,835,836],{},"The election and the in-place restart change how a cluster behaves; they do not change what StoatFlow is. The edges are worth stating plainly.",[384,838,839,849,859,865,875,881],{},[387,840,841,844,845,848],{},[380,842,843],{},"It is still exactly one active."," K standbys are redundancy, not throughput — each is a warm spare, none processes source records. This is the single-instance model intact: you don't pay the ",[359,846,847],{},"distribution tax"," for scale-out you don't need, and to go faster you give the active more cores, not more replicas.",[387,850,851,854,855,858],{},[380,852,853],{},"Each standby is another changelog reader."," A standby stays warm by streaming the changelog the active writes, so K standbys multiply that read fan-out K-fold. ",[424,856,857],{},"ha.max-standbys"," bounds it — pick the redundancy you need, not the maximum you can spell.",[387,860,861,864],{},[380,862,863],{},"A busy active on a shared node has less headroom than it looks."," The co-located drill surfaced this: three engines and a load generator on one worker, and a fan-out workload can starve the active of CPU until a commit misses its window and the exactly-once machinery times the transaction out. That now triggers a clean in-place restart rather than a hung pod — but it is still a restart. Give the active room; the in-place restart makes the failure graceful, not free.",[387,866,867,874],{},[380,868,869,870,873],{},"Node loss on a co-located cluster is a lost ",[359,871,872],{},"active",", not a lost node."," If every standby shares a worker with the active, losing that worker loses all of them. A standby is insurance against losing a node only if it lives on a different one — spread the replicas across nodes for real node-loss protection.",[387,876,877,880],{},[380,878,879],{},"The token is availability; the fence is correctness."," The election reduces how often two pods contend to promote; it does not, and is not relied upon to, guarantee one committer. The transactional fence does that — under a network partition and under a misfiring election alike. Under at-least-once there is no fence, so the token is the whole story; treat it as bounding contention, not eliminating it.",[387,882,883,889,890,892],{},[380,884,885,886,888],{},"A ",[424,887,562],{}," on the demoting pod during a graceful switch is expected, not a fault."," The successor fences the old active's producer while it drains; the demoted pod moves through restore and back to a warm standby without ever reaching ",[424,891,566],{},". It is the fence doing its job on a handoff, not a crash.",[444,894,491],{"id":895},"running-it",[349,897,898,899,902,903,906,907,909,910,913],{},"The topology is a configuration change on the deployment you already run. Hot standby is still one knob — ",[424,900,901],{},"stoatflow.ha.mode: active-standby"," — and going past two is the replica count plus a small amount of intent: ",[424,904,905],{},"ha.desired-standbys"," sets the redundancy the readiness gate protects, ",[424,908,857],{}," caps the changelog fan-out, and ",[424,911,912],{},"ha.failover-priority"," nudges which equally-caught-up standby is preferred. The election, the token, the staleness thresholds, and the in-place demote all have working defaults; a typical deployment sets the mode and the replica count and leaves the rest.",[349,915,916,917,919,920,923,924,926,927,930,931,934],{},"The model is the one the ",[353,918,628],{"href":627}," described, now without its two footnotes. One active, warm spares, the election picks the freshest, the fence backstops — and a planned role change no longer bounces a pod or waits on a cold store. The two footnotes it shipped with — ",[359,921,922],{},"promotion will become lag-aware"," and ",[359,925,570],{}," — are both closed. The ",[353,928,929],{"href":239},"High availability guide"," has the configuration reference and the operator endpoints, and ",[353,932,933],{"href":58},"Exactly-once"," covers the guarantee the fence enforces. The failover drill in the repository runs every scenario in the table above against a live cluster, if you want the numbers on your own hardware rather than mine.",{"title":936,"searchDepth":937,"depth":937,"links":938},"",2,[939,940,941,942,943,944,945,946],{"id":446,"depth":937,"text":447},{"id":494,"depth":937,"text":455},{"id":514,"depth":937,"text":461},{"id":549,"depth":937,"text":467},{"id":597,"depth":937,"text":473},{"id":668,"depth":937,"text":479},{"id":833,"depth":937,"text":485},{"id":895,"depth":937,"text":491},"2026-07-02","The failover-testing post ended on an exploratory idea — rebuild the processing engine in the same process, no JVM exit — and a constraint: stay at two standbys. Both have moved. The in-place restart shipped, and a lag-aware leader election shipped on top, so a StoatFlow HA cluster now runs one active and any number of warm standbys, scaling past two elects exactly one successor instead of crash-looping, and a graceful role swap no longer bounces the pod.","md",{"src":951},"\u002Fassets\u002Fblog\u002Fog\u002Fin-place-restart-multi-standby.png",{},true,"\u002Fblog\u002Fin-place-restart-multi-standby",{"title":336,"description":948},"3.blog\u002F7.in-place-restart-multi-standby","26dnqg0ih9KEFRQutXLSZ529yBTgP08qkWz7xpaOKPc",[959,964],{"title":960,"path":961,"stem":962,"description":963,"children":-1},"Internal consistency on Kafka: emitting a correct answer at every commit","\u002Fblog\u002Finternal-consistency","3.blog\u002F8.internal-consistency","Money can only be moved, never created — so a stream that tracks balances should read total = 0 at every consistent cut. The Flink Table API gets it right 0.035% of the time; our Kafka Streams twin sends total to −1,619 … +1,792. StoatFlow holds it at exactly 0, at every one of its committed cuts. Here is why, and the measured proof.",{"title":965,"path":966,"stem":967,"description":968,"children":-1},"How we obfuscate StoatFlow — and what the public-API boundary taught us","\u002Fblog\u002Fobfuscation-and-the-public-api-boundary","3.blog\u002F6.obfuscation-and-the-public-api-boundary","A drop-in Kafka Streams DSL has to stay byte-stable while the engine beneath it stays hidden. Those two goals pull apart — and every interesting bug we hit while obfuscating StoatFlow lived on the seam between them.",1786987416428]