[{"data":1,"prerenderedAt":1179},["ShallowReactive",2],{"navigation":3,"\u002Fblog\u002Fha-failover-testing":334,"\u002Fblog\u002Fha-failover-testing-surround":1169},[4,43,91,154,181,224,262,293],{"title":5,"path":6,"stem":7,"children":8,"icon":42},"Getting Started","\u002Fdocs\u002Fgetting-started","1.docs\u002F1.getting-started\u002F1.index",[9,12,17,22,27,32,37],{"title":10,"path":6,"stem":7,"icon":11},"Getting started","i-lucide-flag",{"title":13,"path":14,"stem":15,"icon":16},"Modules overview","\u002Fdocs\u002Fgetting-started\u002Fmodules-overview","1.docs\u002F1.getting-started\u002F2.modules-overview","i-lucide-boxes",{"title":18,"path":19,"stem":20,"icon":21},"Installation","\u002Fdocs\u002Fgetting-started\u002Finstallation","1.docs\u002F1.getting-started\u002F3.installation","i-lucide-download",{"title":23,"path":24,"stem":25,"icon":26},"License configuration","\u002Fdocs\u002Fgetting-started\u002Flicense-configuration","1.docs\u002F1.getting-started\u002F4.license-configuration","i-lucide-key-round",{"title":28,"path":29,"stem":30,"icon":31},"Your first app","\u002Fdocs\u002Fgetting-started\u002Ffirst-app","1.docs\u002F1.getting-started\u002F5.first-app","i-lucide-square-play",{"title":33,"path":34,"stem":35,"icon":36},"Project setup","\u002Fdocs\u002Fgetting-started\u002Fproject-setup","1.docs\u002F1.getting-started\u002F6.project-setup","i-lucide-package",{"title":38,"path":39,"stem":40,"icon":41},"AI assistants","\u002Fdocs\u002Fgetting-started\u002Fai-assistants","1.docs\u002F1.getting-started\u002F7.ai-assistants","i-lucide-bot",false,{"title":44,"path":45,"stem":46,"children":47,"icon":50},"Core Concepts","\u002Fdocs\u002Fconcepts","1.docs\u002F2.concepts\u002F1.index",[48,51,56,61,66,71,76,81,86],{"title":49,"path":45,"stem":46,"icon":50},"Core concepts","i-lucide-book-open",{"title":52,"path":53,"stem":54,"icon":55},"Architecture","\u002Fdocs\u002Fconcepts\u002Farchitecture","1.docs\u002F2.concepts\u002F2.architecture","i-lucide-layers",{"title":57,"path":58,"stem":59,"icon":60},"Exactly-once semantics","\u002Fdocs\u002Fconcepts\u002Fexactly-once","1.docs\u002F2.concepts\u002F3.exactly-once","i-lucide-shield-check",{"title":62,"path":63,"stem":64,"icon":65},"Lanes and parallelism","\u002Fdocs\u002Fconcepts\u002Flanes-and-parallelism","1.docs\u002F2.concepts\u002F4.lanes-and-parallelism","i-lucide-split",{"title":67,"path":68,"stem":69,"icon":70},"State and thread-safety","\u002Fdocs\u002Fconcepts\u002Fstate-and-thread-safety","1.docs\u002F2.concepts\u002F5.state-and-thread-safety","i-lucide-lock",{"title":72,"path":73,"stem":74,"icon":75},"Event time and watermarks","\u002Fdocs\u002Fconcepts\u002Fevent-time-and-watermarks","1.docs\u002F2.concepts\u002F6.event-time-and-watermarks","i-lucide-clock",{"title":77,"path":78,"stem":79,"icon":80},"The configuration model","\u002Fdocs\u002Fconcepts\u002Fconfiguration-model","1.docs\u002F2.concepts\u002F7.configuration-model","i-lucide-sliders-horizontal",{"title":82,"path":83,"stem":84,"icon":85},"The error-handling model","\u002Fdocs\u002Fconcepts\u002Ferror-handling-model","1.docs\u002F2.concepts\u002F8.error-handling-model","i-lucide-triangle-alert",{"title":87,"path":88,"stem":89,"icon":90},"How StoatFlow differs from Kafka Streams","\u002Fdocs\u002Fconcepts\u002Fhow-stoatflow-differs-from-ks","1.docs\u002F2.concepts\u002F9.how-stoatflow-differs-from-ks","i-lucide-arrow-left-right",{"title":92,"path":93,"stem":94,"children":95,"icon":98},"Building Topologies","\u002Fdocs\u002Fbuilding","1.docs\u002F3.building\u002F1.index",[96,99,104,109,114,119,124,129,134,139,144,149],{"title":97,"path":93,"stem":94,"icon":98},"Building topologies","i-lucide-blocks",{"title":100,"path":101,"stem":102,"icon":103},"Error handling and DLQ","\u002Fdocs\u002Fbuilding\u002Ferror-handling-dlq","1.docs\u002F3.building\u002F10.error-handling-dlq","i-lucide-circle-x",{"title":105,"path":106,"stem":107,"icon":108},"State stores","\u002Fdocs\u002Fbuilding\u002Fstate-stores","1.docs\u002F3.building\u002F11.state-stores","i-lucide-database",{"title":110,"path":111,"stem":112,"icon":113},"Testing topologies","\u002Fdocs\u002Fbuilding\u002Ftesting","1.docs\u002F3.building\u002F12.testing","i-lucide-flask-conical",{"title":115,"path":116,"stem":117,"icon":118},"Sources and sinks","\u002Fdocs\u002Fbuilding\u002Fstreams-builder","1.docs\u002F3.building\u002F2.streams-builder","i-lucide-import",{"title":120,"path":121,"stem":122,"icon":123},"KStream and KTable operations","\u002Fdocs\u002Fbuilding\u002Fkstream-ktable","1.docs\u002F3.building\u002F3.kstream-ktable","i-lucide-waypoints",{"title":125,"path":126,"stem":127,"icon":128},"Aggregations","\u002Fdocs\u002Fbuilding\u002Faggregations","1.docs\u002F3.building\u002F4.aggregations","i-lucide-sigma",{"title":130,"path":131,"stem":132,"icon":133},"Windowing","\u002Fdocs\u002Fbuilding\u002Fwindowing","1.docs\u002F3.building\u002F5.windowing","i-lucide-calendar-clock",{"title":135,"path":136,"stem":137,"icon":138},"Joins","\u002Fdocs\u002Fbuilding\u002Fjoins","1.docs\u002F3.building\u002F6.joins","i-lucide-git-merge",{"title":140,"path":141,"stem":142,"icon":143},"The Processor API","\u002Fdocs\u002Fbuilding\u002Fprocessor-api","1.docs\u002F3.building\u002F7.processor-api","i-lucide-cpu",{"title":145,"path":146,"stem":147,"icon":148},"Scheduled sources","\u002Fdocs\u002Fbuilding\u002Fscheduled-sources","1.docs\u002F3.building\u002F8.scheduled-sources","i-lucide-timer",{"title":150,"path":151,"stem":152,"icon":153},"Serdes and Avro","\u002Fdocs\u002Fbuilding\u002Fserdes","1.docs\u002F3.building\u002F9.serdes","i-lucide-binary",{"title":155,"path":156,"stem":157,"children":158,"icon":80},"Configuration","\u002Fdocs\u002Fconfiguration","1.docs\u002F4.configuration\u002F1.index",[159,161,166,171,176],{"title":160,"path":156,"stem":157,"icon":80},"How configuration works",{"title":162,"path":163,"stem":164,"icon":165},"Engine configuration (:core)","\u002Fdocs\u002Fconfiguration\u002Fcore-config","1.docs\u002F4.configuration\u002F2.core-config","i-lucide-settings-2",{"title":167,"path":168,"stem":169,"icon":170},"Runtime configuration (:runtime)","\u002Fdocs\u002Fconfiguration\u002Fruntime-config","1.docs\u002F4.configuration\u002F3.runtime-config","i-lucide-server-cog",{"title":172,"path":173,"stem":174,"icon":175},"Defaults, adaptivity, and presets","\u002Fdocs\u002Fconfiguration\u002Fdefaults-and-presets","1.docs\u002F4.configuration\u002F4.defaults-and-presets","i-lucide-gauge",{"title":177,"path":178,"stem":179,"icon":180},"Kafka client configuration","\u002Fdocs\u002Fconfiguration\u002Fkafka-client-config","1.docs\u002F4.configuration\u002F5.kafka-client-config","i-lucide-plug",{"title":182,"path":183,"stem":184,"children":185,"icon":188},"Running in Production","\u002Fdocs\u002Fruntime","1.docs\u002F5.runtime\u002F1.index",[186,189,194,199,204,209,214,219],{"title":187,"path":183,"stem":184,"icon":188},"The runtime","i-lucide-server",{"title":190,"path":191,"stem":192,"icon":193},"The REST API","\u002Fdocs\u002Fruntime\u002Frest-api","1.docs\u002F5.runtime\u002F2.rest-api","i-lucide-globe",{"title":195,"path":196,"stem":197,"icon":198},"Health checks","\u002Fdocs\u002Fruntime\u002Fhealth-checks","1.docs\u002F5.runtime\u002F3.health-checks","i-lucide-heart-pulse",{"title":200,"path":201,"stem":202,"icon":203},"Metrics","\u002Fdocs\u002Fruntime\u002Fmetrics","1.docs\u002F5.runtime\u002F4.metrics","i-lucide-activity",{"title":205,"path":206,"stem":207,"icon":208},"Pause and resume","\u002Fdocs\u002Fruntime\u002Fpause-unpause","1.docs\u002F5.runtime\u002F5.pause-unpause","i-lucide-pause",{"title":210,"path":211,"stem":212,"icon":213},"Plugins and lifecycle hooks","\u002Fdocs\u002Fruntime\u002Fplugins","1.docs\u002F5.runtime\u002F6.plugins","i-lucide-puzzle",{"title":215,"path":216,"stem":217,"icon":218},"Docker images","\u002Fdocs\u002Fruntime\u002Fdocker","1.docs\u002F5.runtime\u002F7.docker","i-lucide-container",{"title":220,"path":221,"stem":222,"icon":223},"GraalVM native image","\u002Fdocs\u002Fruntime\u002Fnative-image","1.docs\u002F5.runtime\u002F8.native-image","i-lucide-zap",{"title":225,"path":226,"stem":227,"children":228,"icon":231},"Deploying & Operating","\u002Fdocs\u002Foperating","1.docs\u002F6.operating\u002F1.index",[229,232,237,242,247,252,257],{"title":230,"path":226,"stem":227,"icon":231},"Deploying and operating","i-lucide-life-buoy",{"title":233,"path":234,"stem":235,"icon":236},"Running on Kubernetes","\u002Fdocs\u002Foperating\u002Fkubernetes","1.docs\u002F6.operating\u002F2.kubernetes","i-lucide-ship",{"title":238,"path":239,"stem":240,"icon":241},"High availability","\u002Fdocs\u002Foperating\u002Fhigh-availability","1.docs\u002F6.operating\u002F3.high-availability","i-lucide-copy",{"title":243,"path":244,"stem":245,"icon":246},"Liveness and readiness probes","\u002Fdocs\u002Foperating\u002Fprobes","1.docs\u002F6.operating\u002F4.probes","i-lucide-stethoscope",{"title":248,"path":249,"stem":250,"icon":251},"Observability","\u002Fdocs\u002Foperating\u002Fobservability","1.docs\u002F6.operating\u002F5.observability","i-lucide-telescope",{"title":253,"path":254,"stem":255,"icon":256},"Tuning under load","\u002Fdocs\u002Foperating\u002Ftuning","1.docs\u002F6.operating\u002F6.tuning","i-lucide-sliders",{"title":258,"path":259,"stem":260,"icon":261},"Production checklist","\u002Fdocs\u002Foperating\u002Fproduction-checklist","1.docs\u002F6.operating\u002F7.production-checklist","i-lucide-clipboard-check",{"title":263,"path":264,"stem":265,"children":266,"icon":90},"Migrating from Kafka Streams","\u002Fdocs\u002Fmigration","1.docs\u002F7.migration\u002F1.index",[267,268,273,278,283,288],{"title":263,"path":264,"stem":265,"icon":90},{"title":269,"path":270,"stem":271,"icon":272},"Automated port","\u002Fdocs\u002Fmigration\u002Fautomated-port","1.docs\u002F7.migration\u002F2.automated-port","i-lucide-wand-sparkles",{"title":274,"path":275,"stem":276,"icon":277},"Migration without carrying state","\u002Fdocs\u002Fmigration\u002Fwithout-data-migration","1.docs\u002F7.migration\u002F3.without-data-migration","i-lucide-sparkles",{"title":279,"path":280,"stem":281,"icon":282},"Migration carrying state","\u002Fdocs\u002Fmigration\u002Fwith-data-migration","1.docs\u002F7.migration\u002F4.with-data-migration","i-lucide-database-backup",{"title":284,"path":285,"stem":286,"icon":287},"The migration tool","\u002Fdocs\u002Fmigration\u002Fmigration-tool","1.docs\u002F7.migration\u002F5.migration-tool","i-lucide-truck",{"title":289,"path":290,"stem":291,"icon":292},"Reusing your Kafka Streams dashboards","\u002Fdocs\u002Fmigration\u002Freusing-kafka-streams-dashboards","1.docs\u002F7.migration\u002F6.reusing-kafka-streams-dashboards","i-lucide-line-chart",{"title":294,"path":295,"stem":296,"children":297,"icon":299},"Reference","\u002Fdocs\u002Freference","1.docs\u002F8.reference\u002F1.index",[298,300,305,310,315,320,325,329],{"title":294,"path":295,"stem":296,"icon":299},"i-lucide-list",{"title":301,"path":302,"stem":303,"icon":304},"Configuration reference","\u002Fdocs\u002Freference\u002Fconfiguration-reference","1.docs\u002F8.reference\u002F2.configuration-reference","i-lucide-table",{"title":306,"path":307,"stem":308,"icon":309},"REST API reference","\u002Fdocs\u002Freference\u002Frest-api-reference","1.docs\u002F8.reference\u002F3.rest-api-reference","i-lucide-network",{"title":311,"path":312,"stem":313,"icon":314},"Gradle plugin reference","\u002Fdocs\u002Freference\u002Fgradle-plugin-reference","1.docs\u002F8.reference\u002F4.gradle-plugin-reference","i-lucide-box",{"title":316,"path":317,"stem":318,"icon":319},"Maven reference","\u002Fdocs\u002Freference\u002Fmaven-reference","1.docs\u002F8.reference\u002F5.maven-reference","i-simple-icons-apachemaven",{"title":321,"path":322,"stem":323,"icon":324},"Kafka Streams compatibility matrix","\u002Fdocs\u002Freference\u002Fks-compatibility-matrix","1.docs\u002F8.reference\u002F6.ks-compatibility-matrix","i-lucide-table-2",{"title":326,"path":327,"stem":328,"icon":175},"Metrics reference","\u002Fdocs\u002Freference\u002Fmetrics-reference","1.docs\u002F8.reference\u002F7.metrics-reference",{"title":330,"path":331,"stem":332,"icon":333},"Glossary","\u002Fdocs\u002Freference\u002Fglossary","1.docs\u002F8.reference\u002F8.glossary","i-lucide-book-a",{"id":335,"title":336,"authors":337,"badge":343,"body":345,"date":1158,"description":1159,"draft":42,"extension":1160,"image":1161,"meta":1163,"navigation":1164,"path":1165,"seo":1166,"stem":1167,"__hash__":1168},"posts\u002F3.blog\u002F5.ha-failover-testing.md","Measuring StoatFlow failover: four scenarios, from the logs",[338],{"name":339,"to":340,"avatar":341},"Hartmut Armbruster","https:\u002F\u002Fwww.linkedin.com\u002Fin\u002Fhartmut-co-uk\u002F",{"src":342},"\u002Fassets\u002Fhartmut_armbruster_monochromatic.jpg",{"label":344},"Engineering",{"type":346,"value":347,"toc":1148},"minimark",[348,363,370,446,451,495,498,528,539,618,645,652,655,658,699,702,705,720,723,858,876,891,894,897,919,935,946,956,962,972,975,1014,1017,1020,1076,1086,1089,1102,1113,1131,1137,1144],[349,350,351,352,357,358,362],"p",{},"The ",[353,354,356],"a",{"href":355},"\u002Fblog\u002Fhot-standby-high-availability","hot-standby"," post made a claim: failover in seconds, independent of state size. A claim like that earns the right to be measured. So I put a live active\u002Fpassive pair under load and triggered every way a StoatFlow active can lose its role — and the first thing the test surfaced is that the obvious way to test failover does not test failover at all. ",[359,360,361],"code",{},"kubectl delete pod"," is a graceful SIGTERM handoff, not a crash.",[349,364,365,366,369],{},"This is the measured failover story: four distinct scenarios, the millisecond timing taken from pod logs — because Prometheus cannot resolve a one-second event — and the things the numbers contradicted, including a single crash that recovers in place without ever failing over. Measured on StoatFlow ",[367,368],"stoatflow-version",{},", an active\u002Fpassive word-count pair on Kubernetes at ~1,000 input records\u002Fs (≈ 44K output records\u002Fs), exactly-once.",[371,372,373,379],"blockquote",{},[349,374,375],{},[376,377,378],"strong",{},"TL;DR",[380,381,382,396,402,408,414,440],"ul",{},[383,384,385,390,391,395],"li",{},[376,386,387,389],{},[359,388,361],{}," is not a crash."," Kubernetes sends SIGTERM first; the JVM shutdown hook drains and hands off gracefully. A real crash needs a SIGKILL of the JVM's host PID ",[392,393,394],"em",{},"from the node",".",[383,397,398,401],{},[376,399,400],{},"Four scenarios, not one:"," graceful switch, SIGTERM\u002Frolling deploy, JVM crash with in-place restart, and node loss with a standby takeover. They behave — and time — very differently.",[383,403,404,407],{},[376,405,406],{},"Graceful is sub-second and lossless:"," ~0.4–0.9 s of no-processing, the in-flight epoch committed before handover, zero reprocessing under exactly-once.",[383,409,410,413],{},[376,411,412],{},"A single JVM crash recovers in place (~3.6 s) and never fails over"," — the container restarts faster than the standby's detection window. The standby takes over only when the active is genuinely gone (~8.5 s, detection-dominated).",[383,415,416,423,424,427,428,431,432,435,436,395],{},[376,417,418,419,422],{},"That is the ",[392,420,421],{},"hard-crash"," tier specifically."," A fatal the engine catches behaves differently again: a ",[392,425,426],{},"restartable"," one (a processing or production failure, with a ",[359,429,430],{},"REPLACE_THREAD"," handler) is absorbed in place and never fails over either; anything else drains and hands off, so it ",[392,433,434],{},"does"," — see ",[353,437,439],{"href":438},"\u002Fdocs\u002Foperating\u002Fhigh-availability#application-faults-under-hot-standby","application faults",[383,441,442,445],{},[376,443,444],{},"Grafana shows the shape; logs show the timing."," A 10-second scrape and a one-minute rate window cannot measure a one-second failover.",[447,448,450],"h2",{"id":449},"in-this-post","In this post",[380,452,453,459,465,471,477,483,489],{},[383,454,455],{},[353,456,458],{"href":457},"#the-obvious-test-isnt-a-crash","The obvious test isn't a crash",[383,460,461],{},[353,462,464],{"href":463},"#four-ways-an-active-loses-its-role","Four ways an active loses its role",[383,466,467],{},[353,468,470],{"href":469},"#expected-versus-measured","Expected versus measured",[383,472,473],{},[353,474,476],{"href":475},"#what-the-numbers-revealed","What the numbers revealed",[383,478,479],{},[353,480,482],{"href":481},"#tradeoffs-and-limits","Tradeoffs and limits",[383,484,485],{},[353,486,488],{"href":487},"#reproduce-it","Reproduce it",[383,490,491],{},[353,492,494],{"href":493},"#whats-coming-next","What's coming next",[447,496,458],{"id":497},"the-obvious-test-isnt-a-crash",[349,499,500,501,503,504,507,508,511,512,515,516,519,520,523,524,527],{},"The instinct, testing failover on Kubernetes, is ",[359,502,361],{},". It feels like pulling the plug. It isn't. Kubernetes deletes a pod gracefully: it sends the container ",[376,505,506],{},"SIGTERM",", waits out the termination grace period, and only then SIGKILLs. StoatFlow installs a JVM shutdown hook on SIGTERM, so a ",[359,509,510],{},"kubectl delete"," runs the ",[392,513,514],{},"graceful"," path — the active drains its in-flight epoch, commits it, publishes a ",[359,517,518],{},"DRAINING"," marker, and the standby promotes on that marker. The logs of the \"crashed\" pod give it away: ",[359,521,522],{},"Graceful shutdown completed in 21ms",", then ",[359,525,526],{},"Stopped … in 99 ms",". That is a clean handoff, not a crash.",[349,529,530,531,534,535,538],{},"To actually crash the process you have to SIGKILL the JVM — and not from inside the container. The JVM runs as PID 1 there, and the kernel protects a PID-namespace init process: an in-container ",[359,532,533],{},"kill -9 1"," is silently dropped. The kill has to come from the node, against the JVM's ",[392,536,537],{},"host"," PID:",[540,541,546],"pre",{"className":542,"code":543,"language":544,"meta":545,"style":545},"language-bash shiki shiki-themes vitesse-light","# on the node hosting the active pod\ncrictl inspect --output go-template --template '{{.info.pid}}' \"$CONTAINER_ID\"  # -> host PID\nkill -9 \"$HOST_PID\"\n","bash","",[359,547,548,557,600],{"__ignoreMap":545},[549,550,553],"span",{"class":551,"line":552},"line",1,[549,554,556],{"class":555},"s8zF2","# on the node hosting the active pod\n",[549,558,560,564,568,572,575,578,582,585,588,591,594,597],{"class":551,"line":559},2,[549,561,563],{"class":562},"sySUi","crictl",[549,565,567],{"class":566},"spphp"," inspect",[549,569,571],{"class":570},"sEi1f"," --output",[549,573,574],{"class":566}," go-template",[549,576,577],{"class":570}," --template",[549,579,581],{"class":580},"sSP4y"," '",[549,583,584],{"class":566},"{{.info.pid}}",[549,586,587],{"class":580},"'",[549,589,590],{"class":580}," \"",[549,592,593],{"class":566},"$CONTAINER_ID",[549,595,596],{"class":580},"\"",[549,598,599],{"class":555},"  # -> host PID\n",[549,601,603,607,610,612,615],{"class":551,"line":602},3,[549,604,606],{"class":605},"su6XF","kill",[549,608,609],{"class":570}," -9",[549,611,590],{"class":580},[549,613,614],{"class":566},"$HOST_PID",[549,616,617],{"class":580},"\"\n",[349,619,620,621,624,625,628,629,632,633,636,637,640,641,644],{},"The second trap is measurement. The temptation is to read failover time off the Grafana dashboard. Don't. The benchmark ServiceMonitor scrapes every 10 seconds, so the ",[359,622,623],{},"stoatflow_ha_role"," gauge only flips at a 10-second boundary — a ±10-second quantisation on a one-second event. Throughput is a ",[359,626,627],{},"rate(…[1m])"," that smooths a sub-second dip into nothing, and the end-to-end latency panel is a sticky high-water gauge that shows the blip's ",[392,630,631],{},"size"," but not ",[392,634,635],{},"when",". Grafana is the right tool for the ",[392,638,639],{},"shape"," of a failover — the role line flipping, the brief throughput dip, the latency spike. For the ",[392,642,643],{},"numbers",", the source of truth is the pod logs: millisecond timestamps on a single NTP-synced cluster clock.",[349,646,647],{},[648,649],"img",{"alt":650,"src":651},"StoatFlow HA role panel across a \u002Fha\u002Fswitch: the active role hands from one pod to the other in a single step — exactly one pod at role = 1 at any moment, never two. The flip lands between two 10-second scrapes, which is why the dashboard can show the handover but not time it.","\u002Fassets\u002Fblog\u002Fha-failover-role-flip.png",[447,653,464],{"id":654},"four-ways-an-active-loses-its-role",[349,656,657],{},"There are four, and conflating them is how you get a wrong number.",[659,660,661,674,683,693],"ol",{},[383,662,663,666,667,670,671,673],{},[376,664,665],{},"Graceful switch"," — an operator ",[359,668,669],{},"POST \u002Fha\u002Fswitch",". The active drains, commits, drops the producer fence, and hands the role to the standby on the ",[359,672,518],{}," signal.",[383,675,676,679,680,682],{},[376,677,678],{},"SIGTERM \u002F pod-delete \u002F rolling deploy"," — the same graceful handoff, but triggered by the JVM shutdown hook instead of an operator command. This is what ",[359,681,361],{}," and a rolling deploy do.",[383,684,685,688,689,692],{},[376,686,687],{},"JVM crash, in-place restart"," — a true SIGKILL of the JVM. Kubernetes restarts the ",[392,690,691],{},"same"," container, so its node-local state is still there: the store delta-restores the aborted epoch rather than rebuilding from the changelog, and the pod re-promotes itself. A warm restart, not a cold start. The standby is never involved.",[383,694,695,698],{},[376,696,697],{},"Node loss, standby failover"," — the active is genuinely gone and cannot come back. The standby stops seeing its heartbeats and promotes itself via staleness detection.",[349,700,701],{},"Scenarios 1 and 2 are graceful: the active drops the fence deliberately and the standby promotes on an explicit signal. Scenarios 3 and 4 are ungraceful: the active dies without warning. The distinction between 3 and 4 is the one that surprised me, and it gets its own section below.",[447,703,470],{"id":704},"expected-versus-measured",[349,706,707,708,711,712,715,716,719],{},"The design — ",[353,709,710],{"href":239},"ADR-125"," — sets the expectations. Promotion is ",[392,713,714],{},"fence-and-assign in parallel, then resume against already-warm state",", which the design measured at ",[376,717,718],{},"~383 ms, independent of state size",". Ungraceful failover adds a detection cost on top: the standby has to notice the active is gone before it can promote.",[349,721,722],{},"Here is what a live pair under load actually did, every figure taken from the logs.",[724,725,726,751],"table",{},[727,728,729],"thead",{},[730,731,732,736,739,742,745,748],"tr",{},[733,734,735],"th",{},"Scenario",[733,737,738],{},"Trigger",[733,740,741],{},"Detection",[733,743,744],{},"Promotion critical path",[733,746,747],{},"Stop-the-world",[733,749,750],{},"Reprocessing",[752,753,754,782,811,835],"tbody",{},[730,755,756,761,765,771,774,779],{},[757,758,759],"td",{},[376,760,665],{},[757,762,763],{},[359,764,669],{},[757,766,767,768,770],{},"immediate (",[359,769,518],{},")",[757,772,773],{},"~0.5–0.8 s",[757,775,776],{},[376,777,778],{},"~0.4–0.9 s",[757,780,781],{},"none",[730,783,784,789,797,801,804,809],{},[757,785,786],{},[376,787,788],{},"SIGTERM \u002F rolling deploy",[757,790,791,793,794],{},[359,792,510],{}," \u002F ",[359,795,796],{},"rollout restart",[757,798,767,799,770],{},[359,800,518],{},[757,802,803],{},"~0.35–0.4 s",[757,805,806],{},[376,807,808],{},"~0.4 s",[757,810,781],{},[730,812,813,818,821,824,827,832],{},[757,814,815],{},[376,816,817],{},"JVM crash, in-place",[757,819,820],{},"SIGKILL the JVM",[757,822,823],{},"— (same pod recovers)",[757,825,826],{},"n\u002Fa (warm restart, self-promotes)",[757,828,829],{},[376,830,831],{},"~3.6 s",[757,833,834],{},"one epoch",[730,836,837,842,845,848,851,856],{},[757,838,839],{},[376,840,841],{},"Node loss, failover",[757,843,844],{},"node gone",[757,846,847],{},"~7 s (staleness)",[757,849,850],{},"~0.8 s",[757,852,853],{},[376,854,855],{},"~8.5 s",[757,857,834],{},[349,859,860,861,864,865,868,869,871,872,875],{},"Two definitions hold the table together. The ",[376,862,863],{},"promotion critical path"," is the work to bring a warm standby to ",[392,866,867],{},"serving"," once it has decided to promote — fence-and-assign, catch the last changelog records, start the engine threads. Measured at 350–820 ms across runs, it brackets the design's ~383 ms; the spread is the changelog catch-up and thread start under load. ",[376,870,747],{}," is the window in which ",[392,873,874],{},"neither"," instance is processing — bounded by the old active stopping and the new active resuming. That is the number that matters for availability, and it is much smaller than it looks from the dashboard, for a reason worth its own paragraph below.",[349,877,878,879,882,883,886,887,890],{},"The graceful paths (switch, SIGTERM) confirm the design cleanly: sub-second, and the in-flight epoch is ",[392,880,881],{},"committed before"," the handover — ",[359,884,885],{},"Phase 2: Final barrier N committed successfully",", then a clean ",[359,888,889],{},"Phase 4b: Commit thread stopped",", in about 21 ms — so the new active resumes from a committed boundary with nothing to reprocess. The ungraceful node-loss path matches the design's other published figure: detection dominates, at roughly seven seconds, and promotion is the fast part.",[447,892,476],{"id":893},"what-the-numbers-revealed",[349,895,896],{},"Three things the measurement contradicted or sharpened.",[349,898,899,902,903,906,907,910,911,914,915,918],{},[376,900,901],{},"A single JVM crash recovers in place — it does not fail over."," Kill the active's JVM and the standby does ",[392,904,905],{},"nothing",". Kubernetes restarts the crashed container in ~1.4 s; StoatFlow restores its node-local state — a warm restart that delta-restores only the aborted epoch's records, not a rebuild from the changelog — and re-promotes itself in ~2.2 s, back to serving in ",[376,908,909],{},"~3.6 s total",". That is well inside the standby's ~7-second detection window, so by the time the standby could have concluded the active was gone, the active is already back. The warm spare earns its keep on ",[392,912,913],{},"node"," loss, not on a process crash that the orchestrator can paper over faster than the failure detector can fire. The cost is one reprocessed epoch — the in-flight transaction was aborted, never committed, never visible to ",[359,916,917],{},"read_committed"," consumers — not a duplicate, not a loss.",[349,920,921,922,925,926,928,929,931,932,934],{},"Read \"crash\" literally, though: this is a SIGKILL, where the JVM dies without running anything. A fatal the engine ",[392,923,924],{},"catches"," takes a different path, and which one depends on whether a rebuilt engine could survive it. A ",[376,927,426],{}," fault — a processing or production failure, with a ",[359,930,430],{}," handler registered — is absorbed in place: the engine is rebuilt inside the live JVM and the pod keeps the active role, so there is no failover at all until the restart budget is spent. Anything else drains, publishes ",[359,933,518],{}," on its way out, and the standby promotes off that signal via the same fast path as a graceful switch — for that case the table's graceful rows are the ones to read, not this one.",[349,936,937,938,941,942,945],{},"There is a sharp edge here: this is the ",[392,939,940],{},"first-crash"," figure. A pod that keeps crashing trips Kubernetes' ",[376,943,944],{},"CrashLoopBackOff",", an exponential restart delay that grows to tens of seconds. The StoatFlow recovery stays ~2 s; the rest is the kubelet holding the container down. If you benchmark crash recovery, reset the restart count first — or you are measuring the kubelet's patience, not the engine's.",[349,947,948,951,952,955],{},[376,949,950],{},"Stop-the-world is not the latency blip."," The dashboard's end-to-end latency spikes to ~2 s on a graceful switch; the measured stop-the-world is ~0.9 s. Both are correct — they measure different things. Stop-the-world is the control-plane gap: no engine running. The latency blip is longer because, the moment processing resumes, the new active has to drain the input that piled up ",[392,953,954],{},"during"," the gap — at 44K records\u002Fs, even a sub-second pause leaves a backlog to chew through. The felt recovery is stop-the-world plus the catch-up; the availability gap is stop-the-world alone. Reporting one as the other overstates the outage by 2–3×. The blip is largest on a node-loss failover, where the gap itself is seconds — the panel below catches one: p95 and p99 climb to ~8 s while the standby takes over, then snap back.",[349,957,958],{},[648,959],{"alt":960,"src":961},"End-to-end latency during a node-loss failover (p50 \u002F p95 \u002F p99 \u002F max): p95 and p99 jump to ~8 s the moment the active goes silent and stay there until the standby promotes and works through the backlog, then return to baseline. The max line is a sticky high-water gauge, so it stays pinned for a while after the event.","\u002Fassets\u002Fblog\u002Fha-failover-latency-blip.png",[349,963,964,967,968,971],{},[376,965,966],{},"Detection is the whole cost of an ungraceful failover."," A clean node-loss measurement — the active cordoned off, SIGKILLed, and force-deleted so it cannot restart — promotes the standby in ~8.5 s of stop-the-world. Of that, ~7 s is detection (",[359,969,970],{},"staleness-threshold-ms"," plus the missed-heartbeat debounce) and only ~0.8 s is the promotion itself. If your recovery-time objective is tight, the lever is the staleness window, not the promotion path — the promotion is already as fast as a warm standby allows.",[447,973,482],{"id":974},"tradeoffs-and-limits",[380,976,977,983,989,999,1005],{},[383,978,979,982],{},[376,980,981],{},"Ungraceful failover is detection-dominated."," The ~7-second window is a deliberate default that trades failover speed for tolerance of transient network jitter. Tightening it speeds up node-loss failover and raises the risk of a spurious promotion on a blip. It is a knob; set it against your recovery-time objective, not by instinct.",[383,984,985,988],{},[376,986,987],{},"A process crash is not a failover."," In-place restart (~3.6 s) is the common case for a crash, and it is faster than the standby would be. The standby is insurance against losing the node, not against the JVM dying.",[383,990,991,994,995,998],{},[376,992,993],{},"Honest crash testing needs node access."," You cannot crash a StoatFlow active faithfully with ",[359,996,997],{},"kubectl"," alone — a real SIGKILL comes from the node. If your test harness can only reach the API server, it is testing the graceful path and calling it a crash.",[383,1000,1001,1004],{},[376,1002,1003],{},"Two is the supported topology."," The pair is one active and one standby. Scaling to three and back leaves a stale peer record in the coordination topic that pushes the survivor onto an unsupported multi-standby code path, where it can decline to promote. Clean it up by recreating the coordination topic, and stay at two.",[383,1006,1007,1010,1011,1013],{},[376,1008,1009],{},"Exactly-once holds across every path."," Graceful commits the in-flight epoch; crash aborts it and reprocesses one epoch. Either way the broker-enforced transactional fence guarantees at most one instance ever commits — no duplicates, no lost committed work, for ",[359,1012,917],{}," consumers.",[447,1015,488],{"id":1016},"reproduce-it",[349,1018,1019],{},"The four scenarios are automated in a single drill that prints the log-based timing — promotion critical path and stop-the-world — rather than the inflated wall-clock:",[540,1021,1023],{"className":542,"code":1022,"language":544,"meta":545,"style":545},"ha-failover-drill.sh --mode switch     # graceful handoff\nha-failover-drill.sh --mode sigterm    # SIGTERM \u002F pod-delete (graceful, not a crash)\nha-failover-drill.sh --mode crash      # true SIGKILL of the JVM -> in-place restart\nha-failover-drill.sh --mode nodeloss   # cordon + SIGKILL + force-delete -> standby failover\n",[359,1024,1025,1039,1051,1063],{"__ignoreMap":545},[549,1026,1027,1030,1033,1036],{"class":551,"line":552},[549,1028,1029],{"class":562},"ha-failover-drill.sh",[549,1031,1032],{"class":570}," --mode",[549,1034,1035],{"class":566}," switch",[549,1037,1038],{"class":555},"     # graceful handoff\n",[549,1040,1041,1043,1045,1048],{"class":551,"line":559},[549,1042,1029],{"class":562},[549,1044,1032],{"class":570},[549,1046,1047],{"class":566}," sigterm",[549,1049,1050],{"class":555},"    # SIGTERM \u002F pod-delete (graceful, not a crash)\n",[549,1052,1053,1055,1057,1060],{"class":551,"line":602},[549,1054,1029],{"class":562},[549,1056,1032],{"class":570},[549,1058,1059],{"class":566}," crash",[549,1061,1062],{"class":555},"      # true SIGKILL of the JVM -> in-place restart\n",[549,1064,1066,1068,1070,1073],{"class":551,"line":1065},4,[549,1067,1029],{"class":562},[549,1069,1032],{"class":570},[549,1071,1072],{"class":566}," nodeloss",[549,1074,1075],{"class":555},"   # cordon + SIGKILL + force-delete -> standby failover\n",[349,1077,351,1078,1081,1082,1085],{},[359,1079,1080],{},"crash"," and ",[359,1083,1084],{},"nodeloss"," modes SIGKILL the JVM's host PID from the node over SSH, resolve timing from the logs, and — for node loss — cordon the node so the pod genuinely cannot come back, then un-cordon afterwards. The full walkthrough, including the dashboard to watch and the gotchas, is in the playground runbook in the repository.",[447,1087,494],{"id":1088},"whats-coming-next",[371,1090,1091],{},[349,1092,1093,1096,1097,1101],{},[376,1094,1095],{},"Update (2026-07-02):"," this shipped. The in-place engine restart is no longer exploratory, and the round-trip described below is no longer always a process restart — an active that loses its role can now rebuild its engine inside the live JVM. The same work lifted the two-standby constraint. See ",[353,1098,1100],{"href":1099},"\u002Fblog\u002Fin-place-restart-multi-standby","In-place engine restart: the primitive behind multi-standby HA",". The section is kept as written for the record.",[349,1103,1104,1105,1108,1109],{},"Across all four scenarios, one round-trip recurs: an active that loses its role ",[392,1106,1107],{},"exits the process and lets Kubernetes restart it",". A graceful demote self-terminates and comes back as a fresh pod; a crash is a container restart; even an in-place recovery is the orchestrator rebuilding the whole container. ",[1110,1111,1112],"del",{},"The engine teardown and re-initialisation happen, but always wrapped in a process restart.",[349,1114,1115,1116,1119,1120,1123,1124,1127,1128,1130],{},"The work in progress — an ",[376,1117,1118],{},"in-place engine restart"," — removes that wrapper. The idea is to tear down and rebuild the processing engine ",[392,1121,1122],{},"in the same process",", without exiting the JVM or recreating the application instance, and resume from the last committed transactional offsets: an in-process crash recovery, exactly-once intact via the same ",[359,1125,1126],{},"transactional.id"," epoch-bump fence that protects a cross-pod handoff today. It is the shared primitive behind three otherwise-separate needs — the faithful analogue of Kafka Streams' ",[359,1129,430],{}," (which StoatFlow currently degrades to a full application shutdown), the hot-standby promotion path, and dynamic lane re-tuning. Build it once, cleanly, and all three get faster.",[349,1132,1133,1136],{},[1110,1134,1135],{},"It is exploratory — a draft on the roadmap, pending a feasibility spike and an ownership decision, not a committed release."," But it is the natural next step the measurements point at: the promotion path is already fast; the remaining cost on a crash is the process round-trip, and that is the round-trip this would remove.",[349,1138,1139,1140,1143],{},"For the shipped behaviour, the ",[353,1141,1142],{"href":239},"High availability guide"," has the configuration reference and the operator endpoints. The full measurement data — every raw timestamp, the stop-the-world boundaries, and the corrections this testing forced — lives in the repository's failover-timing report. The model holds: one active, a warm passive spare, the fence as the backstop, and now numbers behind the word \"seconds\".",[1145,1146,1147],"style",{},"html pre.shiki code .s8zF2, html code.shiki .s8zF2{--shiki-default:#A0ADA0}html pre.shiki code .sySUi, html code.shiki .sySUi{--shiki-default:#59873A}html pre.shiki code .spphp, html code.shiki .spphp{--shiki-default:#B56959}html pre.shiki code .sEi1f, html code.shiki .sEi1f{--shiki-default:#A65E2B}html pre.shiki code .sSP4y, html code.shiki .sSP4y{--shiki-default:#B5695977}html pre.shiki code .su6XF, html code.shiki .su6XF{--shiki-default:#998418}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}",{"title":545,"searchDepth":559,"depth":559,"links":1149},[1150,1151,1152,1153,1154,1155,1156,1157],{"id":449,"depth":559,"text":450},{"id":497,"depth":559,"text":458},{"id":654,"depth":559,"text":464},{"id":704,"depth":559,"text":470},{"id":893,"depth":559,"text":476},{"id":974,"depth":559,"text":482},{"id":1016,"depth":559,"text":488},{"id":1088,"depth":559,"text":494},"2026-06-27","The obvious way to test HA failover — kubectl delete pod — is a graceful SIGTERM handoff, not a crash. Here are the four real failover scenarios, the millisecond timing measured from pod logs (because a 10-second Prometheus scrape cannot resolve a one-second event), and what the numbers revealed: a single JVM crash recovers in place without failing over, and stop-the-world is not the latency blip.","md",{"src":1162},"\u002Fassets\u002Fblog\u002Fog\u002Fha-failover-testing.png",{},true,"\u002Fblog\u002Fha-failover-testing",{"title":336,"description":1159},"3.blog\u002F5.ha-failover-testing","4KDTxDXSjCqPJsLNB0psx0ZSuSKGOIy2MIWw3OFmvVo",[1170,1175],{"title":1171,"path":1172,"stem":1173,"description":1174,"children":-1},"How we obfuscate StoatFlow — and what the public-API boundary taught us","\u002Fblog\u002Fobfuscation-and-the-public-api-boundary","3.blog\u002F6.obfuscation-and-the-public-api-boundary","A drop-in Kafka Streams DSL has to stay byte-stable while the engine beneath it stays hidden. Those two goals pull apart — and every interesting bug we hit while obfuscating StoatFlow lived on the seam between them.",{"title":1176,"path":355,"stem":1177,"description":1178,"children":-1},"Hot standby for StoatFlow: failover in seconds, not a cold restore","3.blog\u002F4.hot-standby-high-availability","StoatFlow recovery has always been fast restart — but restart time scales with state. Hot standby is an opt-in active\u002Fpassive pair that fails over in seconds regardless of state size, with exactly-once preserved across the handoff. Redundancy, not scale-out.",1786987416573]