[{"data":1,"prerenderedAt":1273},["ShallowReactive",2],{"navigation":3,"\u002Fblog\u002Ferror-handling-three-gates-one-transaction":334,"\u002Fblog\u002Ferror-handling-three-gates-one-transaction-surround":1262},[4,43,91,154,181,224,262,293],{"title":5,"path":6,"stem":7,"children":8,"icon":42},"Getting Started","\u002Fdocs\u002Fgetting-started","1.docs\u002F1.getting-started\u002F1.index",[9,12,17,22,27,32,37],{"title":10,"path":6,"stem":7,"icon":11},"Getting started","i-lucide-flag",{"title":13,"path":14,"stem":15,"icon":16},"Modules overview","\u002Fdocs\u002Fgetting-started\u002Fmodules-overview","1.docs\u002F1.getting-started\u002F2.modules-overview","i-lucide-boxes",{"title":18,"path":19,"stem":20,"icon":21},"Installation","\u002Fdocs\u002Fgetting-started\u002Finstallation","1.docs\u002F1.getting-started\u002F3.installation","i-lucide-download",{"title":23,"path":24,"stem":25,"icon":26},"License configuration","\u002Fdocs\u002Fgetting-started\u002Flicense-configuration","1.docs\u002F1.getting-started\u002F4.license-configuration","i-lucide-key-round",{"title":28,"path":29,"stem":30,"icon":31},"Your first app","\u002Fdocs\u002Fgetting-started\u002Ffirst-app","1.docs\u002F1.getting-started\u002F5.first-app","i-lucide-square-play",{"title":33,"path":34,"stem":35,"icon":36},"Project setup","\u002Fdocs\u002Fgetting-started\u002Fproject-setup","1.docs\u002F1.getting-started\u002F6.project-setup","i-lucide-package",{"title":38,"path":39,"stem":40,"icon":41},"AI assistants","\u002Fdocs\u002Fgetting-started\u002Fai-assistants","1.docs\u002F1.getting-started\u002F7.ai-assistants","i-lucide-bot",false,{"title":44,"path":45,"stem":46,"children":47,"icon":50},"Core Concepts","\u002Fdocs\u002Fconcepts","1.docs\u002F2.concepts\u002F1.index",[48,51,56,61,66,71,76,81,86],{"title":49,"path":45,"stem":46,"icon":50},"Core concepts","i-lucide-book-open",{"title":52,"path":53,"stem":54,"icon":55},"Architecture","\u002Fdocs\u002Fconcepts\u002Farchitecture","1.docs\u002F2.concepts\u002F2.architecture","i-lucide-layers",{"title":57,"path":58,"stem":59,"icon":60},"Exactly-once semantics","\u002Fdocs\u002Fconcepts\u002Fexactly-once","1.docs\u002F2.concepts\u002F3.exactly-once","i-lucide-shield-check",{"title":62,"path":63,"stem":64,"icon":65},"Lanes and parallelism","\u002Fdocs\u002Fconcepts\u002Flanes-and-parallelism","1.docs\u002F2.concepts\u002F4.lanes-and-parallelism","i-lucide-split",{"title":67,"path":68,"stem":69,"icon":70},"State and thread-safety","\u002Fdocs\u002Fconcepts\u002Fstate-and-thread-safety","1.docs\u002F2.concepts\u002F5.state-and-thread-safety","i-lucide-lock",{"title":72,"path":73,"stem":74,"icon":75},"Event time and watermarks","\u002Fdocs\u002Fconcepts\u002Fevent-time-and-watermarks","1.docs\u002F2.concepts\u002F6.event-time-and-watermarks","i-lucide-clock",{"title":77,"path":78,"stem":79,"icon":80},"The configuration model","\u002Fdocs\u002Fconcepts\u002Fconfiguration-model","1.docs\u002F2.concepts\u002F7.configuration-model","i-lucide-sliders-horizontal",{"title":82,"path":83,"stem":84,"icon":85},"The error-handling model","\u002Fdocs\u002Fconcepts\u002Ferror-handling-model","1.docs\u002F2.concepts\u002F8.error-handling-model","i-lucide-triangle-alert",{"title":87,"path":88,"stem":89,"icon":90},"How StoatFlow differs from Kafka Streams","\u002Fdocs\u002Fconcepts\u002Fhow-stoatflow-differs-from-ks","1.docs\u002F2.concepts\u002F9.how-stoatflow-differs-from-ks","i-lucide-arrow-left-right",{"title":92,"path":93,"stem":94,"children":95,"icon":98},"Building Topologies","\u002Fdocs\u002Fbuilding","1.docs\u002F3.building\u002F1.index",[96,99,104,109,114,119,124,129,134,139,144,149],{"title":97,"path":93,"stem":94,"icon":98},"Building topologies","i-lucide-blocks",{"title":100,"path":101,"stem":102,"icon":103},"Error handling and DLQ","\u002Fdocs\u002Fbuilding\u002Ferror-handling-dlq","1.docs\u002F3.building\u002F10.error-handling-dlq","i-lucide-circle-x",{"title":105,"path":106,"stem":107,"icon":108},"State stores","\u002Fdocs\u002Fbuilding\u002Fstate-stores","1.docs\u002F3.building\u002F11.state-stores","i-lucide-database",{"title":110,"path":111,"stem":112,"icon":113},"Testing topologies","\u002Fdocs\u002Fbuilding\u002Ftesting","1.docs\u002F3.building\u002F12.testing","i-lucide-flask-conical",{"title":115,"path":116,"stem":117,"icon":118},"Sources and sinks","\u002Fdocs\u002Fbuilding\u002Fstreams-builder","1.docs\u002F3.building\u002F2.streams-builder","i-lucide-import",{"title":120,"path":121,"stem":122,"icon":123},"KStream and KTable operations","\u002Fdocs\u002Fbuilding\u002Fkstream-ktable","1.docs\u002F3.building\u002F3.kstream-ktable","i-lucide-waypoints",{"title":125,"path":126,"stem":127,"icon":128},"Aggregations","\u002Fdocs\u002Fbuilding\u002Faggregations","1.docs\u002F3.building\u002F4.aggregations","i-lucide-sigma",{"title":130,"path":131,"stem":132,"icon":133},"Windowing","\u002Fdocs\u002Fbuilding\u002Fwindowing","1.docs\u002F3.building\u002F5.windowing","i-lucide-calendar-clock",{"title":135,"path":136,"stem":137,"icon":138},"Joins","\u002Fdocs\u002Fbuilding\u002Fjoins","1.docs\u002F3.building\u002F6.joins","i-lucide-git-merge",{"title":140,"path":141,"stem":142,"icon":143},"The Processor API","\u002Fdocs\u002Fbuilding\u002Fprocessor-api","1.docs\u002F3.building\u002F7.processor-api","i-lucide-cpu",{"title":145,"path":146,"stem":147,"icon":148},"Scheduled sources","\u002Fdocs\u002Fbuilding\u002Fscheduled-sources","1.docs\u002F3.building\u002F8.scheduled-sources","i-lucide-timer",{"title":150,"path":151,"stem":152,"icon":153},"Serdes and Avro","\u002Fdocs\u002Fbuilding\u002Fserdes","1.docs\u002F3.building\u002F9.serdes","i-lucide-binary",{"title":155,"path":156,"stem":157,"children":158,"icon":80},"Configuration","\u002Fdocs\u002Fconfiguration","1.docs\u002F4.configuration\u002F1.index",[159,161,166,171,176],{"title":160,"path":156,"stem":157,"icon":80},"How configuration works",{"title":162,"path":163,"stem":164,"icon":165},"Engine configuration (:core)","\u002Fdocs\u002Fconfiguration\u002Fcore-config","1.docs\u002F4.configuration\u002F2.core-config","i-lucide-settings-2",{"title":167,"path":168,"stem":169,"icon":170},"Runtime configuration (:runtime)","\u002Fdocs\u002Fconfiguration\u002Fruntime-config","1.docs\u002F4.configuration\u002F3.runtime-config","i-lucide-server-cog",{"title":172,"path":173,"stem":174,"icon":175},"Defaults, adaptivity, and presets","\u002Fdocs\u002Fconfiguration\u002Fdefaults-and-presets","1.docs\u002F4.configuration\u002F4.defaults-and-presets","i-lucide-gauge",{"title":177,"path":178,"stem":179,"icon":180},"Kafka client configuration","\u002Fdocs\u002Fconfiguration\u002Fkafka-client-config","1.docs\u002F4.configuration\u002F5.kafka-client-config","i-lucide-plug",{"title":182,"path":183,"stem":184,"children":185,"icon":188},"Running in Production","\u002Fdocs\u002Fruntime","1.docs\u002F5.runtime\u002F1.index",[186,189,194,199,204,209,214,219],{"title":187,"path":183,"stem":184,"icon":188},"The runtime","i-lucide-server",{"title":190,"path":191,"stem":192,"icon":193},"The REST API","\u002Fdocs\u002Fruntime\u002Frest-api","1.docs\u002F5.runtime\u002F2.rest-api","i-lucide-globe",{"title":195,"path":196,"stem":197,"icon":198},"Health checks","\u002Fdocs\u002Fruntime\u002Fhealth-checks","1.docs\u002F5.runtime\u002F3.health-checks","i-lucide-heart-pulse",{"title":200,"path":201,"stem":202,"icon":203},"Metrics","\u002Fdocs\u002Fruntime\u002Fmetrics","1.docs\u002F5.runtime\u002F4.metrics","i-lucide-activity",{"title":205,"path":206,"stem":207,"icon":208},"Pause and resume","\u002Fdocs\u002Fruntime\u002Fpause-unpause","1.docs\u002F5.runtime\u002F5.pause-unpause","i-lucide-pause",{"title":210,"path":211,"stem":212,"icon":213},"Plugins and lifecycle hooks","\u002Fdocs\u002Fruntime\u002Fplugins","1.docs\u002F5.runtime\u002F6.plugins","i-lucide-puzzle",{"title":215,"path":216,"stem":217,"icon":218},"Docker images","\u002Fdocs\u002Fruntime\u002Fdocker","1.docs\u002F5.runtime\u002F7.docker","i-lucide-container",{"title":220,"path":221,"stem":222,"icon":223},"GraalVM native image","\u002Fdocs\u002Fruntime\u002Fnative-image","1.docs\u002F5.runtime\u002F8.native-image","i-lucide-zap",{"title":225,"path":226,"stem":227,"children":228,"icon":231},"Deploying & Operating","\u002Fdocs\u002Foperating","1.docs\u002F6.operating\u002F1.index",[229,232,237,242,247,252,257],{"title":230,"path":226,"stem":227,"icon":231},"Deploying and operating","i-lucide-life-buoy",{"title":233,"path":234,"stem":235,"icon":236},"Running on Kubernetes","\u002Fdocs\u002Foperating\u002Fkubernetes","1.docs\u002F6.operating\u002F2.kubernetes","i-lucide-ship",{"title":238,"path":239,"stem":240,"icon":241},"High availability","\u002Fdocs\u002Foperating\u002Fhigh-availability","1.docs\u002F6.operating\u002F3.high-availability","i-lucide-copy",{"title":243,"path":244,"stem":245,"icon":246},"Liveness and readiness probes","\u002Fdocs\u002Foperating\u002Fprobes","1.docs\u002F6.operating\u002F4.probes","i-lucide-stethoscope",{"title":248,"path":249,"stem":250,"icon":251},"Observability","\u002Fdocs\u002Foperating\u002Fobservability","1.docs\u002F6.operating\u002F5.observability","i-lucide-telescope",{"title":253,"path":254,"stem":255,"icon":256},"Tuning under load","\u002Fdocs\u002Foperating\u002Ftuning","1.docs\u002F6.operating\u002F6.tuning","i-lucide-sliders",{"title":258,"path":259,"stem":260,"icon":261},"Production checklist","\u002Fdocs\u002Foperating\u002Fproduction-checklist","1.docs\u002F6.operating\u002F7.production-checklist","i-lucide-clipboard-check",{"title":263,"path":264,"stem":265,"children":266,"icon":90},"Migrating from Kafka Streams","\u002Fdocs\u002Fmigration","1.docs\u002F7.migration\u002F1.index",[267,268,273,278,283,288],{"title":263,"path":264,"stem":265,"icon":90},{"title":269,"path":270,"stem":271,"icon":272},"Automated port","\u002Fdocs\u002Fmigration\u002Fautomated-port","1.docs\u002F7.migration\u002F2.automated-port","i-lucide-wand-sparkles",{"title":274,"path":275,"stem":276,"icon":277},"Migration without carrying state","\u002Fdocs\u002Fmigration\u002Fwithout-data-migration","1.docs\u002F7.migration\u002F3.without-data-migration","i-lucide-sparkles",{"title":279,"path":280,"stem":281,"icon":282},"Migration carrying state","\u002Fdocs\u002Fmigration\u002Fwith-data-migration","1.docs\u002F7.migration\u002F4.with-data-migration","i-lucide-database-backup",{"title":284,"path":285,"stem":286,"icon":287},"The migration tool","\u002Fdocs\u002Fmigration\u002Fmigration-tool","1.docs\u002F7.migration\u002F5.migration-tool","i-lucide-truck",{"title":289,"path":290,"stem":291,"icon":292},"Reusing your Kafka Streams dashboards","\u002Fdocs\u002Fmigration\u002Freusing-kafka-streams-dashboards","1.docs\u002F7.migration\u002F6.reusing-kafka-streams-dashboards","i-lucide-line-chart",{"title":294,"path":295,"stem":296,"children":297,"icon":299},"Reference","\u002Fdocs\u002Freference","1.docs\u002F8.reference\u002F1.index",[298,300,305,310,315,320,325,329],{"title":294,"path":295,"stem":296,"icon":299},"i-lucide-list",{"title":301,"path":302,"stem":303,"icon":304},"Configuration reference","\u002Fdocs\u002Freference\u002Fconfiguration-reference","1.docs\u002F8.reference\u002F2.configuration-reference","i-lucide-table",{"title":306,"path":307,"stem":308,"icon":309},"REST API reference","\u002Fdocs\u002Freference\u002Frest-api-reference","1.docs\u002F8.reference\u002F3.rest-api-reference","i-lucide-network",{"title":311,"path":312,"stem":313,"icon":314},"Gradle plugin reference","\u002Fdocs\u002Freference\u002Fgradle-plugin-reference","1.docs\u002F8.reference\u002F4.gradle-plugin-reference","i-lucide-box",{"title":316,"path":317,"stem":318,"icon":319},"Maven reference","\u002Fdocs\u002Freference\u002Fmaven-reference","1.docs\u002F8.reference\u002F5.maven-reference","i-simple-icons-apachemaven",{"title":321,"path":322,"stem":323,"icon":324},"Kafka Streams compatibility matrix","\u002Fdocs\u002Freference\u002Fks-compatibility-matrix","1.docs\u002F8.reference\u002F6.ks-compatibility-matrix","i-lucide-table-2",{"title":326,"path":327,"stem":328,"icon":175},"Metrics reference","\u002Fdocs\u002Freference\u002Fmetrics-reference","1.docs\u002F8.reference\u002F7.metrics-reference",{"title":330,"path":331,"stem":332,"icon":333},"Glossary","\u002Fdocs\u002Freference\u002Fglossary","1.docs\u002F8.reference\u002F8.glossary","i-lucide-book-a",{"id":335,"title":336,"authors":337,"badge":343,"body":345,"date":1251,"description":1252,"draft":42,"extension":1253,"image":1254,"meta":1256,"navigation":1257,"path":1258,"seo":1259,"stem":1260,"__hash__":1261},"posts\u002F3.blog\u002F11.error-handling-three-gates-one-transaction.md","Three gates, one transaction: error handling in StoatFlow",[338],{"name":339,"to":340,"avatar":341},"Hartmut Armbruster","https:\u002F\u002Fwww.linkedin.com\u002Fin\u002Fhartmut-co-uk\u002F",{"src":342},"\u002Fassets\u002Fhartmut_armbruster_monochromatic.jpg",{"label":344},"Deep Dive",{"type":346,"value":347,"toc":1243},"minimark",[348,426,429,432,435,440,443,482,520,535,538,545,548,552,559,576,579,582,597,600,710,717,721,731,750,753,759,766,769,773,787,801,812,830,837,844,856,866,870,879,896,920,926,961,967,1003,1009,1015,1021,1031,1035,1073,1076,1087,1090,1097,1100,1103,1159,1183,1186,1189,1194,1239],[349,350,351,358],"blockquote",{},[352,353,354],"p",{},[355,356,357],"strong",{},"TL;DR",[359,360,361,380,391,402,412],"ul",{},[362,363,364,367,368,371,372,375,376,379],"li",{},[355,365,366],{},"Three gates:"," a record can fail to ",[355,369,370],{},"deserialize"," on the way in, a ",[355,373,374],{},"processor"," can throw mid-topology, or ",[355,377,378],{},"production"," (serialize + send) can fail on the way out. The handler surface is Kafka Streams' KIP-1033\u002F1034 shape, so the mental model ports.",[362,381,382,385,386,390],{},[355,383,384],{},"Three verdicts, plus retry:"," log-and-continue, log-and-fail (the default for deserialization and processing), or dead-letter. Production alone adds ",[387,388,389],"em",{},"retry"," — transient send errors back off exponentially before giving up.",[362,392,393,396,397,401],{},[355,394,395],{},"The rule:"," every failure resolves inside the transaction. A dead-lettered record rides the same transactional producer and commits on the same barrier as the epoch's output — never lost, never duplicated under ",[398,399,400],"code",{},"read_committed",". A FAIL aborts the in-flight epoch and replays from the last committed barrier.",[362,403,404,407,408,411],{},[355,405,406],{},"The catch:"," a broker-side rejection of an already-sent record (think ",[398,409,410],{},"max.message.bytes",") poisons the transaction. With a DLQ configured — which is the default the moment you set a DLQ topic — the epoch's innocent outputs are dropped: WARN-logged, counted on a dedicated metric, bounded by epoch size, and the source position advances past the poison so the application keeps making progress. Kafka Streams is materially worse on this exact scenario: it drops the same records, never advances, and aborts its own dead-letter record along with the transaction.",[362,413,414,417,418,421,422,425],{},[355,415,416],{},"Above the gates:"," a fatal error reaches the KS-compatible ",[398,419,420],{},"StreamsUncaughtExceptionHandler","; ",[398,423,424],{},"REPLACE_THREAD"," maps to an in-place engine restart, budgeted at 5 per 5 minutes. A failed commit is not configurable at all — abort, exit, restart from the last barrier.",[352,427,428],{},"Error handling is where exactly-once pipelines usually lose their guarantee — not in the transaction protocol, but in the catch block: a dead-letter record published through a second, non-transactional producer, or a skipped record whose offset advances with no trace left behind. Both look harmless in review. Both break the guarantee the rest of the system worked so hard for.",[352,430,431],{},"StoatFlow closes that gap with one rule. Every failure is resolved inside the transaction model: either a handler settles it within the current epoch — and anything the handler emits, dead-letter records included, commits on the same barrier as the epoch's output — or the failure is fatal and the epoch dies whole, replayed from the last committed barrier after recovery. There is no third state where the application limps on with a weakened guarantee.",[352,433,434],{},"This post walks the full model: the three places a record can fail, the verdicts you can configure at each, what a dead-letter queue means when it lives inside the transaction, and the escalation ladder above the handlers — from an in-place engine restart to the process exit that hands recovery to Kubernetes.",[436,437,439],"h2",{"id":438},"three-gates-three-verdicts","Three gates, three verdicts",[352,441,442],{},"A record can fail in three places, and the classes are worth keeping apart because each fails with different evidence in hand.",[359,444,445,451,465],{},[362,446,447,450],{},[355,448,449],{},"Deserialization, on the way in."," The record is still raw bytes, and the configured serde rejects the key or the value. The topology never saw it, so there is nothing to roll back — and the original bytes are still available, untouched, which matters for what a dead-letter record can carry.",[362,452,453,456,457,460,461,464],{},[355,454,455],{},"Processing, inside the topology."," The record deserialized, entered the DAG, and a processor threw — a ",[398,458,459],{},"NullPointerException"," in a ",[398,462,463],{},"mapValues",", a failed enrichment call, a bug. This happens on a processing lane, mid-flight, possibly after state was touched. It sounds like the dangerous one, but the same commit barrier that governs everything else governs those writes too: whatever a failed record did to state belongs to the current epoch, and the epoch commits as a whole or not at all. A skipped record cannot leak half-applied state into the committed snapshot.",[362,466,467,470,471,474,475,478,479,481],{},[355,468,469],{},"Production, on the way out."," The topology emitted a record and the runtime could not publish it. This class splits in two, and the split drives the policy: ",[355,472,473],{},"serialization"," failures are deterministic — the same value fails the same way every time, so StoatFlow never retries them — while ",[355,476,477],{},"send"," failures may be transient: a timeout, a briefly unreachable broker. Production is therefore the only class where ",[387,480,389],{}," is a meaningful verdict; the default production handler retries transient sends with exponential backoff before giving up.",[352,483,484,485,488,489,488,492,495,496,503,504,509,510,515,516,519],{},"Each class has its own exception handler, configured independently — ",[398,486,487],{},"deserialization.exception.handler",", ",[398,490,491],{},"processing.exception.handler",[398,493,494],{},"production.exception.handler",", or the typed builder equivalents. The interfaces follow Kafka Streams deliberately: the ",[497,498,502],"a",{"href":499,"rel":500},"https:\u002F\u002Fcwiki.apache.org\u002Fconfluence\u002Fdisplay\u002FKAFKA\u002FKIP-1033%3A+Add+Kafka+Streams+exception+handler+for+exceptions+occurring+during+processing",[501],"nofollow","KIP-1033"," processing handler, the ",[497,505,508],{"href":506,"rel":507},"https:\u002F\u002Fcwiki.apache.org\u002Fconfluence\u002Fdisplay\u002FKAFKA\u002FKIP-1034%3A+Dead+letter+queue+in+Kafka+Streams",[501],"KIP-1034"," dead-letter design, and ",[497,511,514],{"href":512,"rel":513},"https:\u002F\u002Fcwiki.apache.org\u002Fconfluence\u002Fpages\u002Fviewpage.action?pageId=311627309",[501],"KIP-1065","'s retry option, with the same ",[398,517,518],{},"handle(context, record, exception)"," signature Kafka Streams 4.x uses. Those KIPs got the shape right, and keeping to it means an existing application's error-handling setup ports without relearning anything.",[352,521,522,523,526,527,530,531,534],{},"Every handler chooses between the same verdicts: ",[355,524,525],{},"log and continue"," (skip the record and move on), ",[355,528,529],{},"log and fail"," (stop the application), or ",[355,532,533],{},"dead-letter"," (route the record and its error context to a topic you own, then continue). The defaults are strict on purpose — deserialization and processing both default to log-and-fail. Skipping records should be a decision you make explicitly, not a behaviour you inherit.",[352,536,537],{},"The whole surface fits in one picture:",[352,539,540],{},[541,542],"img",{"alt":543,"src":544},"A record travelling left to right past three gates — deserialize before the topology, process mid-DAG on a lane, and produce, which splits into serialize (never retried) and send (may be transient). Each gate offers the same log-and-continue, log-and-fail and dead-letter verdicts and carries its own default; produce adds a fourth, retry. Every dead-letter verdict feeds one Kafka transaction that commits the output records, the DLQ records and the state plus source offsets together on the same barrier.","\u002Fassets\u002Fdocs\u002Fconcepts\u002Ffailure-gates_20260727.svg",[352,546,547],{},"Look at the right edge. Every dead-letter arrow, from every gate, lands in the same place: one Kafka transaction holding the epoch's output records, its dead-letter records, and its state changes plus source offsets. That box is the model — the rest of this post is what happens inside it, and what happens when something cannot get into it.",[436,549,551],{"id":550},"the-dead-letter-record-rides-the-transaction","The dead-letter record rides the transaction",[352,553,554,555,558],{},"A DLQ handler in StoatFlow does not publish to the dead-letter topic itself. It returns the failed record — wrapped as a ",[398,556,557],{},"ProducerRecord"," targeting your DLQ topic — attached to its verdict, and the engine sends it through the same transactional producer as the epoch's normal output. (There is no separate dead-letter verdict in the response enum; dead-lettering is a continue-or-fail decision that carries records, the same shape KIP-1034 chose.) The DLQ record commits on the same barrier as everything else, which buys two properties a hand-rolled dead-letter path cannot offer:",[359,560,561,567],{},[362,562,563,566],{},[355,564,565],{},"No loss."," If the epoch commits, the DLQ record is on the topic. If the epoch aborts — a crash mid-commit — the DLQ record is discarded with everything else, the source offset never advanced, and the record is re-read and re-handled after restart.",[362,568,569,572,573,575],{},[355,570,571],{},"No duplicates."," The record appears on the DLQ topic exactly once for a ",[398,574,400],{}," consumer — the same guarantee as your real output.",[352,577,578],{},"Compare that with the pattern most teams build by hand: a second producer in the catch block. It has exactly two failure modes. Either the DLQ write lands and the offset commit does not, and after restart the record is dead-lettered again — duplicates on the DLQ. Or the offset commits and the DLQ write did not, and the record is gone with nothing to show for it. Across enough restarts you will meet both, and both are impossible when the DLQ record and the offset advance are one atomic commit.",[352,580,581],{},"Kafka Streams gained the same response-attached DLQ shape with KIP-1034, so the interface is shared; the context differs. Exactly-once is StoatFlow's default processing guarantee, so DLQ-on-the-barrier is the out-of-the-box behaviour rather than a property of a mode you remembered to enable.",[352,583,584,585,588,589,592,593,596],{},"What lands on the topic depends on the gate. Deserialization DLQ records preserve the original raw key and value bytes verbatim — nothing ever deserialized, so the untouched payload is exactly what you need to diagnose or replay. And every DLQ record carries error context in headers under a ",[398,586,587],{},"__stoatflow.errors.*"," namespace: the exception class and message, the source topic, partition and offset, the failing component, optionally the stack trace. The namespace is deliberately not Kafka Streams' ",[398,590,591],{},"__streams.errors.*"," — tooling should be able to tell which engine produced a dead letter. The full header table is in the ",[497,594,595],{"href":101},"error-handling guide",".",[352,598,599],{},"Wiring it is two builder calls. This captures bad input and undeliverable output while keeping processing on its fail-fast default:",[601,602,607],"pre",{"className":603,"code":604,"language":605,"meta":606,"style":606},"language-kotlin shiki shiki-themes vitesse-light","streamsConfigOverrides {\n    \u002F\u002F capture undeserialisable input — original bytes preserved on the DLQ record\n    deserializationExceptionHandler(\n        DeadLetterQueueDeserializationExceptionHandler(dlqTopic = \"orders.deserialization.dlq\"),\n    )\n    \u002F\u002F retry transient sends; dead-letter what cannot be delivered\n    productionExceptionHandler(\n        DefaultProductionExceptionHandler(dlqTopic = \"orders.production.dlq\"),\n    )\n    \u002F\u002F processing stays log-and-fail: a bug should stop the app, not drain into a topic\n}\n","kotlin","",[398,608,609,622,629,638,658,664,670,678,693,698,704],{"__ignoreMap":606},[610,611,614,618],"span",{"class":612,"line":613},"line",1,[610,615,617],{"class":616},"sySUi","streamsConfigOverrides",[610,619,621],{"class":620},"suHK_"," {\n",[610,623,625],{"class":612,"line":624},2,[610,626,628],{"class":627},"s8zF2","    \u002F\u002F capture undeserialisable input — original bytes preserved on the DLQ record\n",[610,630,632,635],{"class":612,"line":631},3,[610,633,634],{"class":616},"    deserializationExceptionHandler",[610,636,637],{"class":620},"(\n",[610,639,641,644,647,651,655],{"class":612,"line":640},4,[610,642,643],{"class":616},"        DeadLetterQueueDeserializationExceptionHandler",[610,645,646],{"class":620},"(dlqTopic ",[610,648,650],{"class":649},"sYZai","=",[610,652,654],{"class":653},"spphp"," \"orders.deserialization.dlq\"",[610,656,657],{"class":620},"),\n",[610,659,661],{"class":612,"line":660},5,[610,662,663],{"class":620},"    )\n",[610,665,667],{"class":612,"line":666},6,[610,668,669],{"class":627},"    \u002F\u002F retry transient sends; dead-letter what cannot be delivered\n",[610,671,673,676],{"class":612,"line":672},7,[610,674,675],{"class":616},"    productionExceptionHandler",[610,677,637],{"class":620},[610,679,681,684,686,688,691],{"class":612,"line":680},8,[610,682,683],{"class":616},"        DefaultProductionExceptionHandler",[610,685,646],{"class":620},[610,687,650],{"class":649},[610,689,690],{"class":653}," \"orders.production.dlq\"",[610,692,657],{"class":620},[610,694,696],{"class":612,"line":695},9,[610,697,663],{"class":620},[610,699,701],{"class":612,"line":700},10,[610,702,703],{"class":627},"    \u002F\u002F processing stays log-and-fail: a bug should stop the app, not drain into a topic\n",[610,705,707],{"class":612,"line":706},11,[610,708,709],{"class":620},"}\n",[352,711,712,713,716],{},"The dead-letter topics are ordinary topics you create, own and retain; the runtime only produces to them and never reads them back, and there is no auto-naming. One boundary worth drawing: a record that parses fine but fails your business rules is not an exception — route those explicitly with a branch and a sink in the topology, where you control the payload and the serde. The ",[497,714,715],{"href":101},"guide"," shows both patterns side by side.",[436,718,720],{"id":719},"fail-means-the-epoch-dies","Fail means the epoch dies",[352,722,723,724,727,728,730],{},"The other half of the model is what ",[387,725,726],{},"fail"," actually does. A FAIL verdict — the default handler's, or your custom handler deciding an exception is not survivable — does not try to unwind one record. It ends the epoch. The in-flight transaction aborts at the broker, and with it everything the epoch had done: output records, changelog writes, state store changes, offset advances, all discarded together. Recovery resumes from the last committed barrier and re-processes everything after it. A ",[398,729,400],{}," consumer never saw the aborted work.",[352,732,733,734,738,739,742,743,745,746,749],{},"One path is exempt, and it is the subject of ",[497,735,737],{"href":736},"#the-honest-part-what-this-model-costs","the honest part"," below: on an asynchronous broker-side production failure the offsets have already advanced by the time the verdict is read, so that epoch is ",[387,740,741],{},"not"," replayed — and ",[387,744,726],{}," loses exactly what ",[387,747,748],{},"continue"," loses.",[352,751,752],{},"FAIL is a controlled crash with crash-recovery semantics, in other words — and that is the feature. There is no degraded mode in which the application keeps running with the guarantee suspended.",[352,754,755],{},[541,756],{"alt":757,"src":758},"A commit barrier cascading downstream through three sub-topologies against a wall-clock axis. The span between two commits is one epoch, and transaction N commits state, output and offsets together — all three or none of them. A crash partway through epoch N plus 1 aborts transaction N plus 1, discarding its state, output and offsets together; that output was never visible to a read_committed consumer. On restart, state rebuilds to the last committed barrier and the consumer resumes from transaction N's offsets, re-processing everything after it into a fresh epoch.","\u002Fassets\u002Fdocs\u002Fconcepts\u002Fcommit-barrier-epochs_20260727.svg",[352,760,761,762,765],{},"The diagram is the same one our docs use to explain exactly-once, and that is no accident: failure recovery ",[387,763,764],{},"is"," the exactly-once mechanism, pointed at a different trigger. An epoch dies the same way whether a processor threw or the process crashed.",[352,767,768],{},"This is also where the strict defaults earn their keep. Under exactly-once, a FAIL costs replay time and nothing else — no duplicates, no partial state, no reconciliation. Failing fast is cheap, so it can be the default. Under an at-least-once default — which is what Kafka Streams ships — a crash means duplicates, and that price pushes teams towards continue-and-hope handlers. Same handler API, different default guarantee, different economics.",[436,770,772],{"id":771},"above-the-gates-replace-the-engine-or-exit-the-process","Above the gates: replace the engine or exit the process",[352,774,775,776,778,779,488,781,488,784,596],{},"A FAIL verdict stops the topology. What happens next follows a ladder, and its top rung is the one Kafka Streams users already know: the fatal error is offered to a ",[398,777,420],{},", the KS-compatible interface with the familiar three answers — ",[398,780,424],{},[398,782,783],{},"SHUTDOWN_CLIENT",[398,785,786],{},"SHUTDOWN_APPLICATION",[352,788,789,790,792,793,796,797,596],{},"There is no stream thread to replace in StoatFlow, so ",[398,791,424],{}," maps to the strongest recovery the architecture allows: an ",[355,794,795],{},"in-place engine restart",". The processing engine is torn down inside the live process — the faulted epoch aborted, exactly as above — and a fresh engine is built and resumes from the last committed barrier. State stores stay open and the JVM stays warm, so recovery costs engine-rebuild time rather than pod-reschedule time. The mechanics have ",[497,798,800],{"href":799},"\u002Fblog\u002Fin-place-restart-multi-standby","their own post",[352,802,803,804,807,808,811],{},"Two constraints keep that honest. Only processing and production failures are restartable — a record that cannot deserialize will not deserialize for a rebuilt engine either, so a deserialization FAIL goes straight to shutdown, as do punctuator failures and commit stalls. And restarts are budgeted — ",[398,805,806],{},"commit-barrier.max-engine-restarts"," (default 5) within ",[398,809,810],{},"commit-barrier.engine-restart-window-ms"," (default 5 minutes) — so a recurring fault escalates to a terminal shutdown instead of looping forever.",[813,814,817],"callout",{"color":815,"icon":816},"info","i-lucide-info",[352,818,819,825,826,829],{},[355,820,821,822,824],{},"A deserialization FAIL on Kubernetes ",[387,823,764],{}," a restart — and that is the point."," The pod exits, the kubelet starts it again, the same record is still there, and it fails again. What you get is not recovery, it is a loud ",[398,827,828],{},"CrashLoopBackOff"," and an alert. Choose that deliberately: it is the right answer when a malformed record means something upstream is broken and you want processing halted until a human looks. If you would rather keep running, handle the poison record — log-and-continue, or route it to a dead-letter topic.",[352,831,832,833,836],{},"Hot standby does not change any of that. Turn it on and a restartable fault is still absorbed in place — the pod keeps the active role, the standby stays a standby, and no failover happens. The role moves only when the budget above is spent, and then it is a graceful hand-off: one failover after a bounded number of local attempts, rather than a role transfer per fault. (An earlier version of this post said the opposite, and said the handler was not consulted under HA. Both were wrong; the ",[497,834,835],{"href":239},"high-availability page"," has the current behaviour.)",[352,838,839,840,843],{},"Everything that is not restartable ends the same way: a graceful shutdown that aborts the in-flight transaction and exits the process, backstopped by a hard-exit timer so that a shutdown which hangs still becomes a clean ",[398,841,842],{},"System.exit(1)"," rather than a zombie pod. From Kubernetes' point of view that is an ordinary container restart. From the data's point of view, it is a resume from the last committed barrier.",[352,845,846,847,850,851,855],{},"At the bottom of the ladder sits the one failure no handler is consulted about: the commit itself failing — a transaction timeout, a broker rejection, a fenced producer. There is no policy hook because there is no safe alternative to abort-and-restart. Kafka Streams has one more move here: it can migrate the fenced task to another instance. StoatFlow does not, and the trade is deliberate. Task migration is part of the ",[387,848,849],{},"distribution tax"," — a rebalance protocol, standby-task placement, and a state-transfer path, all of which you operate and debug — and what replaces it is a shorter contract. The single instance restarts, and the restart is cheap: state stores are node-local and carry their own committed offsets, so recovery ",[497,852,854],{"href":853},"\u002Fblog\u002Fkip-1035-state-store-managed-offsets","delta-restores the aborted epoch"," rather than rebuilding from the changelog, in about the time it takes the container to come back, largely independent of how much state you hold. Every failure, at every rung of the ladder, resolves to the same known-good place: the last committed barrier.",[352,857,858,859,862,863,865],{},"(If you run ",[497,860,861],{"href":239},"hot-standby HA",", there ",[387,864,764],{}," another instance — a warm standby that takes the role over. That is an availability layer bolted onto the same contract, not the distribution model coming back: still exactly one instance processing, still no partition-level task migration.)",[436,867,869],{"id":868},"the-honest-part-what-this-model-costs","The honest part: what this model costs",[352,871,872,875,876,878],{},[355,873,874],{},"One production path used to drop records that did nothing wrong. It no longer does — and this section\nsaid otherwise when the post first went out."," Everything above describes synchronous failures, caught\nbefore or during the send. A broker-side rejection is different: the record was already handed to the\nproducer and rejected asynchronously — the classic case is a record exceeding the topic's\n",[398,877,410],{},". That rejection poisons the whole in-flight transaction; nothing in it can commit any\nmore.",[352,880,881,882,885,886,888,889,891,892,895],{},"The original version of this post described what StoatFlow then did: abort the epoch, commit a small\nsecondary transaction carrying only the poison's dead letter ",[355,883,884],{},"and the epoch's offset advance",", and move\non — losing every innocent output in that epoch, on ",[387,887,726],{}," exactly as on ",[387,890,748],{},". That was accurate,\nand it was the wrong design. It rested on the claim that dropping the epoch is the only terminating\nsemantic for a deterministic poison under batched exactly-once, which is true only ",[387,893,894],{},"without replay\nmachinery",". StoatFlow has that machinery: the in-place engine restart that absorbs a restartable fault\nhigher up this same ladder.",[352,897,898,899,902,903,906,907,910,911,913,914,916,917,919],{},"So the semantics changed. The offsets are now ",[355,900,901],{},"held"," on every verdict, and the epoch is ",[355,904,905],{},"replayed","\nwith the poison skipped: the secondary transaction carries the dead letter alone, the poison's\n",[398,908,909],{},"(topic, partition, offset)"," goes into a quarantine that outlives the engine, the engine restarts in\nplace, and the epoch is reprocessed with exactly that record dropped. Its epoch-mates commit normally.\n",[387,912,726],{}," now means what it says — stop before further harm, with the offsets held so an operator restart\nreplays the epoch — and it is no longer eligible for the in-place restart rung, which used to make it\nquietly equivalent to ",[387,915,748],{}," for anyone running a ",[398,918,424],{}," handler.",[352,921,922],{},[541,923],{"alt":924,"src":925},"A wall-clock timeline of one epoch. Source records are read while innocent output records and state writes accumulate, and one output record is sent that the broker rejects asynchronously. At the commit barrier the whole transaction aborts, but the source offsets are held rather than advanced: a secondary transaction commits only the poison's dead-letter record, the engine restarts in place, and the epoch is replayed with the poison record skipped so the innocent records commit normally.","\u002Fassets\u002Fdocs\u002Fconcepts\u002Fpoison-epoch-abort_20260813.svg",[352,927,928,931,932,935,936,939,940,943,944,946,947,949,950,953,954,957,958,960],{},[355,929,930],{},"What it still costs, because \"no data is lost\" would be the same kind of overstatement."," Quarantining a\nsource record skips ",[387,933,934],{},"all"," of its outputs, so a record that fans out to several sinks loses the good sends\nwith the bad one — read the guarantee as ",[387,937,938],{},"loss is bounded to the poison record's own outputs",". And a poison\nderived from ",[355,941,942],{},"accumulated state",", an aggregate that outgrew ",[398,945,410],{},", does not converge:\nskipping the triggering record does not shrink the accumulator, so the next record on that key reproduces\nit. Each replay costs a full engine restart and spends a budget that is deliberately in-memory, so\nexhausting it ends the process, the pod restarts, and the budget comes back clean — ",[387,948,748],{}," is bounded\n",[355,951,952],{},"per process",", never end to end. Alert on ",[398,955,956],{},"stoatflow.dlq.poison.replays.total",". The thing that prevents\nthe case entirely is still sizing the target topic's ",[398,959,410],{}," for your largest output.",[352,962,963,964,966],{},"Two cases cannot be replayed at all and stop the instance immediately, offsets held: a ",[387,965,726],{}," verdict, and\na poison emitted by something with no source record behind it — a punctuator, a window close, a suppression\nflush, a timer, a scheduled source. Those fire on their own schedule and would re-poison every attempt, so\nthe runtime names the plane and stops rather than burning the budget discovering it.",[352,968,969,972,973,975,976,979,980,983,984,987,988,993,994,996,997,999,1000,596],{},[355,970,971],{},"A poisoned exactly-once transaction has no gentle exit anywhere, but the exits differ."," Kafka Streams\n4.3.1 honours ",[387,974,748],{}," locally and then loses on three counts. Its offsets ride\n",[398,977,978],{},"sendOffsetsToTransaction",", which throws once the producer is in ",[398,981,982],{},"ABORTABLE_ERROR",", so nothing reaches\n",[398,985,986],{},"__consumer_offsets"," and the restart replays straight back onto the same poison — the\n",[497,989,992],{"href":990,"rel":991},"https:\u002F\u002Fissues.apache.org\u002Fjira\u002Fbrowse\u002FKAFKA-15259",[501],"open bug"," for exactly this. Its own KIP-1034\ndead-letter record is appended to the transaction the rejection already doomed, so it is aborted with\neverything else and no ",[398,995,400],{}," consumer ever sees it. And the commit throws, the stream thread\ndies, and the default ",[398,998,783],{}," stops the client — under a supervisor, a restart loop onto the same\nrecord with no evidence written anywhere. StoatFlow now also stops advancing, deliberately; the difference\nis what survives it. We keep a Testcontainers probe against real Kafka Streams 4.3.1 so that comparison\nstays true; on the run behind this post it observed a dead-letter topic with no dead letter for the poison,\na committed offset still pointing at the poison, and a client in ",[398,1001,1002],{},"ERROR",[352,1004,1005,1008],{},[355,1006,1007],{},"Log-and-continue is data loss with a log line."," Without a DLQ, a skipped record leaves a log entry and a metric tick, nothing else. If there is any chance you will want the record back, dead-letter it instead.",[352,1010,1011,1014],{},[355,1012,1013],{},"Processing dead letters are not the original bytes."," Only the deserialization DLQ preserves the raw payload. A processing failure happens after deserialization, so its DLQ record carries the key and value rendered to strings — good for diagnosis, not a byte-faithful replay source.",[352,1016,1017,1020],{},[355,1018,1019],{},"The DLQ topics are yours."," You create them, size their retention, and monitor them; nothing is auto-created. And because the DLQ handlers take the topic as a constructor argument, they cannot be configured from YAML, which can only instantiate no-arg handlers — wire them in code, or write a small no-arg subclass that hard-codes the topic.",[352,1022,1023,1026,1027,1030],{},[355,1024,1025],{},"An exhausted restart budget is downtime."," Five faults in five minutes ends the process, and unless you run the opt-in ",[497,1028,1029],{"href":239},"hot standby",", recovery is a cold start with state restoration ahead of it.",[436,1032,1034],{"id":1033},"what-you-see-when-it-breaks","What you see when it breaks",[352,1036,1037,1038,1041,1042,1045,1046,1048,1049,1052,1053,1056,1057,1060,1061,1064,1065,1068,1069,1072],{},"Every verdict leaves a signal, and the metric names are worth knowing before you need them: ",[398,1039,1040],{},"stoatflow.error.total"," counts errors by type and topic, ",[398,1043,1044],{},"stoatflow.dropped.records.total"," is the Kafka Streams-parity dropped-records counter, ",[398,1047,956],{}," counts the epoch replays above (with ",[398,1050,1051],{},"stoatflow.dlq.poison.quarantined.total"," and ",[398,1054,1055],{},"stoatflow.dlq.poison.quarantine.size"," showing what is being skipped), ",[398,1058,1059],{},"stoatflow.engine.restart.total"," tags each in-place restart with its trigger, and ",[398,1062,1063],{},"stoatflow.barrier.failed.total"," plus ",[398,1066,1067],{},"stoatflow.commit.stall.detected.total"," cover the commit path. All are exposed on ",[398,1070,1071],{},"\u002Fmetrics"," in Prometheus form.",[352,1074,1075],{},"If you set one alert on day one, make it this one:",[601,1077,1081],{"className":1078,"code":1079,"language":1080,"meta":606,"style":606},"language-promql shiki shiki-themes vitesse-light","rate(stoatflow_engine_restart_total{trigger=\"replace_thread\"}[10m]) > 0\n","promql",[398,1082,1083],{"__ignoreMap":606},[610,1084,1085],{"class":612,"line":613},[610,1086,1079],{},[352,1088,1089],{},"An in-place restart is self-healing, but a fault that recurs is still a fault — and the budget means five of them in five minutes will end the process. Alerting on the first buys you the investigation window.",[352,1091,1092,1093,1096],{},"Alert on ",[398,1094,1095],{},"rate(stoatflow_dlq_poison_replays_total[10m]) > 0"," for the same reason, and with more urgency: a poison replay is self-healing exactly once per poison, its budget is in-memory, and a poison that does not converge will spend that budget and crash-loop the pod.",[352,1098,1099],{},"The commit path also watches itself: a commit-pipeline watchdog (45-second stall threshold by default) and bounded waits on every commit-critical call turn a silent freeze into a loud failure — a stall exception with a thread dump attached, then the restart path above. In practice that is the difference between a consumer-lag graph climbing while the process looks healthy, and a process that tells you what it was stuck on before recovering.",[352,1101,1102],{},"And when a record lands on a DLQ topic, the headers make it self-describing:",[601,1104,1108],{"className":1105,"code":1106,"language":1107,"meta":606,"style":606},"language-bash shiki shiki-themes vitesse-light","kafka-console-consumer.sh --bootstrap-server localhost:9092 \\\n  --topic orders.deserialization.dlq --from-beginning \\\n  --property print.headers=true --property print.key=true\n","bash",[398,1109,1110,1125,1138],{"__ignoreMap":606},[610,1111,1112,1115,1119,1122],{"class":612,"line":613},[610,1113,1114],{"class":616},"kafka-console-consumer.sh",[610,1116,1118],{"class":1117},"sEi1f"," --bootstrap-server",[610,1120,1121],{"class":653}," localhost:9092",[610,1123,1124],{"class":1117}," \\\n",[610,1126,1127,1130,1133,1136],{"class":612,"line":624},[610,1128,1129],{"class":1117},"  --topic",[610,1131,1132],{"class":653}," orders.deserialization.dlq",[610,1134,1135],{"class":1117}," --from-beginning",[610,1137,1124],{"class":1117},[610,1139,1140,1143,1146,1150,1153,1156],{"class":612,"line":631},[610,1141,1142],{"class":1117},"  --property",[610,1144,1145],{"class":653}," print.headers=",[610,1147,1149],{"class":1148},"sbBg2","true",[610,1151,1152],{"class":1117}," --property",[610,1154,1155],{"class":653}," print.key=",[610,1157,1158],{"class":1148},"true\n",[352,1160,1161,1164,1165,1168,1169,1168,1172,1175,1176,1052,1179,1182],{},[398,1162,1163],{},"__stoatflow.errors.type"," names the gate that fired, ",[398,1166,1167],{},"__stoatflow.errors.topic"," \u002F ",[398,1170,1171],{},".partition",[398,1173,1174],{},".offset"," point at the exact source record, and ",[398,1177,1178],{},"__stoatflow.errors.exception",[398,1180,1181],{},".message"," carry the why — enough to triage from the console before any tooling gets involved.",[352,1184,1185],{},"A restart, throughout all of this, is the recovery path rather than an incident: the readiness probe stays down while a restarted instance restores state, so traffic waits until it has caught up. The design assumes restarts happen and makes them boring.",[352,1187,1188],{},"That is the whole model. Three gates a record can fail at, the same verdicts at each, a dead-letter path that lives inside the exactly-once transaction instead of beside it — and above the handlers, a ladder of engine restart, process exit and orchestrator restart where every rung ends at the last committed barrier. Failures are handled inside the epoch, or the epoch dies. Nothing in between.",[352,1190,1191],{},[355,1192,1193],{},"Read on:",[359,1195,1196,1201,1206,1212,1221,1227,1233],{},[362,1197,1198,1200],{},[497,1199,82],{"href":83}," — the behavioural reference this post narrates.",[362,1202,1203,1205],{},[497,1204,100],{"href":101}," — handler classes, config keys, the full header table, and custom handlers in Kotlin and Java.",[362,1207,1208,1211],{},[497,1209,1210],{"href":58},"Exactly-once"," — the commit barrier the whole model hangs off.",[362,1213,1214,1217,1218,1220],{},[497,1215,1216],{"href":799},"In-place engine restart: the primitive behind multi-standby HA"," — what ",[398,1219,424],{}," actually does.",[362,1222,1223],{},[497,1224,1226],{"href":499,"rel":1225},[501],"KIP-1033: Add Kafka Streams exception handler for exceptions occurring during processing",[362,1228,1229],{},[497,1230,1232],{"href":506,"rel":1231},[501],"KIP-1034: Dead letter queue in Kafka Streams",[362,1234,1235],{},[497,1236,1238],{"href":512,"rel":1237},[501],"KIP-1065: Add \"retry\" return-option to ProductionExceptionHandler",[1240,1241,1242],"style",{},"html pre.shiki code .sySUi, html code.shiki .sySUi{--shiki-default:#59873A}html pre.shiki code .suHK_, html code.shiki .suHK_{--shiki-default:#393A34}html pre.shiki code .s8zF2, html code.shiki .s8zF2{--shiki-default:#A0ADA0}html pre.shiki code .sYZai, html code.shiki .sYZai{--shiki-default:#999999}html pre.shiki code .spphp, html code.shiki .spphp{--shiki-default:#B56959}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html pre.shiki code .sEi1f, html code.shiki .sEi1f{--shiki-default:#A65E2B}html pre.shiki code .sbBg2, html code.shiki .sbBg2{--shiki-default:#1E754F}",{"title":606,"searchDepth":624,"depth":624,"links":1244},[1245,1246,1247,1248,1249,1250],{"id":438,"depth":624,"text":439},{"id":550,"depth":624,"text":551},{"id":719,"depth":624,"text":720},{"id":771,"depth":624,"text":772},{"id":868,"depth":624,"text":869},{"id":1033,"depth":624,"text":1034},"2026-07-27","A record can fail on the way in, in the middle, or on the way out. StoatFlow gives every gate the same verdicts — continue, fail, or dead-letter — and settles them all inside the exactly-once transaction: DLQ records commit on the same barrier as your output, and what cannot be handled kills the epoch, never the guarantee. The full model, from one bad record to a Kubernetes restart.","md",{"src":1255},"\u002Fassets\u002Fblog\u002Fog\u002Ferror-handling-three-gates-one-transaction.png",{},true,"\u002Fblog\u002Ferror-handling-three-gates-one-transaction",{"title":336,"description":1252},"3.blog\u002F11.error-handling-three-gates-one-transaction","j9B0N_vo6SAahe7plthehvArSRQquqeAp6RKfar0AMY",[1263,1268],{"title":1264,"path":1265,"stem":1266,"description":1267,"children":-1},"From first alpha to release candidate: the road to StoatFlow 1.0.0","\u002Fblog\u002Froad-to-1-0-0","3.blog\u002F12.road-to-1-0-0","StoatFlow 1.0.0-rc.1 is cut and the feature set is frozen — what separates the candidate from GA is proof, not features. The twelve weeks from the first alpha: 26 releases, 1,086 commits, a compatibility matrix grown from 417 to 619 tracked entries, three full-codebase review rounds — and the framework integration that kept us honest.",{"title":1269,"path":1270,"stem":1271,"description":1272,"children":-1},"llms.txt for StoatFlow: docs your AI agent can fetch","\u002Fblog\u002Fllms-txt-machine-readable-docs","3.blog\u002F10.llms-txt-machine-readable-docs","The StoatFlow documentation is now published as llms.txt, llms-full.txt, and raw markdown — the retrieval-side complement to the AI Assistant Skills pack. Nothing to install.",1786987415769]