{"id":232436,"date":"2026-10-01T13:29:35","date_gmt":"2026-10-01T13:29:35","guid":{"rendered":"https:\/\/www.intrinsec.com\/?p=232436"},"modified":"2026-10-02T08:40:52","modified_gmt":"2026-10-02T08:40:52","slug":"ai-assisted-development-what-we-changed-and-why","status":"publish","type":"post","link":"https:\/\/www.intrinsec.com\/en\/ai-assisted-development-what-we-changed-and-why\/","title":{"rendered":"Industrializing AI-Assisted Development: What We Changed and Why"},"content":{"rendered":"<!-- =============================================================\n     Intrinsec blog \u2014 \"Industrialising AI-Assisted Development\"\n     Paste this WHOLE block into a Gutenberg \"Custom HTML\" block.\n     The post TITLE is set separately in WordPress (do not re-add an H1 here).\n     All CSS is scoped under .iagen-post so nothing touches the theme.\n     ============================================================= -->\n<style>\n@import url('https:\/\/fonts.googleapis.com\/css2?family=Poppins:ital,wght@0,300;0,400;0,500;0,600;0,700;1,400&display=swap');\n\n.iagen-post{\n  --paper:#ffffff; --paper-alt:#f6f9fb; --band:#0a3055;\n  --ink:#070707; --body:#39454f; --muted:#6e7a84;\n  --rule:#e6e9ec; --rule-strong:#cfd6dc;\n  --navy:#0a3055; --red:#c41718; --accent:#d1345b;\n  --link:#0a3055; --code-bg:#f4f6f8; --code-fg:#0a3055;\n  --fig-bg:#fbfcfd; --fig-line:#cfd6dc; --fig-soft:#eef2f5;\n  --text-w:clamp(20rem,54vw,62rem);\n\n  background:#ffffff; color:var(--body);\n  font-family:\"Poppins\",-apple-system,BlinkMacSystemFont,\"Segoe UI\",Roboto,Helvetica,Arial,sans-serif;\n  font-size:16px; line-height:1.75; font-weight:400;\n  -webkit-font-smoothing:antialiased;\n  max-width:calc(var(--text-w) + 3rem); margin:1.5rem auto; padding:1.5rem;\n  overflow-x:hidden;\n}\n\/* Light design only, forced to match the Intrinsec blog (white content area).\n   No prefers-color-scheme:dark block, so the article never flips to a dark\n   palette regardless of the reader's or editor's browser theme. *\/\n.iagen-post *{box-sizing:border-box}\n\n\/* one measure for everything, figures included *\/\n.iagen-post > *,\n.iagen-post > section > *{\n  max-width:var(--text-w); margin-left:auto; margin-right:auto;\n}\n\n\/* intro \/ lead *\/\n.iagen-post .lead{\n  font-size:1.14rem; line-height:1.55; color:var(--ink); font-weight:300;\n  margin:0 0 1.4rem;\n}\n.iagen-post .lead em{font-style:italic}\n.iagen-post .byline{\n  font-size:.82rem; letter-spacing:.06em; text-transform:uppercase;\n  color:var(--muted); font-weight:500; margin:0 0 .5rem;\n}\n.iagen-post .intro-rule{border:0; border-top:2px solid var(--red); margin:1.6rem auto 2rem; max-width:var(--text-w)}\n\n\/* type *\/\n.iagen-post h2{\n  font-size:clamp(1.45rem,2.4vw,1.85rem); line-height:1.175; letter-spacing:-.027em;\n  font-weight:700; color:var(--ink); margin:3.6rem auto 1.1rem; padding-top:1.6rem;\n  position:relative;\n}\n.iagen-post h2::before{\n  content:\"\"; position:absolute; top:0; left:0; width:46px; height:3px; background:var(--red);\n}\n.iagen-post h3{\n  font-size:1.12rem; line-height:1.3; letter-spacing:-.02em; font-weight:600;\n  color:var(--ink); margin:2.4rem auto .7rem;\n}\n.iagen-post p{margin:0 auto 1.15rem}\n.iagen-post a{color:var(--link); text-decoration-thickness:1px; text-underline-offset:.18em}\n.iagen-post a:hover{color:var(--red)}\n.iagen-post strong{font-weight:600; color:var(--ink)}\n.iagen-post em{font-style:italic}\n.iagen-post ul,.iagen-post ol{margin:0 auto 1.3rem; padding-left:1.3rem}\n.iagen-post li{margin:0 0 .6rem; padding-left:.2rem}\n.iagen-post li::marker{color:var(--accent)}\n.iagen-post li > ul,.iagen-post li > ol{margin-top:.6rem}\n\n.iagen-post blockquote{\n  margin:2rem auto; padding:1.5rem 1.75rem;\n  border-left:4px solid var(--red); background:var(--paper-alt);\n  border-radius:0 8px 8px 0;\n}\n.iagen-post blockquote p{\n  margin:0; font-size:1.2rem; line-height:1.5; font-weight:300;\n  color:var(--ink); letter-spacing:-.015em;\n}\n\n.iagen-post code{\n  font-family:ui-monospace,SFMono-Regular,Menlo,Consolas,\"Liberation Mono\",monospace;\n  font-size:.86em; background:var(--code-bg); color:var(--code-fg);\n  padding:.12em .38em; border-radius:4px;\n}\n.iagen-post pre{\n  background:var(--code-bg); color:var(--code-fg); border:1px solid var(--rule);\n  border-left:4px solid var(--navy);\n  padding:1.15rem 1.3rem; border-radius:6px; overflow-x:auto;\n  margin:0 auto 1.6rem; font-size:.9rem; line-height:1.6;\n}\n.iagen-post pre code{background:none; padding:0; font-size:1em}\n\n.iagen-post .table-wrap{overflow-x:auto; margin:0 auto 1.9rem; -webkit-overflow-scrolling:touch}\n.iagen-post table{border-collapse:collapse; width:100%; min-width:36rem; font-size:.94rem; line-height:1.55}\n.iagen-post th,.iagen-post td{text-align:left; vertical-align:top; padding:.75rem .85rem; border-bottom:1px solid var(--rule)}\n.iagen-post th{\n  font-weight:600; color:var(--ink); font-size:.76rem; letter-spacing:.09em;\n  text-transform:uppercase; border-bottom:2px solid var(--rule-strong); white-space:nowrap;\n}\n.iagen-post tbody tr:hover{background:var(--paper-alt)}\n\n\/* table of contents *\/\n.iagen-post nav.toc{\n  margin:0 auto 1rem; padding:1.6rem 1.9rem;\n  background:var(--paper-alt); border:1px solid var(--rule); border-radius:8px;\n}\n.iagen-post nav.toc h2{\n  font-size:.72rem; letter-spacing:.18em; text-transform:uppercase; font-weight:600;\n  color:var(--muted); margin:0 0 1rem; padding:0;\n}\n.iagen-post nav.toc h2::before{display:none}\n.iagen-post nav.toc ul{\n  list-style:none; margin:0; padding:0; columns:2; column-gap:3rem; font-size:.94rem;\n}\n.iagen-post nav.toc li{margin:0 0 .45rem; padding:0; break-inside:avoid}\n.iagen-post nav.toc a{text-decoration:none; color:var(--body)}\n.iagen-post nav.toc a:hover{color:var(--red); text-decoration:underline}\n@media (max-width:44rem){ .iagen-post nav.toc ul{columns:1} }\n\n\/* figures *\/\n.iagen-post figure.fig{\n  margin:2.4rem auto 2.6rem; padding:1.6rem 1.6rem 1.1rem;\n  background:var(--fig-bg); border:1px solid var(--rule); border-radius:8px;\n}\n.iagen-post figure.fig svg{display:block; width:100%; height:auto; overflow:visible}\n.iagen-post figure.fig figcaption{\n  margin-top:1.1rem; padding-top:.9rem; border-top:1px solid var(--rule);\n  font-size:.86rem; line-height:1.55; color:var(--muted);\n}\n\/* svg primitives *\/\n.iagen-post .s-box{fill:var(--fig-soft); stroke:var(--fig-line); stroke-width:1.25}\n.iagen-post .s-card{fill:var(--paper); stroke:var(--fig-line); stroke-width:1.25}\n.iagen-post .s-ring{fill:none; stroke:var(--fig-line); stroke-width:1.25}\n.iagen-post .s-ring.r5{stroke-dasharray:5 4}\n.iagen-post .s-solid{fill:var(--navy)}\n.iagen-post .s-accent{fill:var(--red)}\n.iagen-post .s-arrow{fill:none; stroke:var(--fig-line); stroke-width:1.5}\n.iagen-post .s-arrow-head{fill:var(--fig-line)}\n.iagen-post .s-axis{fill:none; stroke:var(--fig-line); stroke-width:1.25}\n.iagen-post .s-curve{fill:none; stroke:var(--red); stroke-width:2.5; stroke-linecap:round}\n.iagen-post .s-dot{fill:var(--red)}\n.iagen-post .s-dot-mut{fill:var(--fig-line)}\n.iagen-post .s-sep{stroke:var(--rule); stroke-width:1; stroke-dasharray:4 4}\n.iagen-post .t-big{font-size:20px; font-weight:700; fill:var(--ink); letter-spacing:-.02em}\n.iagen-post .t-lbl{font-size:12px; font-weight:600; fill:var(--ink); letter-spacing:.13em}\n.iagen-post .t-sm{font-size:14px; font-weight:500; fill:var(--ink)}\n.iagen-post .t-xs{font-size:12px; font-weight:400; fill:var(--muted)}\n.iagen-post .t-mut{fill:var(--muted)}\n.iagen-post .t-acc{fill:var(--red)}\n.iagen-post .t-inv{fill:#fff}\n.iagen-post .t-inv-mut{fill:rgba(255,255,255,.78)}\n\n\/* stat tiles *\/\n.iagen-post .stats{\n  display:grid; grid-template-columns:repeat(auto-fit,minmax(11rem,1fr));\n  gap:1px; background:var(--rule); border:1px solid var(--rule);\n  border-radius:8px; overflow:hidden; margin:2.2rem auto 2.4rem;\n}\n.iagen-post .stat{background:var(--paper-alt); padding:1.3rem 1.25rem}\n.iagen-post .stat b{\n  display:block; font-size:1.85rem; line-height:1.1; font-weight:700;\n  letter-spacing:-.03em; color:var(--red); margin-bottom:.4rem;\n}\n.iagen-post .stat span{display:block; font-size:.85rem; line-height:1.45; color:var(--muted)}\n\n\/* footer \/ credits *\/\n.iagen-post .iagen-footer{\n  margin-top:4rem; padding:2rem 0 .5rem; border-top:3px solid var(--red);\n}\n.iagen-post .iagen-footer p{margin:0 auto .8rem; font-size:.9rem; color:var(--muted)}\n.iagen-post .iagen-footer .credit strong{color:var(--body); font-weight:600}\n.iagen-post .iagen-footer .copyright{\n  font-size:.8rem; letter-spacing:.04em; padding-top:.9rem; border-top:1px solid var(--rule);\n}\n\n@media (max-width:40rem){\n  .iagen-post{font-size:16.5px; padding:1rem}\n  .iagen-post figure.fig{padding:1.1rem 1rem .9rem}\n  .iagen-post h2{margin-top:2.8rem}\n}\n<\/style>\n\n<div class=\"iagen-post\">\n  <p>Over the past year the way we build software internally has changed shape. Code is largely produced by agents, and what a developer contributes has moved somewhere else. Along the way we built an internal framework, <strong>iagen-dev<\/strong>, to make that shift repeatable instead of dependent on who happens to be driving.<\/p>\n  <p>This article is the reasoning behind it. Not the installation procedure, not our internal tooling: the <em>why<\/em>. Most of it should transfer to any organization trying to do this seriously.<\/p>\n  <hr class=\"intro-rule\">\n\n  <nav class=\"toc\" aria-label=\"Table of contents\">\n    <h2>Contents<\/h2>\n    <ul>\n      <li><a href=\"#the-posture-shift\">1. The posture shift<\/a><\/li>\n      <li><a href=\"#the-founding-assumption-safety-engineering\">2. The finding assumption: safety engineering<\/a><\/li>\n      <li><a href=\"#why-a-framework-at-all\">3. Why a framework at all<\/a><\/li>\n      <li><a href=\"#specifying-is-coding\">4. Specifying is coding<\/a><\/li>\n      <li><a href=\"#the-documentation-model\">5. The documentation model<\/a><\/li>\n      <li><a href=\"#low-entropy-engineering-why-the-technical-defaults-look-like-this\">6. Low-entropy engineering: why the technical defaults look like this<\/a><\/li>\n      <li><a href=\"#scaling-the-constraint-tiers\">7. Scaling the constraint: third party<\/a><\/li>\n      <li><a href=\"#multi-agent-review\">8. Multi-agent review<\/a><\/li>\n      <li><a href=\"#what-actually-goes-wrong\">9. What actually goes wrong<\/a><\/li>\n      <li><a href=\"#domain-experts-come-inside-the-loop\">10. Domain experts come inside the loop<\/a><\/li>\n      <li><a href=\"#adopting-this-in-a-team\">11. Adopting this in a team<\/a><\/li>\n      <li><a href=\"#what-is-not-yet-solved\">12. What is not (yet) solved<\/a><\/li>\n      <li><a href=\"#sources\">Sources<\/a><\/li>\n    <\/ul>\n  <\/nav>\n\n  <section id=\"the-posture-shift\">\n<h2>1. The posture shift<\/h2>\n<p>This is a change in the job itself, and it is not optional. Your trade is no longer writing code. It is:<\/p>\n<ul>\n<li><strong>Specifying<\/strong>: producing a good, stable, testable statement of the need.<\/li>\n<li><strong>Framing the agents<\/strong>: rules, context, the right task at the right size, in the right model.<\/li>\n<li><strong>Putting verification and guardrails in place<\/strong>: tests, linters, CI, reviews. You build the harness, the agents work inside it.<\/li>\n<li><strong>Running the project<\/strong>: sequencing, arbitrating, keeping scope under control.<\/li>\n<li><strong>Holding a product view of your own project<\/strong>: what it is for, who benefits, what you deliberately will not build, when you stop.<\/li>\n<li><strong>Owning the whole lifecycle<\/strong>: deployment, monitoring, sustainment, end of life.<\/li>\n<\/ul>\n<p>Your unit of value is no longer source code. It is a good, stable, testable specification that a team of agents implements. You express the need, you review the spec, then you verify functionally, on end-to-end behavior and real runs, not line by line.<\/p>\n<p>Nobody is demoted by this. The work moves from writing syntax to designing constraints, arbitrating trade-offs and auditing systems. That is a harder job than the one it replaces.<\/p>\n<p>Two consequences people consistently underestimate. First, the output is ordinary software: compiled Go, a Vue\/Nuxt frontend, built, tested, deployed on Kubernetes. No LLM at runtime, no magic. What changed is the production method, not the artifact. Second, the bottleneck moves to you. When agents struggle it is almost never executed, it is definition. \u201cThe problem is behind the keyboard\u201d is the most frequent root cause we see.<\/p>\n<figure class=\"fig\" data-fig=\"posture\">\n  <svg viewbox=\"0 0 820 330\" role=\"img\" aria-label=\"Before: most effort spent writing code. Now: effort spent specifying, framing, harnessing and verifying, with code generated.\">\n    <text class=\"t-lbl\" x=\"30\" y=\"30\">BEFORE<\/text>\n    <text class=\"t-lbl t-acc\" x=\"470\" y=\"30\">NOW<\/text>\n    <rect class=\"s-box\" x=\"30\" y=\"48\" width=\"300\" height=\"40\" rx=\"6\"><\/rect>\n    <text class=\"t-sm\" x=\"180\" y=\"73\" text-anchor=\"middle\">Requirements, half-written<\/text>\n    <rect class=\"s-box\" x=\"30\" y=\"96\" width=\"300\" height=\"40\" rx=\"6\"><\/rect>\n    <text class=\"t-sm\" x=\"180\" y=\"121\" text-anchor=\"middle\">Design, mostly in your head<\/text>\n    <rect class=\"s-solid\" x=\"30\" y=\"144\" width=\"300\" height=\"150\" rx=\"6\"><\/rect>\n    <text class=\"t-big t-inv\" x=\"180\" y=\"212\" text-anchor=\"middle\">WRITING CODE<\/text>\n    <text class=\"t-sm t-inv\" x=\"180\" y=\"240\" text-anchor=\"middle\">where the hours went<\/text>\n    <path class=\"s-arrow\" d=\"M362 175 H438\"><\/path>\n    <path class=\"s-arrow-head\" d=\"M438 175 l-11 -6 v12 z\"><\/path>\n    <text class=\"t-xs t-mut\" x=\"400\" y=\"162\" text-anchor=\"middle\">the work<\/text>\n    <text class=\"t-xs t-mut\" x=\"400\" y=\"200\" text-anchor=\"middle\">moves up<\/text>\n    <rect class=\"s-accent\" x=\"470\" y=\"48\" width=\"320\" height=\"180\" rx=\"6\"><\/rect>\n    <text class=\"t-big t-inv\" x=\"630\" y=\"86\" text-anchor=\"middle\">SPECIFYING<\/text>\n    <text class=\"t-sm t-inv\" x=\"630\" y=\"116\" text-anchor=\"middle\">framing the agents<\/text>\n    <text class=\"t-sm t-inv\" x=\"630\" y=\"141\" text-anchor=\"middle\">building the harness<\/text>\n    <text class=\"t-sm t-inv\" x=\"630\" y=\"166\" text-anchor=\"middle\">verifying functionally<\/text>\n    <text class=\"t-sm t-inv\" x=\"630\" y=\"191\" text-anchor=\"middle\">owning the lifecycle<\/text>\n    <text class=\"t-xs t-inv-mut\" x=\"630\" y=\"216\" text-anchor=\"middle\">designing constraints, arbitrating trade-offs<\/text>\n    <rect class=\"s-box\" x=\"470\" y=\"240\" width=\"320\" height=\"54\" rx=\"6\"><\/rect>\n    <text class=\"t-sm\" x=\"630\" y=\"264\" text-anchor=\"middle\">Code, generated<\/text>\n    <text class=\"t-xs t-mut\" x=\"630\" y=\"284\" text-anchor=\"middle\">ordinary software: compiled, tested, deployed<\/text>\n  <\/svg>\n  <figcaption>Where the value moves. What you are judged on stops being source code, and becomes the specification and the harness around it.<\/figcaption>\n<\/figure>\n  <\/section>\n\n  <section id=\"the-founding-assumption-safety-engineering\">\n<h2>2. The finding assumption: safety engineering<\/h2>\n<p>The framework borrows its structure from safety engineering, as practiced in avionics, automotive and medical software. Its founding assumption is one sentence:<\/p>\n<blockquote>\n<p>The entity producing the code is not reliable. It makes mistakes. So you put a harness around it.<\/p>\n<\/blockquote>\n<p>That was already true of humans. It is true of agents. Well harnessed, an unreliable producer yields a reliable result. This is not a new idea, it is why aviation software is trustworthy despite being written by people who have bad days.<\/p>\n<p>Everything in iagen-dev is harness: linters, compiled languages, tests, CI, per-project rule files, multi-agent review. Nothing in it assumes the agent is good. It assumes the agent is productive and fallible, and builds the checks accordingly.<\/p>\n<figure class=\"fig\" data-fig=\"harness\">\n  <svg viewbox=\"0 0 820 350\" role=\"img\" aria-label=\"Concentric layers of guardrails around a fallible producer: types and compiler, zero-diff linting, tests and TDD, CI gates, cross-model review.\">\n    <rect class=\"s-ring r5\" x=\"20\" y=\"20\" width=\"780\" height=\"230\" rx=\"10\"><\/rect>\n    <rect class=\"s-ring r4\" x=\"72\" y=\"52\" width=\"676\" height=\"166\" rx=\"9\"><\/rect>\n    <rect class=\"s-ring r3\" x=\"124\" y=\"84\" width=\"572\" height=\"102\" rx=\"8\"><\/rect>\n    <rect class=\"s-ring r2\" x=\"176\" y=\"112\" width=\"468\" height=\"56\" rx=\"7\"><\/rect>\n    <text class=\"t-xs t-mut\" x=\"36\" y=\"40\">CROSS-MODEL REVIEW<\/text>\n    <text class=\"t-xs t-mut\" x=\"88\" y=\"72\">CI GATES: the last resort, not the discovery<\/text>\n    <text class=\"t-xs t-mut\" x=\"140\" y=\"104\">TESTS \/ TDD: you define the contract<\/text>\n    <text class=\"t-xs t-mut\" x=\"192\" y=\"130\">ZERO-DIFF LINTING<\/text>\n    <rect class=\"s-solid\" x=\"252\" y=\"136\" width=\"316\" height=\"26\" rx=\"5\"><\/rect>\n    <text class=\"t-sm t-inv\" x=\"410\" y=\"154\" text-anchor=\"middle\">TYPES &amp; COMPILE<\/text>\n    <text class=\"t-lbl t-acc\" x=\"410\" y=\"292\" text-anchor=\"middle\">THE PRODUCER: HUMAN OR AGENT<\/text>\n    <text class=\"t-sm t-mut\" x=\"410\" y=\"316\" text-anchor=\"middle\">assumed unreliable. It makes mistakes.<\/text>\n    <text class=\"t-sm t-mut\" x=\"410\" y=\"338\" text-anchor=\"middle\">Well-harnessed, an unreliable producer yields a reliable result.<\/text>\n  <\/svg>\n  <figcaption>The harness. Nothing in it assumes the producer is good. It assumes the producer is productive and fallible.<\/figcaption>\n<\/figure>\n<p>That framing settles a lot of arguments before they start. \u201cShould we trust AI-generated code?\u201d is the wrong question. You do not trust human-generated code either. You test it, lint it, review it, and run it in an environment that survives its failures. Apply the same discipline, calibrated to a producer that is faster, cheaper, tireless and differently wrong.<\/p>\n<p>It also has a belief that comes back every time a language choice is discussed: that picking a memory-safe language removes the class of problem. It doesn&#039;t. Cloudflare&#039;s outage of 18 November 2025 is a good public example, and <a href=\"https:\/\/blog.cloudflare.com\/18-november-2025-outage\/\">their postmortem<\/a> is worth reading in full. A database permissions change made a query return duplicate columns, the Bot Management configuration file doubled in size, it crossed a hardcoded limit of 200 features against about 60 in use, and the proxy handling it, written in Rust, called <code>unwrap()<\/code> on an error and panicked. Memory safety is not correct.<\/p>\n<p>The harder version of the same point is that a correct implementation of a wrong specification fails just as hard. In 1993 year A320 <a href=\"https:\/\/en.wikipedia.org\/wiki\/Lufthansa_Flight_2904\">landed at Warsaw<\/a> on one gear in a crosswind, on a wet runway. Ground spoilers required 6.3 tons on each main gear or wheels spinning above 72 knots, and aquaplaning meant neither condition was met. The braking logic remained inhibited for about nine seconds and the aircraft ran off the runway, killing two people. The software did exactly what its specification said. The specification&#039;s definition of being on the ground does not survive contact with that landing.<\/p>\n<p>That is why the spec is the artifact that gets reviewed. A harness is only as good as the assumptions encoded in it, and no language checks those for you.<\/p>\n  <\/section>\n\n  <section id=\"why-a-framework-at-all\">\n<h2>3. Why a framework at all<\/h2>\n<p>Developer adoption of AI tooling is near universal while trust is falling. In the <a href=\"https:\/\/survey.stackoverflow.co\/2025\/ai\">Stack Overflow Developer Survey 2025<\/a>, across 49,000 respondents, 84% use or plan to use AI tools, up from 76% the year before. In the same survey 46% do not trust the accuracy of AI output, up from 31% in the 2024 edition, 66% name <em>\u201cAI solutions that are almost right, but not quite\u201d<\/em> as their top frustration, and 72% say vibe coding is not part of their professional work.<\/p>\n<div class=\"stats\">\n  <div class=\"stat\"><b>84%<\/b><span>use or plan to use AI tools, up from 76%<\/span><\/div>\n  <div class=\"stat\"><b>46%<\/b><span>do not trust its accuracy, up from 31%<\/span><\/div>\n  <div class=\"stat\"><b>66%<\/b><span>name \u00abalmost right, but not quite\u00bb as their top frustration<\/span><\/div>\n  <div class=\"stat\"><b>~\u2153<\/b><span>of improvement requests are about understanding the codebase<\/span><\/div>\n<\/div>\n<p>This disappointment is legitimate and well documented. It is also misattributed.<\/p>\n<p>The top improvement developers ask for is better contextual understanding. In Qodo&#039;s <a href=\"https:\/\/www.qodo.ai\/reports\/state-of-ai-code-quality\/\">State of AI Code Quality 2025<\/a>, a third of all improvement requests are about the tool understanding the codebase, the team&#039;s norms and the project structure, and half the developers reporting a context gap work in organizations of ten people or fewer. This is not a big-company problem.<\/p>\n<p>The concrete experience is familiar to anyone who has tried. On an existing service, the AI generates code that compiles but ignores the project&#039;s error patterns, reinvents existing abstractions, reaches for third-party libraries instead of the standard library, and writes tests that test the implementation rather than the behavior. It works but does not integrate, and you spend more time fixing it than you would have spent writing it. The rational conclusion is that AI does not work, when in fact it is the absence of a frame that does not work.<\/p>\n<p>Everything the framework ships (per-project rule files, the documentation model, the lint gates, the project tiers) exists to supply that missing frame. It is distributed as a versioned skill pack that agents fetch and self-update from, so a fix found on one project propagates to the others.<\/p>\n  <\/section>\n\n  <section id=\"specifying-is-coding\">\n<h2>4. Specifying is coding<\/h2>\n<p>The spec is no longer a preliminary document you write to satisfy a process. It is the artifact that determines the result.<\/p>\n<p>This is a shared observation among the people doing it, not a surprise: a spec co-written with an agent, interactively, comes out better than what an experienced developer produces alone. The agent challenges assumptions, surfaces the cases you skipped, and forces you to answer questions you would otherwise have left implicit. Those are the questions that normally get discovered three weeks later, in a demo.<\/p>\n<p>The loop is: deep research and state of the art, then lay out the ideas and the problems, then draft the spec with the agent, then read it back critically. It is long, and a large share of the value sits there.<\/p>\n<p>One prompt does more work than any other. Append it to every feature request:<\/p>\n<pre><code>Before you answer, tell me what you need to know to answer well, and point out any assumptions you&#039;d otherwise make.<\/code><\/pre>\n<p>It converts silent bad assumptions into explicit questions, which is the failure mode that costs weeks.<\/p>\n<p>A practical asymmetry appears once the framework is in place: the WHAT gets constantly challenged, the HOW less and less. The spec determines the functionality and the tests, so that is where the effort belongs, and where a bad decision is expensive because everything downstream inherits it. The architecture gets challenged less over time, not from complacency, but because the rules are already encoded in the framework. With them in place the AI produces clean, testable architectures by default.<\/p>\n<p>That asymmetry is where the framework pays for itself. Writing the rules demands real architecture skills and deep knowledge of the stacks, but they are written once and everyone benefits from them on every project. Individual developers stop re-litigating layering, error handling and testability on every feature, and spend their attention on the requirement instead.<\/p>\n  <\/section>\n\n  <section id=\"the-documentation-model\">\n<h2>5. The documentation model<\/h2>\n<p>This is the core of agentic development and the biggest single lever on result quality. Four document types, split by topic, with a global index, a per-folder index and a code map.<\/p>\n<div class=\"table-wrap\"><table>\n<colgroup><col style=\"width: 33%\"><col style=\"width: 33%\"><col style=\"width: 33%\"><\/colgroup>\n<thead><tr><th>Document<\/th><th>Answers<\/th><th>Thrilled<\/th><\/tr><\/thead>\n<tbody>\n<tr><td><strong>Spec<\/strong><\/td><td>the WHAT<\/td><td>Requirements. Source of truth. Only the WHAT, never the HOW. Ideally testable<\/td><\/tr>\n<tr><td><strong>Design<\/strong><\/td><td>the HOW<\/td><td>Implementation, chosen patterns, design system, stacks<\/td><\/tr>\n<tr><td><strong>ADR<\/strong><\/td><td>the WHY<\/td><td>Why a decision was made, under which constraints, which limitations were knowingly accepted<\/td><\/tr>\n<tr><td><strong>Runbook<\/strong><\/td><td>the HOW TO OPERATE<\/td><td>Procedures around the project, not the code: deploy, back up, recover<\/td><\/tr>\n<\/tbody>\n<\/table><\/div>\n<figure class=\"fig\" data-fig=\"docs\">\n  <svg viewbox=\"0 0 820 380\" role=\"img\" aria-label=\"Four document types (spec, design, ADR, runbook) feeding an index and code map, with transient plans deleted.\">\n    <rect class=\"s-card\" x=\"20\" y=\"20\" width=\"380\" height=\"104\" rx=\"8\"><\/rect>\n    <text class=\"t-lbl t-acc\" x=\"42\" y=\"48\">SPEC<\/text>\n    <text class=\"t-big\" x=\"378\" y=\"50\" text-anchor=\"end\">WHAT<\/text>\n    <text class=\"t-sm t-mut\" x=\"42\" y=\"76\">Requirements. Source of truth.<\/text>\n    <text class=\"t-sm t-mut\" x=\"42\" y=\"98\">Current state only, no history. Ideally testable.<\/text>\n    <rect class=\"s-card\" x=\"420\" y=\"20\" width=\"380\" height=\"104\" rx=\"8\"><\/rect>\n    <text class=\"t-lbl t-acc\" x=\"442\" y=\"48\">DESIGN<\/text>\n    <text class=\"t-big\" x=\"778\" y=\"50\" text-anchor=\"end\">HOW<\/text>\n    <text class=\"t-sm t-mut\" x=\"442\" y=\"76\">Implementation, chosen patterns,<\/text>\n    <text class=\"t-sm t-mut\" x=\"442\" y=\"98\">design system, stacks.<\/text>\n    <rect class=\"s-card\" x=\"20\" y=\"140\" width=\"380\" height=\"104\" rx=\"8\"><\/rect>\n    <text class=\"t-lbl t-acc\" x=\"42\" y=\"168\">ADR<\/text>\n    <text class=\"t-big\" x=\"378\" y=\"170\" text-anchor=\"end\">WHY<\/text>\n    <text class=\"t-sm t-mut\" x=\"42\" y=\"196\">Decisions, their constraints, and the<\/text>\n    <text class=\"t-sm t-mut\" x=\"42\" y=\"218\">limitations knowingly accepted. Immutable.<\/text>\n    <rect class=\"s-card\" x=\"420\" y=\"140\" width=\"380\" height=\"104\" rx=\"8\"><\/rect>\n    <text class=\"t-lbl t-acc\" x=\"442\" y=\"168\">RUNBOOK<\/text>\n    <text class=\"t-big\" x=\"778\" y=\"170\" text-anchor=\"end\">OPERATE<\/text>\n    <text class=\"t-sm t-mut\" x=\"442\" y=\"196\">Procedures around the project, not the<\/text>\n    <text class=\"t-sm t-mut\" x=\"442\" y=\"218\">code: deploy, back up, recover.<\/text>\n    <path class=\"s-arrow\" d=\"M210 252 V282\"><\/path>\n    <path class=\"s-arrow\" d=\"M610 252 V282\"><\/path>\n    <path class=\"s-arrow-head\" d=\"M210 284 l-6 -11 h12 z\"><\/path>\n    <path class=\"s-arrow-head\" d=\"M610 284 l-6 -11 h12 z\"><\/path>\n    <rect class=\"s-solid\" x=\"20\" y=\"288\" width=\"780\" height=\"40\" rx=\"6\"><\/rect>\n    <text class=\"t-sm t-inv\" x=\"410\" y=\"313\" text-anchor=\"middle\">GLOBAL INDEX + PER-FOLDER INDEX + MAP CODE<\/text>\n    <text class=\"t-xs t-mut\" x=\"410\" y=\"352\" text-anchor=\"middle\">Every fresh agent starts here, and gets a short exploration phase instead of re-walking the tree.<\/text>\n    <text class=\"t-xs t-acc\" x=\"410\" y=\"370\" text-anchor=\"middle\">Transient specs and plans live outside this model, and are deleted once the task is done.<\/text>\n  <\/svg>\n  <figcaption>Four document types, one index. The split is what makes starting every task from a clean session affordable.<\/figcaption>\n<\/figure>\n<p>The split is not bureaucracy. Each document answers a different question, and mixing them is what produces documentation nobody maintains:<\/p>\n<ul>\n<li>Merge WHAT and HOW, and every implementation change invalidates the requirements.<\/li>\n<li>Lose the WHY, and six months later someone \u201csimplifies\u201d a constraint that existed for a reason. An agent will do it in an afternoon.<\/li>\n<li>Keep transient plans around, and the next agent reads a stale plan as current state.<\/li>\n<li>Let change history creep into the spec, and the agent can no longer tell what is true now from what was true before. The spec carries the current state only: no history, no \u201cpreviously we did X\u201d. History belongs in ADRs and in Git.<\/li>\n<\/ul>\n<h3 id=\"why-the-indexes-matter\">Why the indexes matter<\/h3>\n<p>The goal is to start every task from a clean session. That only pays off if a fresh agent can understand the project without burning its context rediscovering it. Index, code map and minimal docs give a short exploration phase for every new agent. Without them each fresh agent re-explores the tree, and you have paid the reset cost without getting the benefit.<\/p>\n<p>Two rules go with it. No transient documents in the repository&#039;s docs: the iterative specs and plans produced during a task are verbose and bound to it, so delete them once it is done. The implementation plan in particular is never re-read and misleads more than it helps. And docs are written for agents first. They do not have to read beautifully for humans. If you have a question, ask the agent rather than digging.<\/p>\n<h3 id=\"the-context-failure-mode\">The context failure mode<\/h3>\n<p>The central failure mode of agentic development: when the agent&#039;s context fills up, it goes off the rails. Three things happen, in increasing order of damage.<\/p>\n<p>The context burps. As the window grows, the early tokens, which is exactly where the rules were loaded, carry less and less weight against thousands of tokens of recent output. Agents do compress and summarize, but they re-read their original instructions poorly, and compaction drops constraints first because constraints look like boilerplate.<\/p>\n<p>Then it starts inventing. Gaps get filled with plausible reconstructions instead of checked facts, which is the visible symptom everyone complains about.<\/p>\n<p>Worst, because it is silent, it stops applying the constraints the framework exists to enforce. The branch it was told to work on, the linter it was told to re-run, the document it was told to update. Nothing announces this. The work simply comes back subtly off policy, and you find out at review time or later.<\/p>\n<p>That last point is the real argument for giving every agent a clean context, and it is not about saving tokens. In a small context the rules hold a large share of the model&#039;s attention. In a saturated one they are a rounding error. A sub-agent that receives only its task and the rules that bind it will respect them. The same instruction, given at hour three of a filled session, often will not.<\/p>\n<p>Countermeasures, in order of impact:<\/p>\n<ul>\n<li>One task, one session. Reset context between tasks.<\/li>\n<li>Sub-agents with clean contexts. When the agent offers to execute in sub-agents, say yes almost always. The main agent drives (instructs, verifies), sub-agents do the work in isolated contexts. The exception is small, tightly coupled edits, where inline keeps coherence better.<\/li>\n<li>Even then the main agent eventually pollutes itself and drives worse. Start a fresh session.<\/li>\n<\/ul>\n<p>A project whose documentation is not organized this way makes all of it worse: the agent burns context rediscovering the codebase, so saturation arrives sooner and the rules go first. What looks like a hallucinating model is usually a saturated one, and that is preventable.<\/p>\n<p>Here is the execution prompt we use:<\/p>\n<pre><code>Execute the whole plan in sub-agents with clear context. Use sub-agents adapted to the difficulty of the task. You, the main agent, are responsible for the correct execution and verification of all tasks and sub-tasks. There is no need to parallelize, execute sub-agents sequentially in your recommended order. Do not blindly acknowledge a sub-agent&#039;s work, every work must be challenged.<\/code><\/pre>\n<p>Sub-agents rather than inline buys isolated contexts. Sequential rather than parallel matters more than it looks: left alone, the agent fans everything out at once, which burns context faster, triggers provider overload errors much more often, and makes recovery after a quota exhaustion painful. Resuming one agent is simple, making sure ten restart correctly is not. And challenging every deliverable is not optional. Never settle for a sub-agent&#039;s own report of its own success.<\/p>\n  <\/section>\n\n  <section id=\"low-entropy-engineering-why-the-technical-defaults-look-like-this\">\n<h2>6. Low-entropy engineering: why the technical defaults look like this<\/h2>\n<p>Nothing in the framework is arbitrary, and the reasoning matters more than the rule. Once you know why a default exists, you know when it legitimately does not apply. Every default can be deviated from, with a written justification in the project spec.<\/p>\n<h3 id=\"language-models-work-better-in-low-entropy-environments\">Language models work better in low-entropy environments<\/h3>\n<p>Our backend, CLI and tooling default is Go. Its advantage is not richness, it is its deliberately low abstraction ceiling: fewer syntactic choices, one idiomatic way to do each thing, an exhaustive standard library, rigid static typing, a single concurrency model. Generated code follows the same structural patterns every time.<\/p>\n<p>In high-optionality environments the model faces a large combinatorial space: dozens of competing frameworks, disparate typing approaches, fragmented utility libraries. That excess optionality is what makes agents hallucinate hybrid architectures mixing incompatible paradigms.<\/p>\n<figure class=\"fig\" data-fig=\"entropy\">\n  <svg viewbox=\"0 0 820 300\" role=\"img\" aria-label=\"Agent reliability falls as language optionality rises: Go and strict TypeScript on the low-entropy end, unconstrained Python and JavaScript on the high-entropy end.\">\n    <text class=\"t-xs t-mut\" x=\"20\" y=\"26\">AGENT RELIABILITY<\/text>\n    <path class=\"s-axis\" d=\"M60 40 V236 H790\"><\/path>\n    <path class=\"s-curve\" d=\"M60 62 C 240 74, 330 118, 420 152 S 640 214, 790 226\"><\/path>\n    <circle class=\"s-dot\" cx=\"92\" cy=\"66\" r=\"6\"><\/circle>\n    <circle class=\"s-dot\" cx=\"240\" cy=\"96\" r=\"6\"><\/circle>\n    <circle class=\"s-dot\" cx=\"430\" cy=\"157\" r=\"6\"><\/circle>\n    <circle class=\"s-dot-mut\" cx=\"600\" cy=\"200\" r=\"6\"><\/circle>\n    <circle class=\"s-dot-mut\" cx=\"742\" cy=\"222\" r=\"6\"><\/circle>\n    <text class=\"t-sm\" x=\"92\" y=\"52\" text-anchor=\"middle\">Go<\/text>\n    <text class=\"t-sm\" x=\"240\" y=\"82\" text-anchor=\"middle\">TS strict<\/text>\n    <text class=\"t-sm\" x=\"430\" y=\"143\" text-anchor=\"middle\">Python + strict typing<\/text>\n    <text class=\"t-sm t-mut\" x=\"600\" y=\"186\" text-anchor=\"middle\">Python, untyped<\/text>\n    <text class=\"t-sm t-mut\" x=\"742\" y=\"208\" text-anchor=\"middle\">JS, no gate<\/text>\n    <text class=\"t-xs t-mut\" x=\"60\" y=\"258\">LOW OPTIONALITY<\/text>\n    <text class=\"t-xs t-mut\" x=\"790\" y=\"258\" text-anchor=\"end\">HIGH OPTIONALITY<\/text>\n    <text class=\"t-xs t-mut\" x=\"60\" y=\"276\">one idiomatic way \u00b7 compiler catches it<\/text>\n    <text class=\"t-xs t-mut\" x=\"790\" y=\"276\" text-anchor=\"end\">competing frameworks \u00b7 hallucinated hybrid architectures<\/text>\n  <\/svg>\n  <figcaption>Agent reliability tracks how few ways there are to express the same thing.<\/figcaption>\n<\/figure>\n<p>Three properties matter, in this order:<\/p>\n<ol type=\"1\">\n<li>Tooling to constrain the agent: linting, static analysis, vulnerability scanning, build. More harness available.<\/li>\n<li>Compilation: a large share of agent mistakes explode at compile time instead of at runtime.<\/li>\n<li>Low entropy: explicit beats clever when a machine is writing it.<\/li>\n<\/ol>\n<p>An honest nuance. The claim that Go has had no breaking structural change in a decade needs tempering. Generics were a paradigm shift that forked practice into pre- and post-generics code and partially fragmented the training corpus.<\/p>\n<h3 id=\"why-not-rust\">Why not Rust<\/h3>\n<p>The question comes up constantly, so here is the full answer rather than a dismissal.<\/p>\n<p>Compilation time is the first reason, and it matters more here than it would elsewhere. An agent recompiles dozens of times within a single task, so build time does not add to the cycle, it multiplies. Fast iteration is most of the game with agents, and Rust is where that loop gets slow.<\/p>\n<p>The second reason is that nobody here could audit the result. We have no Rust practice and little low-level practice generally. In an agentic setting that is not a statement about what we could write, it is a statement about what we could read: the agent produces the code, and a human has to be able to open it when something goes wrong at three in the morning. That human is the last link in the harness. Adding a second compiled language also means a second CI chain, a second review pool and a second sustainment perimeter. This is also why the earlier example of a team shipping components without reading the code is not a contradiction: that was an internal benchmarking tool, not a service with customers behind it, and the tier decides how much unread code is acceptable.<\/p>\n<p>Third, our workloads do not ask for it. These are API services on Kubernetes, bound by I\/O and by the database, at moderate throughput. Rust&#039;s real advantages are predictable latency without a garbage collector, tight memory budgets, and parsing untrusted input. Those advantages exist, they are simply not what constrains us. Where they would constrain us, a parser exposed to hostile input, a hot path genuinely limited by CPU, a component with a hard memory ceiling, Rust is the right answer and the framework says so.<\/p>\n<p>Fourth, on our own criterion, Rust is the higher-optionality environment. Traits, macros, generics and a choice of async runtime give more ways to express the same thing than Go does, which is the property we deliberately optimized against.<\/p>\n<p>The serious counter-argument deserves stating. Rust&#039;s compiler is the strictest harness on the market, and this article argues that types and compilers are harness, so Rust should follow. Our experience is that the compiler does catch more, but the cost per attempt is higher, and borrow checker and lifetime errors are where we see agents flail, changing signatures until something compiles rather than understanding the constraint. That is an observation from our practice, not a law, and it is the argument most likely to change.<\/p>\n<p>Last and least, generated Rust suffers from ecosystem churn more than from corpus size: competing async runtimes and successive generations of error-handling idioms mean models mix conventions that do not belong together.<\/p>\n<p>None of this makes Rust a bad language. It makes it the wrong default for what we build.<\/p>\n<p>Python is restricted, not banned. Go has no mature ML ecosystem, and data engineering, ML pipelines and scientific processing legitimately live in Python. The rule is: if it can be done in Go, do it in Go. When Python is approved, deterministic lockfiles and reproducible environments are mandatory, and strict typing is required. Strict typing claws back some of the entropy that made Python the weaker choice for generation in the first place.<\/p>\n<p>On the frontend, TypeScript with strict mode mandatory, for the same reason Go wins on the backend: types are a harness that catches agent mistakes before runtime.<\/p>\n<h3 id=\"a-formal-contract-between-backend-and-frontend\">A formal contract between backend and frontend<\/h3>\n<p>Between a Go service and its frontend, the contract is an OpenAPI specification acting as single source of truth, with generated types on both sides. The backend changes an endpoint, the spec is regenerated, a frontend CI job re-runs the generator and blocks the merge on an uncommitted diff, and components using a renamed field fail to compile before anything deploys. This kills an entire bug class, contract drift from hand-copied types, which is the kind of mistake an agent makes silently and repeatedly.<\/p>\n<h3 id=\"crash-early-let-the-orchestrator-handle-it\">Crash early, let the orchestrator handle it<\/h3>\n<p>The default error model is <em>let it crash<\/em>, borrowed from Erlang\/OTP and adapted to Kubernetes. Transient errors get one to three local retries with backoff. Past the limit, log an error and exit non-zero. Structural errors (missing configuration, absent dependency, inconsistent state) crash immediately, because the problem will not resolve itself.<\/p>\n<p>Why crash rather than cope? Kubernetes is the recovery orchestrator, with restart policies, backoff, probes and rescheduling. Reimplementing that in application code produces complex, hard-to-test, frequently buggy logic that does the orchestrator&#039;s job worse. A service that crashes cleanly with a clear message is more reliable than one that survives in a degraded state at all costs.<\/p>\n<p>This matters doubly with agents. \u201cHandle every error gracefully\u201d is the kind of instruction that makes a model generate elaborate defensive machinery nobody asked for.<\/p>\n<h3 id=\"zero-diff-linting-or-linting-against-the-machine\">Zero-diff linting, or linting against the machine<\/h3>\n<p>AI-generated code has different anti-patterns from human code: excessive defensive code, orphan imports, unsolicited recreation of existing abstractions, and security flaws hidden under syntactically perfect surfaces. Code-quality analyzes put readability problems around 3 times more frequent and formatting problems around 2.66 times more frequent than in human-written code.<\/p>\n<p>Two decisions follow. The linter belongs inside the agent&#039;s loop, with an explicit instruction not to stop until the checks pass, so a failing linter is redirected to the agent as error feedback and forces self-remediation. And zero diff is allowed on formatting and static rules. The point is not style, it is removing dead code, capping complexity, catching obvious bugs and filtering security issues before a human review. CI is the last-resort guardrail, not the discovery mechanism.<\/p>\n<h3 id=\"tdd-and-dependency-injection-as-control-mechanisms\">TDD and dependency injection as control mechanisms<\/h3>\n<p>TDD inverts the model&#039;s natural failure mode. Unconstrained, generative models produce defensive, over-architected code with useless imports and abstractions anticipating improbable scenarios. Requiring a failing test before any implementation forces the agent to satisfy one verifiable constraint. It cannot wander into unrequested features.<\/p>\n<p>TDD also reduces comprehension debt, because the tests are living, executable documentation of intended behavior. And it is the best answer to \u201cI don&#039;t trust AI code\u201d: you define the contract, the AI implements it, the tests verify. If the AI produces bad code, tests fail immediately, and there is no need to read 200 lines hunting for the bug.<\/p>\n<p>Dependency injection is what makes that possible, so it stops being merely good design and becomes a control mechanism. It buys explicit interface contracts, since a constructor signature reveals the component&#039;s requirements and the agent cannot hide coupling inside business logic. It buys interface segregation, which has to be enforced because LLMs optimizes for generation fluency and drift toward monoliths. And it buys strict side-effect isolation, with network and database calls pushed to the boundaries and injected, so the agent iterates on pure logic without touching external state.<\/p>\n<p>Test observable behavior, never internal implementation details, otherwise every minor refactor invalidates dozens of tests.<\/p>\n<h3 id=\"spec-and-tests-together\">Spec and tests together<\/h3>\n<p>Specification-driven development alone shows real weaknesses on existing codebases. Each intermediate textual step amplifies hallucination, so a decision validated during requirements gets quietly altered by the model at design time, because there is no strict behavioral coupling between spec and generated code. TDD alone has the opposite gap: it verifies behavior rigorously but cannot encode business intent, non-functional constraints or architectural decisions.<\/p>\n<p>The opposition is a false dilemma. Textual specs carry the WHAT and the WHY, automated tests encode the verifiable HOW. That combination prevents spec drift while preserving traceability, and it is why the documentation model separates spec from design from ADR in the first place.<\/p>\n<h3 id=\"the-workflow-adapts-to-the-task\">The workflow adapts to the task<\/h3>\n<p>The canonical loop is a backbone, not a liturgy. Bug fixes skip product shaping and architecture review entirely, and they are the one category mature enough today for near-autonomous ticket-to-agent execution. Backend features get the full loop, because they are testable end to end. The judgment call stays with you: applying the heaviest workflow everywhere burns tokens and goodwill, applying the lightest one to a production backend feature is how you ship a mess.<\/p>\n<p>UI work is the clearest case for dropping the process, and it earns its own mode. Running the full spec and TDD treatment to move a button is a tank for a nail: the agent writes a test, fails it, changes the color, greens the test, runs every linter, and then you look at it and want a different color. That is a lot of tokens for nothing. Worse, UX is much harder to describe to an agent than code, so it usually lands off target and the only loop that works is seeing the rendering. So we switch the process off explicitly:<\/p>\n<pre><code>Switch to UI\/UX editing. Create a branch, commit all edits but do not push, and do not run checks or tests until I say that we have finished UX editing.<\/code><\/pre>\n<p>Then the frontend runs with hot reload, changes are visible live, and you iterate by looking rather than by specifying. When it is right, switch the process back on:<\/p>\n<pre><code>Stop UI\/UX editing; make local checks and tests. If OK, then squash all commits from this UX\/UI branch while keeping a detailed commit message, and push.<\/code><\/pre>\n<p>The checks and the tests still run, and the branch still lands as one reviewable commit. They just stop running dozens of times while you are deciding what you want. This is worth transposing to any problem where you only know what you want once you see it, which is exactly the shape of the dashboard difficulty below.<\/p>\n  <\/section>\n\n  <section id=\"scaling-the-constraint-tiers\">\n<h2>7. Scaling the constraint: third party<\/h2>\n<p>A single standard is either too heavy for a personal script or too light for client production. So projects declare a tier, and the constraint scales with the blast radius.<\/p>\n<div class=\"table-wrap\"><table>\n<colgroup><col style=\"width: 33%\"><col style=\"width: 33%\"><col style=\"width: 33%\"><\/colgroup>\n<thead><tr><th>Third<\/th><th>Nature<\/th><th>Constraints<\/th><\/tr><\/thead>\n<tbody>\n<tr><td><strong>A<\/strong><\/td><td>Personal project<\/td><td>Light rule file, few sharing constraints. Optimized for iteration speed<\/td><\/tr>\n<tr><td><strong>B<\/strong><\/td><td>Shared, internal audience<\/td><td>Intermediate. Documentation and changelog expected<\/td><\/tr>\n<tr><td><strong>C<\/strong><\/td><td>Shared and going to production<\/td><td>Heavy: monitoring, log management, failure handling, dashboards, SLA<\/td><\/tr>\n<\/tbody>\n<\/table><\/div>\n<p>Pick honestly. Tier C on a throwaway script wastes your week, tier A on something that reaches a client is how incidents happen. The cost of tier C is real, and paying it on a disposable tool is as much a mistake as skipping it on something a customer depends on.<\/p>\n  <\/section>\n\n  <section id=\"multi-agent-review\">\n<h2>8. Multi-agent review<\/h2>\n<p>This is the quality practice with the best return, and the one that catches the rules your main agent forgot.<\/p>\n<p>The protocol:<\/p>\n<ol type=\"1\">\n<li>The driving agent asks his questions and produces a review plan plus three independent prompts.<\/li>\n<li>Three reviews run in clean, mutually blind contexts, written to separate files. No reviewer sees another&#039;s output.<\/li>\n<li>One of them then writes the synthesis, with the instruction to validate each claim raised by the others, ending in a remediation plan.<\/li>\n<\/ol>\n<figure class=\"fig\" data-fig=\"review\">\n  <svg viewbox=\"0 0 820 300\" role=\"img\" aria-label=\"One artifact reviewed by three mutually blind reviewers, then a synthesis validating every claim, producing a remediation plan.\">\n    <rect class=\"s-card\" x=\"20\" y=\"110\" width=\"150\" height=\"72\" rx=\"8\"><\/rect>\n    <text class=\"t-sm\" x=\"95\" y=\"142\" text-anchor=\"middle\">The artifact<\/text>\n    <text class=\"t-xs t-mut\" x=\"95\" y=\"164\" text-anchor=\"middle\">code \u00b7 spec \u00b7 plan<\/text>\n    <path class=\"s-arrow\" d=\"M172 146 C 210 146, 210 60, 248 60\"><\/path>\n    <path class=\"s-arrow\" d=\"M172 146 H248\"><\/path>\n    <path class=\"s-arrow\" d=\"M172 146 C 210 146, 210 232, 248 232\"><\/path>\n    <path class=\"s-arrow-head\" d=\"M250 60 l-11 -6 v12 z\"><\/path>\n    <path class=\"s-arrow-head\" d=\"M250 146 l-11 -6 v12 z\"><\/path>\n    <path class=\"s-arrow-head\" d=\"M250 232 l-11 -6 v12 z\"><\/path>\n    <rect class=\"s-box\" x=\"252\" y=\"30\" width=\"200\" height=\"60\" rx=\"8\"><\/rect>\n    <text class=\"t-sm\" x=\"352\" y=\"58\" text-anchor=\"middle\">Reviewer A<\/text>\n    <text class=\"t-xs t-mut\" x=\"352\" y=\"78\" text-anchor=\"middle\">clean context, own file<\/text>\n    <rect class=\"s-box\" x=\"252\" y=\"116\" width=\"200\" height=\"60\" rx=\"8\"><\/rect>\n    <text class=\"t-sm\" x=\"352\" y=\"144\" text-anchor=\"middle\">Reviewer B<\/text>\n    <text class=\"t-xs t-mut\" x=\"352\" y=\"164\" text-anchor=\"middle\">clean context, own file<\/text>\n    <rect class=\"s-box\" x=\"252\" y=\"202\" width=\"200\" height=\"60\" rx=\"8\"><\/rect>\n    <text class=\"t-sm\" x=\"352\" y=\"230\" text-anchor=\"middle\">Reviewer C<\/text>\n    <text class=\"t-xs t-mut\" x=\"352\" y=\"250\" text-anchor=\"middle\">clean context, own file<\/text>\n    <text class=\"t-xs t-acc\" x=\"352\" y=\"189\" text-anchor=\"middle\">no reviewer sees another&#039;s output<\/text>\n    <path class=\"s-arrow\" d=\"M454 60 C 492 60, 492 146, 530 146\"><\/path>\n    <path class=\"s-arrow\" d=\"M454 146 H530\"><\/path>\n    <path class=\"s-arrow\" d=\"M454 232 C 492 232, 492 146, 530 146\"><\/path>\n    <path class=\"s-arrow-head\" d=\"M532 146 l-11 -6 v12 z\"><\/path>\n    <rect class=\"s-accent\" x=\"534\" y=\"110\" width=\"150\" height=\"72\" rx=\"8\"><\/rect>\n    <text class=\"t-sm t-inv\" x=\"609\" y=\"142\" text-anchor=\"middle\">Synthesis<\/text>\n    <text class=\"t-xs t-inv-mut\" x=\"609\" y=\"164\" text-anchor=\"middle\">validate every claim<\/text>\n    <path class=\"s-arrow\" d=\"M688 146 H746\"><\/path>\n    <path class=\"s-arrow-head\" d=\"M748 146 l-11 -6 v12 z\"><\/path>\n    <text class=\"t-sm\" x=\"800\" y=\"142\" text-anchor=\"end\">Remediation<\/text>\n    <text class=\"t-sm\" x=\"800\" y=\"164\" text-anchor=\"end\">plan<\/text>\n  <\/svg>\n  <figcaption>Multi-agent review. The value is not in any one reviewer, it is in the blindness between them.<\/figcaption>\n<\/figure>\n<p>For pure code review the major model families are equivalent. Each finds shared issues and independent, relevant ones. The point is not picking the best, it is crossing them. Even single-provider crossing works: a mid-tier and a top-tier model of the same family already surface different things.<\/p>\n<p>Two observations from practice. On code that has been reworked several times, reviewers often find nothing, which is a good sign and a useful maturity signal. On recent, less mature areas, typically deployment code, each of them surfaces different things.<\/p>\n<p>A concrete example of what it buys. On a monitoring CLI, the main agent defaulted to logging on stdout, following the service rule it had learned. All three independent reviewers flagged the violation: a CLI runs in a terminal, not a container, so stdout logging pollutes the output, and it should log to a file or syslog instead. The main agent had not seen it itself.<\/p>\n<p>Cost framing matters here. Agentic usage consumes 5 to 20 times the tokens of plain completions. A cross-audit by three or four frontier models over a significant codebase is not free. Use it for critical components and major architectural decisions, not for every routine merge request. A single second-model pass on a structural change gives roughly 80% of the benefit for 20% of the cost.<\/p>\n<p>On economics generally, one platform built over five weeks would have cost well over $10,000 in tokens billed per API call, versus a handful of flat-rate subscriptions actually used over that period. Pay-per-use API pricing is not manageable for daily agent work.<\/p>\n  <\/section>\n\n  <section id=\"what-actually-goes-wrong\">\n<h2>9. What actually goes wrong<\/h2>\n<p>Each of these costs someone real time.<\/p>\n<p><strong>Do not stack up tasks.<\/strong> Early on a large project, results came so fast that a whole batch was handed over at once, with automatic relaunch at every quota reset so it would keep going unsupervised overnight. Roughly half of what was produced had to be thrown away, plus a large amount of time spent unpicking the off-topic parts. One subject at a time, verify, consolidate the docs, clear the context, next subject. \u201cFigure the whole thing out on your own\u201d does not work on a complex project.<\/p>\n<p><strong>Never give a demo mockup as a frontend spec.<\/strong> An interactive mockup containing fake JavaScript was handed over as input for coding the frontend. The agent mixed the demo code into the real code, and it nearly cost the whole codebase. Give the visual plus a description of the steps and interactions, advance piece by piece, never the demo code.<\/p>\n<p><strong>Never trust the agent&#039;s self-reported verification, especially on the frontend.<\/strong> On a simple frontend feature, an agent was explicitly asked to verify visually with a browser automation tool. He claimed it had, and that the result matched. Shown an actual screenshot, it admitted the result was nothing like what had been asked, and even after a second attempt it still was not right. On the frontend, agents do somewhat what they want despite clear specs. On the backend (RPC, message queues, mutual TLS, token handling) the same team reports essentially no problems, because those technologies are testable. Check the rendering yourself.<\/p>\n<p><strong>Rewriting an existing project: never ask for a line-by-line translation.<\/strong> Three attempts at porting an internal CLI from Python and shell to Go:<\/p>\n<ol type=\"1\">\n<li><em>\u201cRewrite this script in Go, iso-functional, go ahead\u201d<\/em>: total failure. The agent never understood how it worked. All thrown away.<\/li>\n<li>Feeding it a one-hour transcript of the original author explaining the project: slightly better, still far short. Even the author had not managed to explain everything in an hour.<\/li>\n<li>What worked: drop the idea of rewriting. Explore the code together with the agent, confronting your own understanding against it, until you have a spec of the actual need, discarding the features that only existed because of the original author&#039;s particular design. Once that spec was clear, developing from scratch was straightforward.<\/li>\n<\/ol>\n<p><strong>The agent does not always know where it is pointing.<\/strong> A request to test a new component on a pre-production pipeline led to two chained mistakes. The agent asked for a message to be pasted into a message-queue UI, it was done without thinking and in the wrong UI, so the message went to production. Then the agent pushed its own test messages missing a required identifier, which blocked the connector feeding the search cluster. Nearly the entire infrastructure of that team went down. Read what you execute, and state the target environment, and what is production, explicitly to the agent. Note the inverse observation elsewhere: on other projects the agent asks for confirmation systematically, even on pre-production. Behavior depends on the framing, so never rely on it.<\/p>\n<p><strong>\u201cGo ahead, figure it out\u201d can go very far.<\/strong> Asked to fix a firewall problem on a personal project, an agent noticed SSH was open with the user&#039;s key, connected to the machine, pulled the scripts and started operating on production. It went fine. It could have gone very badly.<\/p>\n<p><strong>Check the pipeline is entirely green.<\/strong> The agent often looks at the first job, sees green and concludes it is done while later jobs fail. Ask it to iterate until fully green, and expect it to skip local checks before pushing despite explicit rules. CI is the real guardrail.<\/p>\n<h3 id=\"where-agents-shine-and-where-they-do-not\">Where agents shine, and where they do not<\/h3>\n<p>They work well on standalone, well-framed projects with defined inputs and outputs. A team with no ML background brainstormed the subject with an agent, which then built a local test and benchmark platform, ran whole nights of model tuning, and produced two qualification components that work well, without anyone reading the code. Also: infrastructure-as-code reviews, sysadmin companionship, capitalizing operations into the docs, investigating live application logs, driving version-control CLIs.<\/p>\n<p>Dashboards are the contested case, and reports differ enough between teams that we do not treat this as settled. Some get usable first drafts, others spend more time explaining what they want than they would spend building it by hand. The split seems to follow the data more than the tool: already-curated metrics give the agent something to reason about, while raw log data needs meaning assigned to it before a dashboard means anything. Where it does go badly, the diagnosis is definition rather than execution, because with a dashboard you discover which indicator is missing while building it, and the round-trip with the agent breaks that loop. Two leads: ask the teams getting good drafts what they do differently, and transpose the live-rendering workflow that works for UI.<\/p>\n  <\/section>\n\n  <section id=\"domain-experts-come-inside-the-loop\">\n<h2>10. Domain experts come inside the loop<\/h2>\n<p>This is one of the largest wins, and it forces a question about roles that deserve to be asked out loud. The chain that disappears is one a product manager often stood in the middle of, translating business intent into something a developer could act on. That translation function is the part that loses its reason to exist. What does not disappear, and becomes more valuable, is the rest of the job: arbitrating between demands that cannot all be satisfied, sequencing, holding the product line, defining what success would look like, and saying no. The mistake would be to read this as the role becoming redundant. It is the intermediate half that is.<\/p>\n<p>The classic way a feature went wrong had nothing to do with technology. A domain expert tried to put a need into words. An intermediate translated it. A developer specified it in their own terms. Someone implemented it. Weeks later a demo revealed the result was not really what was wanted, and nobody had lied at any step. The need had simply been re-encoded four times, losing a little at each hop.<\/p>\n<p>Agentic development collapses that chain. The person who holds the need can now produce something concrete, a mockup, a working prototype, a rough script that does the real thing on realistic data, without waiting for a developer to be available and without learning to code first.<\/p>\n<figure class=\"fig\" data-fig=\"chain\">\n  <svg viewbox=\"0 0 820 290\" role=\"img\" aria-label=\"The old five-hop chain from domain expert to demo, losing fidelity at each hop, replaced by a direct loop between the expert and an agent.\">\n    <text class=\"t-lbl t-mut\" x=\"20\" y=\"26\">BEFORE: FOUR RE-ENCODINGS<\/text>\n    <rect class=\"s-box\" x=\"20\" y=\"42\" width=\"128\" height=\"52\" rx=\"6\"><\/rect>\n    <text class=\"t-sm\" x=\"84\" y=\"66\" text-anchor=\"middle\">Domain<\/text>\n    <text class=\"t-sm\" x=\"84\" y=\"84\" text-anchor=\"middle\">expert<\/text>\n    <rect class=\"s-box\" x=\"180\" y=\"42\" width=\"128\" height=\"52\" rx=\"6\"><\/rect>\n    <text class=\"t-sm\" x=\"244\" y=\"74\" text-anchor=\"middle\">Intermediary<\/text>\n    <rect class=\"s-box\" x=\"340\" y=\"42\" width=\"128\" height=\"52\" rx=\"6\"><\/rect>\n    <text class=\"t-sm\" x=\"404\" y=\"74\" text-anchor=\"middle\">Developer<\/text>\n    <rect class=\"s-box\" x=\"500\" y=\"42\" width=\"128\" height=\"52\" rx=\"6\"><\/rect>\n    <text class=\"t-sm\" x=\"564\" y=\"74\" text-anchor=\"middle\">Implementation<\/text>\n    <rect class=\"s-box\" x=\"660\" y=\"42\" width=\"140\" height=\"52\" rx=\"6\"><\/rect>\n    <text class=\"t-sm\" x=\"730\" y=\"66\" text-anchor=\"middle\">Demo,<\/text>\n    <text class=\"t-sm\" x=\"730\" y=\"84\" text-anchor=\"middle\">weeks later<\/text>\n    <path class=\"s-arrow\" d=\"M150 68 H176\"><\/path><path class=\"s-arrow-head\" d=\"M178 68 l-11 -6 v12 z\"><\/path>\n    <path class=\"s-arrow\" d=\"M310 68 H336\"><\/path><path class=\"s-arrow-head\" d=\"M338 68 l-11 -6 v12 z\"><\/path>\n    <path class=\"s-arrow\" d=\"M470 68 H496\"><\/path><path class=\"s-arrow-head\" d=\"M498 68 l-11 -6 v12 z\"><\/path>\n    <path class=\"s-arrow\" d=\"M630 68 H656\"><\/path><path class=\"s-arrow-head\" d=\"M658 68 l-11 -6 v12 z\"><\/path>\n    <text class=\"t-xs t-acc\" x=\"163\" y=\"116\" text-anchor=\"middle\">\u2212<\/text>\n    <text class=\"t-xs t-acc\" x=\"323\" y=\"116\" text-anchor=\"middle\">\u2212<\/text>\n    <text class=\"t-xs t-acc\" x=\"483\" y=\"116\" text-anchor=\"middle\">\u2212<\/text>\n    <text class=\"t-xs t-acc\" x=\"643\" y=\"116\" text-anchor=\"middle\">\u2212<\/text>\n    <text class=\"t-xs t-mut\" x=\"410\" y=\"136\" text-anchor=\"middle\">a little of the need is lost at every hop<\/text>\n    <line class=\"s-sep\" x1=\"20\" y1=\"158\" x2=\"800\" y2=\"158\"><\/line>\n    <text class=\"t-lbl t-acc\" x=\"20\" y=\"186\">NOW: ONE LOOP<\/text>\n    <rect class=\"s-accent\" x=\"180\" y=\"202\" width=\"200\" height=\"62\" rx=\"8\"><\/rect>\n    <text class=\"t-sm t-inv\" x=\"280\" y=\"228\" text-anchor=\"middle\">Domain expert<\/text>\n    <text class=\"t-xs t-inv-mut\" x=\"280\" y=\"250\" text-anchor=\"middle\">holds the need, judges the result<\/text>\n    <rect class=\"s-solid\" x=\"440\" y=\"202\" width=\"200\" height=\"62\" rx=\"8\"><\/rect>\n    <text class=\"t-sm t-inv\" x=\"540\" y=\"228\" text-anchor=\"middle\">Agent<\/text>\n    <text class=\"t-xs t-inv-mut\" x=\"540\" y=\"250\" text-anchor=\"middle\">challenges, drafts, builds<\/text>\n    <path class=\"s-arrow\" d=\"M384 218 H434\"><\/path><path class=\"s-arrow-head\" d=\"M436 218 l-11 -6 v12 z\"><\/path>\n    <path class=\"s-arrow\" d=\"M436 248 H386\"><\/path><path class=\"s-arrow-head\" d=\"M384 248 l11 -6 v12 z\"><\/path>\n    <text class=\"t-xs t-mut\" x=\"700\" y=\"224\">a mockup people can<\/text>\n    <text class=\"t-xs t-mut\" x=\"700\" y=\"242\">react to, in a day<\/text>\n  <\/svg>\n  <figcaption>The translation chain collapses. Nobody lied at any step. The need was simply re-encoded four times.<\/figcaption>\n<\/figure>\n<p>That changes what a domain expert is expected to produce. A mockup or prototype is worth more than a paragraph of requirements, because it is something people can react to. \u201cNot that, more like this\u201d is a more reliable signal than any specification review. They are also the best person to write the spec&#039;s WHAT, since they know what correct looks like on real data, and the best person to verify functionally, since judging whether the output is right requires no ability to read the code.<\/p>\n<h3 id=\"the-risk-shadow-development\">The risk: shadow development<\/h3>\n<p>Without a frame, this produces shadow development, the technical equivalent of shadow IT: unversioned, untested, unmaintained code accumulating invisible technical debt until it becomes critical. The goal is not to forbid it, it is to channel the creative energy into safe rails.<\/p>\n<p>Four principles, and a classification.<\/p>\n<ol type=\"1\">\n<li><strong>Approved tools only.<\/strong> No unvalidated tool keys internal or sensitive data. In a cybersecurity context this is not negotiable.<\/li>\n<li><strong>All code lives in version control<\/strong>, with a minimal CI. Deep Git mastery is not required. Branch, merge request, automated review is enough.<\/li>\n<li><strong>Mandatory review before deployment.<\/strong> Non-developer contributions are treated as drafts: the AI produced a first pass, a developer validates architecture, security and integration.<\/li>\n<li><strong>Templates and starters.<\/strong> A pre-configured project (structure, linters, CI, rule file) so the agent works inside a constrained frame rather than from a blank page. The rules apply to AI-generated code whoever prompts it.<\/li>\n<\/ol>\n<div class=\"table-wrap\"><table>\n<colgroup><col style=\"width: 33%\"><col style=\"width: 33%\"><col style=\"width: 33%\"><\/colgroup>\n<thead><tr><th>Category<\/th><th>Examples<\/th><th>Governance<\/th><\/tr><\/thead>\n<tbody>\n<tr><td><strong>Exploration<\/strong><\/td><td>UX prototypes, throwaway POCs, one-off analyzes<\/td><td>Version control optional. No deployment. Internal use only<\/td><\/tr>\n<tr><td><strong>Internal tooling<\/strong><\/td><td>Automation scripts, dashboards, log parsers<\/td><td>Version control mandatory. Automatic linting. Developer review before production use<\/td><\/tr>\n<tr><td><strong>Production code<\/strong><\/td><td>Platform components, business rules, APIs<\/td><td>Full workflow: version control, TDD, CI, review, squash merge<\/td><\/tr>\n<\/tbody>\n<\/table><\/div>\n<p>Be honest about which row you are in, and note that things move down the table over time. The script \u201cjust for me\u201d that a colleague starts relying on has become internal tooling.<\/p>\n<p>Two cautions apply specifically to this path. A mockup is an input for discussion, never an input for code, as the war story above shows. And the data rules apply identically to everyone: no secrets, no client data, anonymize first.<\/p>\n  <\/section>\n\n  <section id=\"adopting-this-in-a-team\">\n<h2>11. Adopting this in a team<\/h2>\n<p>The condensed playbook:<\/p>\n<ol type=\"1\">\n<li><strong>Start from the chore they hate<\/strong>, not from the tool. Regression tests, boilerplate reviews, endpoint documentation. The message is not \u201ctrust me\u201d, it is \u201clook at this diff\u201d.<\/li>\n<li><strong>Make the rule files the tangible proof.<\/strong> Run the same task without a project rule file, with its hallucinated conventions and reinvented components, then with it. Context is the strongest trust lever there is.<\/li>\n<li><strong>TDD as a psychological safety net.<\/strong> The agent cannot cheat if it must pass tests you wrote. You define the contract, the AI implements it, the tests verify.<\/li>\n<li><strong>Pilot on a real brownfield task<\/strong>, not a greenfield demo. The disappointment came from existing codebases, so that is where the proof has to land. Pick something the skeptic knows well.<\/li>\n<li><strong>Quantify fast.<\/strong> Two or three light metrics before the pilot: time per merge request, review comments, coverage. Move the conversation from \u201cAI doesn\u2019t work\u201d to something observable.<\/li>\n<li><strong>Respect the expertise.<\/strong> Never \u201cthe AI will code for you\u201d, rather \u201cthe AI codes under your orders, and your value moves from writing syntax to designing constraints and auditing systems\u201d.<\/li>\n<\/ol>\n<p>A workable ramp for a small team: weeks 1-2, silent foundations, meaning rule files and zero-diff linting. Weeks 3-4, one volunteer on one real brownfield task, documented and shared. Weeks 5-8, a second use case, each person picking their own. Month 3 onwards, standardize what worked and drop what did not, without guilt.<\/p>\n<p>One prompt worth reusing for unattended runs, overnight or while you are in a meeting:<\/p>\n<pre><code>Do everything you can without asking me questions. If you have any doubt that is not blocking, note it in a file, and we&#039;ll review it when I&#039;m back.<\/code><\/pre>\n  <\/section>\n\n  <section id=\"what-is-not-yet-solved\">\n<h2>12. What is not (yet) solved<\/h2>\n<p>Writing code by hand has become obsolete for us. In a year of practice, one hand-written equation on a personal project, because explaining it was slower than typing it.<\/p>\n<p>Deployment is different. It still demands substantial expertise and real knowledge of the stacks to steer agents correctly. The framework keeps improving on that front, but it is not solved. Budget human expertise there, and expect a multi-agent review to keep finding things in deployment code long after it finds nothing in application code.<\/p>\n<p>A closing deposit that has nothing to do with AI capability. Internal development targets our own efficiency and innovation. It is not meant to rebuild what specialized vendors already do well. Build internally when the need is too specific to how you work, or when a market tool would create excessive vendor lock-in, and arbitrate anything substantial.<\/p>\n<p>Development being easy is not a reason to develop anything and everything.<\/p>\n  <\/section>\n\n  <section id=\"sources\">\n<h2>Sources<\/h2>\n<ul>\n<li>Stack Overflow, <a href=\"https:\/\/survey.stackoverflow.co\/2025\/ai\">Developer Survey 2025, AI section<\/a>, 49,000 respondents. Adoption, trust and frustration figures.<\/li>\n<li>Qodo, <a href=\"https:\/\/www.qodo.ai\/reports\/state-of-ai-code-quality\/\">State of AI Code Quality 2025<\/a>. Contextual understanding as the top improvement request.<\/li>\n<li>Cloudflare, <a href=\"https:\/\/blog.cloudflare.com\/18-november-2025-outage\/\">Cloudflare outage on November 18, 2025<\/a>. The configuration file, the hardcoded limit and the panic.<\/li>\n<li><a href=\"https:\/\/en.wikipedia.org\/wiki\/Lufthansa_Flight_2904\">Lufthansa Flight 2904<\/a>, Warsaw, September 14, 1993. The ground-detection logic and the nine-second delay.<\/li>\n<\/ul>\n  <\/section>\n\n<\/div>\n<!-- =============================================================\n     END of block. Post title to set in WordPress:\n     \"Industrialising AI-Assisted Development: What We Changed and Why\"\n     Suggested meta description:\n     \"One year of practice industrialising AI-assisted development: the\n     posture shift, the harness, the documentation model, and the mistakes\n     that shaped an internal framework.\"\n     ============================================================= -->\n\n\n\n<p class=\"wp-block-paragraph\">Written from a year of building software with agents. The framework described here is internal, the reasoning is not.<\/p>\n\n\n\n<p class=\"wp-block-paragraph\">Written by <strong>Stany MARCEL<\/strong> with <strong>Claude<\/strong>, from Stany MARCEL&#039;s research work and the agentic development framework built by Intrinsec&#039;s Software Engineering team.<\/p>","protected":false},"excerpt":{"rendered":"<p>Over the past year the way we build software internally has changed shape. Code is [&hellip;]<\/p>\n","protected":false},"author":60,"featured_media":232579,"comment_status":"closed","ping_status":"open","sticky":false,"template":"","format":"standard","meta":{"footnotes":""},"categories":[12,17],"tags":[374,378,380,377,375,382,383,381,379],"class_list":["post-232436","post","type-post","status-publish","format-standard","has-post-thumbnail","hentry","category-engineering","category-recherche-et-developpement","tag-ai","tag-ai-coding-agents","tag-ai-security","tag-ai-assisted-development","tag-artificial-intelligence","tag-developer-experience","tag-safety-engineering","tag-secure-development","tag-software-development"],"yoast_head":"<!-- This site is optimized with the Yoast SEO Premium plugin v28.6 (Yoast SEO v28.6) - https:\/\/yoast.com\/product\/yoast-seo-premium-wordpress\/ -->\n<title>AI-Assisted Development: What We Changed and Why | Intrinsec<\/title>\n<meta name=\"description\" content=\"Learn how we industrialised AI-assisted development, what changed after 1 year of practice, and the lessons that shaped our internal framework\" \/>\n<meta name=\"robots\" content=\"index, follow, max-snippet:-1, max-image-preview:large, max-video-preview:-1\" \/>\n<link rel=\"canonical\" href=\"https:\/\/www.intrinsec.com\/en\/ai-assisted-development-what-we-changed-and-why\/\" \/>\n<meta property=\"og:locale\" content=\"en_US\" \/>\n<meta property=\"og:type\" content=\"article\" \/>\n<meta property=\"og:title\" content=\"Industrialising AI-Assisted Development: What We Changed and Why\" \/>\n<meta property=\"og:description\" content=\"Learn how we industrialised AI-assisted development, what changed after 1 year of practice, and the lessons that shaped our internal framework\" \/>\n<meta property=\"og:url\" content=\"https:\/\/www.intrinsec.com\/en\/ai-assisted-development-what-we-changed-and-why\/\" \/>\n<meta property=\"og:site_name\" content=\"INTRINSEC\" \/>\n<meta property=\"article:published_time\" content=\"2026-10-01T13:29:35+00:00\" \/>\n<meta property=\"article:modified_time\" content=\"2026-10-02T08:40:52+00:00\" \/>\n<meta property=\"og:image\" content=\"https:\/\/www.intrinsec.com\/wp-content\/uploads\/2026\/10\/guttest192_Premium_editorial_hero_image_for_a_French_cybersec_f7683194-d9b7-46e5-b4c6-d31ac4acc684_3.png\" \/>\n\t<meta property=\"og:image:width\" content=\"1232\" \/>\n\t<meta property=\"og:image:height\" content=\"928\" \/>\n\t<meta property=\"og:image:type\" content=\"image\/png\" \/>\n<meta name=\"author\" content=\"Stany MARCEL\" \/>\n<meta name=\"twitter:card\" content=\"summary_large_image\" \/>\n<meta name=\"twitter:creator\" content=\"@Intrinsec\" \/>\n<meta name=\"twitter:site\" content=\"@Intrinsec\" \/>\n<meta name=\"twitter:label1\" content=\"Written by\" \/>\n\t<meta name=\"twitter:data1\" content=\"Stany MARCEL\" \/>\n\t<meta name=\"twitter:label2\" content=\"Est. reading time\" \/>\n\t<meta name=\"twitter:data2\" content=\"35 minutes\" \/>\n<script type=\"application\/ld+json\" class=\"yoast-schema-graph\">{\"@context\":\"https:\\\/\\\/schema.org\",\"@graph\":[{\"@type\":\"Article\",\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/#article\",\"isPartOf\":{\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/\"},\"author\":{\"name\":\"Stany MARCEL\",\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/#\\\/schema\\\/person\\\/e54654e7916b88eab240b5d456c1c809\"},\"headline\":\"Industrialising AI-Assisted Development: What We Changed and Why\",\"datePublished\":\"2026-10-01T13:29:35+00:00\",\"dateModified\":\"2026-10-02T08:40:52+00:00\",\"mainEntityOfPage\":{\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/\"},\"wordCount\":6726,\"publisher\":{\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/#organization\"},\"image\":{\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/#primaryimage\"},\"thumbnailUrl\":\"https:\\\/\\\/www.intrinsec.com\\\/wp-content\\\/uploads\\\/2026\\\/10\\\/guttest192_Premium_editorial_hero_image_for_a_French_cybersec_f7683194-d9b7-46e5-b4c6-d31ac4acc684_3.png\",\"keywords\":[\"AI\",\"AI coding agents\",\"AI Security\",\"AI-assisted development\",\"Artificial Intelligence\",\"Developer Experience\",\"Safety Engineering\",\"Secure Development\",\"Software Development\"],\"articleSection\":[\"Engineering\",\"Recherche et D\u00e9veloppement\"],\"inLanguage\":\"en-US\"},{\"@type\":\"WebPage\",\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/\",\"url\":\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/\",\"name\":\"AI-Assisted Development: What We Changed and Why | Intrinsec\",\"isPartOf\":{\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/#website\"},\"primaryImageOfPage\":{\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/#primaryimage\"},\"image\":{\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/#primaryimage\"},\"thumbnailUrl\":\"https:\\\/\\\/www.intrinsec.com\\\/wp-content\\\/uploads\\\/2026\\\/10\\\/guttest192_Premium_editorial_hero_image_for_a_French_cybersec_f7683194-d9b7-46e5-b4c6-d31ac4acc684_3.png\",\"datePublished\":\"2026-10-01T13:29:35+00:00\",\"dateModified\":\"2026-10-02T08:40:52+00:00\",\"description\":\"Learn how we industrialised AI-assisted development, what changed after 1 year of practice, and the lessons that shaped our internal framework\",\"breadcrumb\":{\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/#breadcrumb\"},\"inLanguage\":\"en-US\",\"potentialAction\":[{\"@type\":\"ReadAction\",\"target\":[\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/\"]}]},{\"@type\":\"ImageObject\",\"inLanguage\":\"en-US\",\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/#primaryimage\",\"url\":\"https:\\\/\\\/www.intrinsec.com\\\/wp-content\\\/uploads\\\/2026\\\/10\\\/guttest192_Premium_editorial_hero_image_for_a_French_cybersec_f7683194-d9b7-46e5-b4c6-d31ac4acc684_3.png\",\"contentUrl\":\"https:\\\/\\\/www.intrinsec.com\\\/wp-content\\\/uploads\\\/2026\\\/10\\\/guttest192_Premium_editorial_hero_image_for_a_French_cybersec_f7683194-d9b7-46e5-b4c6-d31ac4acc684_3.png\",\"width\":1232,\"height\":928,\"caption\":\"Industrialising AI-Assisted Development\"},{\"@type\":\"BreadcrumbList\",\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/ai-assisted-development-what-we-changed-and-why\\\/#breadcrumb\",\"itemListElement\":[{\"@type\":\"ListItem\",\"position\":1,\"name\":\"Accueil\",\"item\":\"https:\\\/\\\/www.intrinsec.com\\\/\"},{\"@type\":\"ListItem\",\"position\":2,\"name\":\"Industrialising AI-Assisted Development: What We Changed and Why\"}]},{\"@type\":\"WebSite\",\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/#website\",\"url\":\"https:\\\/\\\/www.intrinsec.com\\\/\",\"name\":\"INTRINSEC\",\"description\":\"Notre m\u00e9tier , Prot\u00e9ger le v\u00f4tre\",\"publisher\":{\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/#organization\"},\"potentialAction\":[{\"@type\":\"SearchAction\",\"target\":{\"@type\":\"EntryPoint\",\"urlTemplate\":\"https:\\\/\\\/www.intrinsec.com\\\/?s={search_term_string}\"},\"query-input\":{\"@type\":\"PropertyValueSpecification\",\"valueRequired\":true,\"valueName\":\"search_term_string\"}}],\"inLanguage\":\"en-US\"},{\"@type\":\"Organization\",\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/#organization\",\"name\":\"INTRINSEC\",\"alternateName\":\"ISEC\",\"url\":\"https:\\\/\\\/www.intrinsec.com\\\/\",\"logo\":{\"@type\":\"ImageObject\",\"inLanguage\":\"en-US\",\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/#\\\/schema\\\/logo\\\/image\\\/\",\"url\":\"https:\\\/\\\/www.intrinsec.com\\\/wp-content\\\/uploads\\\/2025\\\/02\\\/libellule.png\",\"contentUrl\":\"https:\\\/\\\/www.intrinsec.com\\\/wp-content\\\/uploads\\\/2025\\\/02\\\/libellule.png\",\"width\":1322,\"height\":1322,\"caption\":\"INTRINSEC\"},\"image\":{\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/#\\\/schema\\\/logo\\\/image\\\/\"},\"sameAs\":[\"https:\\\/\\\/x.com\\\/Intrinsec\",\"https:\\\/\\\/fr.linkedin.com\\\/company\\\/intrinsec\",\"https:\\\/\\\/www.youtube.com\\\/channel\\\/UC0trUZAHNZOUbxYnNdecM4A\"],\"description\":\"soci\u00e9t\u00e9 de consulting, pure player cybers\u00e9curit\u00e9 fran\u00e7ais et europ\u00e9en depuis plus de 30ans, sp\u00e9cialiste dans la s\u00e9curit\u00e9 offensive & audit (pentest\\\/red team), GRC, et services IMSS comme le SOC, CTI et CERT Intrinsec est qualifi\u00e9 PASSI Elev\u00e9, PRIS Elev\u00e9 et PACS par l'ANSSI\",\"email\":\"contact@intrinsec.com\"},{\"@type\":\"Person\",\"@id\":\"https:\\\/\\\/www.intrinsec.com\\\/#\\\/schema\\\/person\\\/e54654e7916b88eab240b5d456c1c809\",\"name\":\"Stany MARCEL\",\"image\":{\"@type\":\"ImageObject\",\"inLanguage\":\"en-US\",\"@id\":\"https:\\\/\\\/secure.gravatar.com\\\/avatar\\\/56a6dedb03014d323bcdc4e57e398b0bdf6873e6f7f2dcffebe8bfab47701928?s=96&d=retro&r=g\",\"url\":\"https:\\\/\\\/secure.gravatar.com\\\/avatar\\\/56a6dedb03014d323bcdc4e57e398b0bdf6873e6f7f2dcffebe8bfab47701928?s=96&d=retro&r=g\",\"contentUrl\":\"https:\\\/\\\/secure.gravatar.com\\\/avatar\\\/56a6dedb03014d323bcdc4e57e398b0bdf6873e6f7f2dcffebe8bfab47701928?s=96&d=retro&r=g\",\"caption\":\"Stany MARCEL\"},\"url\":\"https:\\\/\\\/www.intrinsec.com\\\/en\\\/author\\\/stany-marcel\\\/\"}]}<\/script>\n<!-- \/ Yoast SEO Premium plugin. -->","yoast_head_json":{"title":"AI-Assisted Development: What We Changed and Why | Intrinsic","description":"Learn how we industrialized AI-assisted development, what changed after 1 year of practice, and the lessons that shaped our internal framework","robots":{"index":"index","follow":"follow","max-snippet":"max-snippet:-1","max-image-preview":"max-image-preview:large","max-video-preview":"max-video-preview:-1"},"canonical":"https:\/\/www.intrinsec.com\/en\/ai-assisted-development-what-we-changed-and-why\/","og_locale":"en_US","og_type":"article","og_title":"Industrialising AI-Assisted Development: What We Changed and Why","og_description":"Learn how we industrialised AI-assisted development, what changed after 1 year of practice, and the lessons that shaped our internal framework","og_url":"https:\/\/www.intrinsec.com\/en\/ai-assisted-development-what-we-changed-and-why\/","og_site_name":"INTRINSEC","article_published_time":"2026-10-01T13:29:35+00:00","article_modified_time":"2026-10-02T08:40:52+00:00","og_image":[{"width":1232,"height":928,"url":"https:\/\/www.intrinsec.com\/wp-content\/uploads\/2026\/10\/guttest192_Premium_editorial_hero_image_for_a_French_cybersec_f7683194-d9b7-46e5-b4c6-d31ac4acc684_3.png","type":"image\/png"}],"author":"Stany MARCEL","twitter_card":"summary_large_image","twitter_creator":"@Intrinsec","twitter_site":"@Intrinsec","twitter_misc":{"Written by":"Stany MARCEL","Est. reading time":"35 minutes"},"schema":{"@context":"https:\/\/schema.org","@graph":[{"@type":"Article","@id":"https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/#article","isPartOf":{"@id":"https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/"},"author":{"name":"Stany MARCEL","@id":"https:\/\/www.intrinsec.com\/#\/schema\/person\/e54654e7916b88eab240b5d456c1c809"},"headline":"Industrialising AI-Assisted Development: What We Changed and Why","datePublished":"2026-10-01T13:29:35+00:00","dateModified":"2026-10-02T08:40:52+00:00","mainEntityOfPage":{"@id":"https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/"},"wordCount":6726,"publisher":{"@id":"https:\/\/www.intrinsec.com\/#organization"},"image":{"@id":"https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/#primaryimage"},"thumbnailUrl":"https:\/\/www.intrinsec.com\/wp-content\/uploads\/2026\/10\/guttest192_Premium_editorial_hero_image_for_a_French_cybersec_f7683194-d9b7-46e5-b4c6-d31ac4acc684_3.png","keywords":["AI","AI coding agents","AI Security","AI-assisted development","Artificial Intelligence","Developer Experience","Safety Engineering","Secure Development","Software Development"],"articleSection":["Engineering","Recherche et D\u00e9veloppement"],"inLanguage":"en-US"},{"@type":"WebPage","@id":"https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/","url":"https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/","name":"AI-Assisted Development: What We Changed and Why | Intrinsic","isPartOf":{"@id":"https:\/\/www.intrinsec.com\/#website"},"primaryImageOfPage":{"@id":"https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/#primaryimage"},"image":{"@id":"https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/#primaryimage"},"thumbnailUrl":"https:\/\/www.intrinsec.com\/wp-content\/uploads\/2026\/10\/guttest192_Premium_editorial_hero_image_for_a_French_cybersec_f7683194-d9b7-46e5-b4c6-d31ac4acc684_3.png","datePublished":"2026-10-01T13:29:35+00:00","dateModified":"2026-10-02T08:40:52+00:00","description":"Learn how we industrialized AI-assisted development, what changed after 1 year of practice, and the lessons that shaped our internal framework","breadcrumb":{"@id":"https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/#breadcrumb"},"inLanguage":"en-US","potentialAction":[{"@type":"ReadAction","target":["https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/"]}]},{"@type":"ImageObject","inLanguage":"en-US","@id":"https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/#primaryimage","url":"https:\/\/www.intrinsec.com\/wp-content\/uploads\/2026\/10\/guttest192_Premium_editorial_hero_image_for_a_French_cybersec_f7683194-d9b7-46e5-b4c6-d31ac4acc684_3.png","contentUrl":"https:\/\/www.intrinsec.com\/wp-content\/uploads\/2026\/10\/guttest192_Premium_editorial_hero_image_for_a_French_cybersec_f7683194-d9b7-46e5-b4c6-d31ac4acc684_3.png","width":1232,"height":928,"caption":"Industrialising AI-Assisted Development"},{"@type":"BreadcrumbList","@id":"https:\/\/www.intrinsec.com\/ai-assisted-development-what-we-changed-and-why\/#breadcrumb","itemListElement":[{"@type":"ListItem","position":1,"name":"Accueil","item":"https:\/\/www.intrinsec.com\/"},{"@type":"ListItem","position":2,"name":"Industrialising AI-Assisted Development: What We Changed and Why"}]},{"@type":"WebSite","@id":"https:\/\/www.intrinsec.com\/#website","url":"https:\/\/www.intrinsec.com\/","name":"INTRINSEC","description":"Our job is to protect yours.","publisher":{"@id":"https:\/\/www.intrinsec.com\/#organization"},"potentialAction":[{"@type":"SearchAction","target":{"@type":"EntryPoint","urlTemplate":"https:\/\/www.intrinsec.com\/?s={search_term_string}"},"query-input":{"@type":"PropertyValueSpecification","valueRequired":true,"valueName":"search_term_string"}}],"inLanguage":"en-US"},{"@type":"Organization","@id":"https:\/\/www.intrinsec.com\/#organization","name":"INTRINSEC","alternateName":"ISEC","url":"https:\/\/www.intrinsec.com\/","logo":{"@type":"ImageObject","inLanguage":"en-US","@id":"https:\/\/www.intrinsec.com\/#\/schema\/logo\/image\/","url":"https:\/\/www.intrinsec.com\/wp-content\/uploads\/2025\/02\/libellule.png","contentUrl":"https:\/\/www.intrinsec.com\/wp-content\/uploads\/2025\/02\/libellule.png","width":1322,"height":1322,"caption":"INTRINSEC"},"image":{"@id":"https:\/\/www.intrinsec.com\/#\/schema\/logo\/image\/"},"sameAs":["https:\/\/x.com\/Intrinsec","https:\/\/fr.linkedin.com\/company\/intrinsec","https:\/\/www.youtube.com\/channel\/UC0trUZAHNZOUbxYnNdecM4A"],"description":"Intrinsec, a consulting firm and pure-play French and European cybersecurity provider for over 30 years, specializes in offensive security and auditing (penetration testing\/red teams), GRC, and IMSS services such as SOC, CTI, and CERT. Intrinsec is qualified at PASSI High, PRIS High, and PACS levels by ANSSI.","email":"contact@intrinsec.com"},{"@type":"Person","@id":"https:\/\/www.intrinsec.com\/#\/schema\/person\/e54654e7916b88eab240b5d456c1c809","name":"Stany MARCEL","image":{"@type":"ImageObject","inLanguage":"en-US","@id":"https:\/\/secure.gravatar.com\/avatar\/56a6dedb03014d323bcdc4e57e398b0bdf6873e6f7f2dcffebe8bfab47701928?s=96&d=retro&r=g","url":"https:\/\/secure.gravatar.com\/avatar\/56a6dedb03014d323bcdc4e57e398b0bdf6873e6f7f2dcffebe8bfab47701928?s=96&d=retro&r=g","contentUrl":"https:\/\/secure.gravatar.com\/avatar\/56a6dedb03014d323bcdc4e57e398b0bdf6873e6f7f2dcffebe8bfab47701928?s=96&d=retro&r=g","caption":"Stany MARCEL"},"url":"https:\/\/www.intrinsec.com\/en\/author\/stany-marcel\/"}]}},"_links":{"self":[{"href":"https:\/\/www.intrinsec.com\/en\/wp-json\/wp\/v2\/posts\/232436","targetHints":{"allow":["GET"]}}],"collection":[{"href":"https:\/\/www.intrinsec.com\/en\/wp-json\/wp\/v2\/posts"}],"about":[{"href":"https:\/\/www.intrinsec.com\/en\/wp-json\/wp\/v2\/types\/post"}],"author":[{"embeddable":true,"href":"https:\/\/www.intrinsec.com\/en\/wp-json\/wp\/v2\/users\/60"}],"replies":[{"embeddable":true,"href":"https:\/\/www.intrinsec.com\/en\/wp-json\/wp\/v2\/comments?post=232436"}],"version-history":[{"count":3,"href":"https:\/\/www.intrinsec.com\/en\/wp-json\/wp\/v2\/posts\/232436\/revisions"}],"predecessor-version":[{"id":232575,"href":"https:\/\/www.intrinsec.com\/en\/wp-json\/wp\/v2\/posts\/232436\/revisions\/232575"}],"wp:featuredmedia":[{"embeddable":true,"href":"https:\/\/www.intrinsec.com\/en\/wp-json\/wp\/v2\/media\/232579"}],"wp:attachment":[{"href":"https:\/\/www.intrinsec.com\/en\/wp-json\/wp\/v2\/media?parent=232436"}],"wp:term":[{"taxonomy":"category","embeddable":true,"href":"https:\/\/www.intrinsec.com\/en\/wp-json\/wp\/v2\/categories?post=232436"},{"taxonomy":"post_tag","embeddable":true,"href":"https:\/\/www.intrinsec.com\/en\/wp-json\/wp\/v2\/tags?post=232436"}],"curies":[{"name":"wp","href":"https:\/\/api.w.org\/{rel}","templated":true}]}}