<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:dc="http://purl.org/dc/elements/1.1/">
  <channel>
    <title>DEV Community: Sabith KS</title>
    <description>The latest articles on DEV Community by Sabith KS (@sks).</description>
    <link>https://dev.to/sks</link>
    <image>
      <url>https://media2.dev.to/dynamic/image/width=90,height=90,fit=cover,gravity=auto,format=auto/https:%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F180380%2F5047aa95-dce9-4b00-9d75-5f59434b5059.jpeg</url>
      <title>DEV Community: Sabith KS</title>
      <link>https://dev.to/sks</link>
    </image>
    <atom:link rel="self" type="application/rss+xml" href="https://dev.to/feed/sks"/>
    <language>en</language>
    <item>
      <title>Evidence-Gated RCA — Prove, Then Narrate</title>
      <dc:creator>Sabith KS</dc:creator>
      <pubDate>Wed, 08 Jul 2026 17:00:00 +0000</pubDate>
      <link>https://dev.to/sks/evidence-gated-rca-prove-then-narrate-150</link>
      <guid>https://dev.to/sks/evidence-gated-rca-prove-then-narrate-150</guid>
      <description>&lt;p&gt;Evidence-gated multi-plane RCA — fixed DAG, structural evals, and token-aware tool loops for production agent workflows.&lt;/p&gt;

</description>
      <category>aiagents</category>
      <category>compoundai</category>
      <category>orchestration</category>
      <category>evaluation</category>
    </item>
    <item>
      <title>Just realized the JSON pain has been already solved thanks to github.com/kaptinlin/jsonrepair 

Shout out to @kaptinlin</title>
      <dc:creator>Sabith KS</dc:creator>
      <pubDate>Fri, 03 Jul 2026 18:41:46 +0000</pubDate>
      <link>https://dev.to/sks/just-realized-the-json-pain-has-been-already-solved-thanks-to-githubcomkaptinlinjsonrepair-eo6</link>
      <guid>https://dev.to/sks/just-realized-the-json-pain-has-been-already-solved-thanks-to-githubcomkaptinlinjsonrepair-eo6</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/sks/why-we-chose-go-for-our-ai-agent-platform-when-everyone-else-picked-python-4ok2" class="crayons-story__hidden-navigation-link"&gt;Why We Chose Go for Our AI Agent Platform (When Everyone Else Picked Python)&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/sks" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F180380%2F5047aa95-dce9-4b00-9d75-5f59434b5059.jpeg" alt="sks profile" class="crayons-avatar__image" width="233" height="233"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/sks" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Sabith KS
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Sabith KS
                
              
              &lt;div id="story-author-preview-content-4045427" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/sks" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F180380%2F5047aa95-dce9-4b00-9d75-5f59434b5059.jpeg" class="crayons-avatar__image" alt="" width="233" height="233"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Sabith KS&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/sks/why-we-chose-go-for-our-ai-agent-platform-when-everyone-else-picked-python-4ok2" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Jul 1&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/sks/why-we-chose-go-for-our-ai-agent-platform-when-everyone-else-picked-python-4ok2" id="article-link-4045427"&gt;
          Why We Chose Go for Our AI Agent Platform (When Everyone Else Picked Python)
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/go"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;go&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/aiagents"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;aiagents&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/architecture"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;architecture&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/llm"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;llm&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
            &lt;a href="https://dev.to/sks/why-we-chose-go-for-our-ai-agent-platform-when-everyone-else-picked-python-4ok2#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            1 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
      <category>github</category>
      <category>go</category>
      <category>opensource</category>
      <category>tooling</category>
    </item>
    <item>
      <title>You Can’t Debug What You Can’t See — Observability for AI Agents</title>
      <dc:creator>Sabith KS</dc:creator>
      <pubDate>Wed, 01 Jul 2026 16:00:00 +0000</pubDate>
      <link>https://dev.to/sks/you-cant-debug-what-you-cant-see-observability-for-ai-agents-2hl0</link>
      <guid>https://dev.to/sks/you-cant-debug-what-you-cant-see-observability-for-ai-agents-2hl0</guid>
      <description>&lt;p&gt;Traditional APM can’t tell you why your agent spent $4.72 asking the same question three times.&lt;/p&gt;

&lt;p&gt;We’ve been running AI agents in production for 4 months. The hardest part isn’t building them — it’s understanding what they’re doing when they go wrong. Agents don’t crash with stack traces. They loop, hallucinate, burn tokens, and produce plausible-looking output that’s subtly wrong.&lt;/p&gt;

&lt;p&gt;Here’s the observability stack we built to see inside.&lt;/p&gt;




&lt;h2&gt;
  
  
  Why Standard Monitoring Falls Short
&lt;/h2&gt;

&lt;p&gt;Standard application monitoring answers questions like:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Is the service up? (health check)&lt;/li&gt;
&lt;li&gt;How fast are responses? (latency P50/P99)&lt;/li&gt;
&lt;li&gt;Are there errors? (error rate)&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Agent monitoring needs to answer:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;&lt;strong&gt;Why did this task cost $12 when it usually costs $0.50?&lt;/strong&gt;&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Why did the agent call the same tool 7 times?&lt;/strong&gt;&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Did the agent actually do what it said it did?&lt;/strong&gt;&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Which model is best for this task type?&lt;/strong&gt;&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Is the agent learning, or is it making the same mistakes?&lt;/strong&gt;&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These are fundamentally different questions. You can’t answer them with Prometheus counters and Grafana dashboards alone.&lt;/p&gt;




&lt;h2&gt;
  
  
  The Three Pillars for Agents
&lt;/h2&gt;

&lt;h3&gt;
  
  
  1. Traces — The Session Timeline
&lt;/h3&gt;

&lt;p&gt;Every agent session produces a trace. Not an APM trace — an &lt;strong&gt;agent trace&lt;/strong&gt; that captures the full decision history:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Session: "Investigate production latency spike"
├─ LLM Call #1: Plan generation (tokens: 1200 in, 340 out, $0.004)
├─ Tool: web_search("production latency monitoring") → 3 results
├─ LLM Call #2: Analyze results (tokens: 2100 in, 890 out, $0.008)
├─ Tool: run_shell("kubectl top pods -n production") → approved, 1.2s
├─ Sub-agent: "Check database metrics"
│ ├─ LLM Call #3: Sub-plan (tokens: 800 in, 200 out, $0.002)
│ ├─ Tool: run_shell("psql -c 'SELECT * FROM pg_stat_activity'")
│ └─ LLM Call #4: Analysis (tokens: 3400 in, 1200 out, $0.012)
├─ LLM Call #5: Synthesize findings (tokens: 4200 in, 1800 out, $0.018)
└─ Tool: send_message("Root cause: connection pool exhaustion...")

Total: 5 LLM calls, 3 tool calls, 1 sub-agent, $0.044, 47 seconds

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;We use &lt;a href="https://langfuse.com" rel="noopener noreferrer"&gt;Langfuse&lt;/a&gt; as our trace backend. Every LLM call, tool execution, and sub-agent delegation is a span. Traces nest — sub-agent traces are children of the parent trace.&lt;/p&gt;

&lt;p&gt;Trace delivery is non-blocking — we use OpenTelemetry’s batch exporter pipeline so tool execution is never waiting on a synchronous HTTP POST to the tracing backend. The OTel SDK buffers spans in memory and flushes them periodically. On pod shutdown, a graceful drain hook flushes remaining spans before the process exits. If the trace backend is temporarily unreachable, spans are dropped after the buffer fills — you lose telemetry, not availability.&lt;/p&gt;

&lt;h3&gt;
  
  
  2. Costs — The $4.72 Question
&lt;/h3&gt;

&lt;p&gt;Token costs are the unit economics of agents. We track:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight yaml"&gt;&lt;code&gt;&lt;span class="na"&gt;Per-session&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
  &lt;span class="s"&gt;Total cost $0.044&lt;/span&gt;
  &lt;span class="s"&gt;Input tokens 11,700&lt;/span&gt;
  &lt;span class="s"&gt;Output tokens 4,430&lt;/span&gt;
  &lt;span class="s"&gt;Model breakdown&lt;/span&gt;&lt;span class="err"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;claude-sonnet&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;$0.038 (4 calls)&lt;/span&gt;
    &lt;span class="na"&gt;gemini-flash&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;$0.006 (1 call, efficiency task)&lt;/span&gt;

&lt;span class="na"&gt;Per-agent (daily)&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
  &lt;span class="na"&gt;sre-copilot&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;$12.40 (28 sessions)&lt;/span&gt;
  &lt;span class="na"&gt;security-analyst&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;$3.20 (7 sessions)&lt;/span&gt;
  &lt;span class="na"&gt;dev-assistant&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;$18.90 (42 sessions)&lt;/span&gt;

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;&lt;strong&gt;Why this matters:&lt;/strong&gt; An agent that loops — calling the same tool repeatedly because it can’t make progress — burns tokens geometrically. A 10-iteration loop on Claude Sonnet costs 10× a single call. Without cost monitoring, you discover this when the invoice arrives.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Proactive guardrails:&lt;/strong&gt; Reactive alerting alone isn’t fast enough — a tight exception loop in a parallel agent can burn through dollars in seconds before a Slack webhook fires. So cost management is layered: hard iteration caps (&lt;code&gt;MaxIterations&lt;/code&gt;), per-tool call budgets (&lt;code&gt;ToolBudgets&lt;/code&gt;), and loop detection middleware that blocks identical consecutive calls all act as &lt;strong&gt;pre-flight circuit breakers&lt;/strong&gt; in the execution pipeline. Alerts are the second line of defense, not the first.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Alerting:&lt;/strong&gt; We alert when a single session exceeds 3× the rolling average cost for that agent. This catches the slower-burning anomalies — hallucination spirals, model routing errors, and gradually accumulating context — that slip past the hard limits.&lt;/p&gt;

&lt;h3&gt;
  
  
  3. Audit — The Immutable Record
&lt;/h3&gt;

&lt;p&gt;Every tool call, governance decision, and memory operation is logged to an append-only NDJSON file:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight json"&gt;&lt;code&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="nl"&gt;"ts"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"..."&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"event"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"tool_call"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"tool"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"run_shell"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"args"&lt;/span&gt;&lt;span class="p"&gt;:{&lt;/span&gt;&lt;span class="nl"&gt;"cmd"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"kubectl get pods"&lt;/span&gt;&lt;span class="p"&gt;},&lt;/span&gt;&lt;span class="nl"&gt;"decision"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"auto_approved"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"middleware_ms"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="mi"&gt;2&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="nl"&gt;"ts"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"..."&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"event"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"tool_result"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"tool"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"run_shell"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"status"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"success"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"output_bytes"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="mi"&gt;1247&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"redacted"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="kc"&gt;false&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="nl"&gt;"ts"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"..."&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"event"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"hitl_pending"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"tool"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"kubectl_apply"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"request_id"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"abc123"&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="nl"&gt;"ts"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"..."&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"event"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"hitl_approved"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"request_id"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"abc123"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"approver"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"sabith"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"latency_s"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="mf"&gt;6.2&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="nl"&gt;"ts"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"..."&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"event"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"memory_store"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"type"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"episodic"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"goal"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="s2"&gt;"investigate latency"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="nl"&gt;"confidence"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="mf"&gt;0.82&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;

&lt;/span&gt;&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The audit trail is PII-redacted. Tool outputs that contain sensitive data are sanitized before logging. You can trace what happened without exposing credentials.&lt;/p&gt;




&lt;h2&gt;
  
  
  The Diagnostic CLI
&lt;/h2&gt;

&lt;p&gt;We built &lt;code&gt;genie doctor&lt;/code&gt; (think &lt;code&gt;brew doctor&lt;/code&gt;) — a diagnostic command that checks agent health:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight console"&gt;&lt;code&gt;&lt;span class="gp"&gt;$&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;genie doctor
&lt;span class="go"&gt;
✅ Model connectivity: claude-sonnet, gemini-flash (2/2 reachable)
✅ Vector store: qdrant at localhost:6334 (healthy, 1,247 vectors)
✅ HITL: 0 pending approvals
✅ Memory: 42 episodic memories, 12 skills, 8 notes
⚠️ Langfuse: connected but 3 failed trace uploads in last hour
❌ MCP server "datadog": connection refused
   → Last successful connection: 2 hours ago
   → Try: npx -y @datadog/mcp-server --check

&lt;/span&gt;&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;One command tells you if the agent’s dependencies are healthy. No digging through logs, no checking 5 dashboards.&lt;/p&gt;




&lt;h2&gt;
  
  
  Trace Analysis — Automated Session Reviews
&lt;/h2&gt;

&lt;p&gt;Raw traces are useful for debugging individual sessions. But with 50+ agents running hundreds of sessions daily, you can’t review them all manually.&lt;/p&gt;

&lt;p&gt;We built a trace analyzer that produces automated session breakdowns:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight markdown"&gt;&lt;code&gt;&lt;span class="gu"&gt;## Session Analysis: sre-copilot (session-abc123)&lt;/span&gt;

&lt;span class="gs"&gt;**Request:**&lt;/span&gt;"Why is the API slow?"
&lt;span class="gs"&gt;**Duration:**&lt;/span&gt; 47s | &lt;span class="gs"&gt;**Cost:**&lt;/span&gt; $0.044 | &lt;span class="gs"&gt;**Tools:**&lt;/span&gt; 3 | &lt;span class="gs"&gt;**LLM calls:**&lt;/span&gt; 5

&lt;span class="gu"&gt;### Efficiency Assessment&lt;/span&gt;
&lt;span class="p"&gt;-&lt;/span&gt; ✅ No tool loops detected
&lt;span class="p"&gt;-&lt;/span&gt; ✅ Sub-agent completed successfully
&lt;span class="p"&gt;-&lt;/span&gt; ⚠️ High input token count on call #5 (4,200 tokens)
  → Consider: Trim sub-agent output before synthesis

&lt;span class="gu"&gt;### Cost Breakdown&lt;/span&gt;
| Call | Model | Tokens (in/out) | Cost |
|------|-------|-----------------|------|
| Plan | claude-sonnet | 1,200 / 340 | $0.004 |
| Analyze | claude-sonnet | 2,100 / 890 | $0.008 |
| Sub-plan | gemini-flash | 800 / 200 | $0.002 |
| Sub-analyze | claude-sonnet | 3,400 / 1,200 | $0.012 |
| Synthesize | claude-sonnet | 4,200 / 1,800 | $0.018 |

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The analyzer runs on every trace. Anomalous sessions (loops, high cost, tool errors) are flagged for human review.&lt;/p&gt;




&lt;h2&gt;
  
  
  Prometheus Metrics
&lt;/h2&gt;

&lt;p&gt;For real-time dashboards and alerting, we export metrics:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight prometheus"&gt;&lt;code&gt;&lt;span class="c"&gt;# Tool call success/failure rates&lt;/span&gt;
&lt;span class="n"&gt;toolwrap_tool_call_total&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="na"&gt;tool&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s2"&gt;"run_shell"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="na"&gt;outcome&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s2"&gt;"success"&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt; &lt;span class="mi"&gt;142&lt;/span&gt;
&lt;span class="n"&gt;toolwrap_tool_call_total&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="na"&gt;tool&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s2"&gt;"run_shell"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="na"&gt;outcome&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s2"&gt;"failure"&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt; &lt;span class="mi"&gt;3&lt;/span&gt;

&lt;span class="c"&gt;# Per-agent session costs&lt;/span&gt;
&lt;span class="n"&gt;agent_session_cost_usd&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="na"&gt;agent&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s2"&gt;"sre-copilot"&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt; &lt;span class="mf"&gt;0.044&lt;/span&gt;

&lt;span class="c"&gt;# HITL approval latency&lt;/span&gt;
&lt;span class="n"&gt;hitl_approval_latency_seconds&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="na"&gt;quantile&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s2"&gt;"0.5"&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt; &lt;span class="mf"&gt;6.2&lt;/span&gt;
&lt;span class="n"&gt;hitl_approval_latency_seconds&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="na"&gt;quantile&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s2"&gt;"0.99"&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt; &lt;span class="mf"&gt;45.1&lt;/span&gt;

&lt;span class="c"&gt;# Semantic router classification&lt;/span&gt;
&lt;span class="n"&gt;semantic_router_classification_total&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="na"&gt;route&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s2"&gt;"operations"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="na"&gt;tier&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s2"&gt;"L1"&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt; &lt;span class="mi"&gt;89&lt;/span&gt;
&lt;span class="n"&gt;semantic_router_classification_total&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="na"&gt;route&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s2"&gt;"jailbreak"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="na"&gt;tier&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s2"&gt;"L0"&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt; &lt;span class="mi"&gt;2&lt;/span&gt;

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;These feed into standard Grafana dashboards for the operations team. They don’t replace Langfuse traces — they complement them with real-time alerting.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;A warning on cardinality:&lt;/strong&gt; Keep Prometheus labels low-cardinality. Labels like &lt;code&gt;tool="run_shell"&lt;/code&gt; or &lt;code&gt;agent="sre-copilot"&lt;/code&gt; are safe — they have bounded values. Never put unique identifiers like &lt;code&gt;session_id&lt;/code&gt; or &lt;code&gt;run_id&lt;/code&gt; into Prometheus labels. A production system running thousands of agent sessions daily will cause a high-cardinality explosion that bloats memory and crashes the Prometheus server. Leave per-session details to your tracing backend (Langfuse) or structured logs.&lt;/p&gt;




&lt;h2&gt;
  
  
  What We Monitor
&lt;/h2&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;What&lt;/th&gt;
&lt;th&gt;How&lt;/th&gt;
&lt;th&gt;Alert Threshold&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;Session cost&lt;/td&gt;
&lt;td&gt;Langfuse traces&lt;/td&gt;
&lt;td&gt;&amp;gt; 3× rolling average&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Tool loop&lt;/td&gt;
&lt;td&gt;Middleware counter&lt;/td&gt;
&lt;td&gt;&amp;gt; 2 identical consecutive calls&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;HITL latency&lt;/td&gt;
&lt;td&gt;Prometheus histogram&lt;/td&gt;
&lt;td&gt;&amp;gt; 5 minutes (approval stale)&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Model errors&lt;/td&gt;
&lt;td&gt;Prometheus counter&lt;/td&gt;
&lt;td&gt;&amp;gt; 5% error rate in 5 minutes&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Vector store health&lt;/td&gt;
&lt;td&gt;
&lt;code&gt;genie doctor&lt;/code&gt; cron&lt;/td&gt;
&lt;td&gt;Connection refused&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;MCP server health&lt;/td&gt;
&lt;td&gt;Heartbeat check&lt;/td&gt;
&lt;td&gt;3 missed heartbeats&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Token burn rate&lt;/td&gt;
&lt;td&gt;Daily aggregation&lt;/td&gt;
&lt;td&gt;&amp;gt; 2× daily budget&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Audit file growth&lt;/td&gt;
&lt;td&gt;Filesystem monitor&lt;/td&gt;
&lt;td&gt;&amp;gt; 1GB/day (potential loop)&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;




&lt;h2&gt;
  
  
  Lessons Learned
&lt;/h2&gt;

&lt;ol&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Cost is your canary.&lt;/strong&gt; Sudden cost spikes almost always indicate a bug — loops, model routing errors, or unbounded context accumulation. Alert on cost first, debug second.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Traces are for debugging, metrics are for alerting.&lt;/strong&gt; Don’t try to alert on traces (too detailed) or debug with metrics (too aggregated). Use both.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Audit PII-redaction is non-negotiable.&lt;/strong&gt; Your audit trail will be queried during incident reviews. If it contains credentials or PII, your observability tool becomes a liability.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Build a diagnostic CLI.&lt;/strong&gt; &lt;code&gt;genie doctor&lt;/code&gt; saves more time than any dashboard. One command, all dependencies, clear pass/fail.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Automate trace analysis.&lt;/strong&gt; You can’t review 200 sessions a day manually. Let the analyzer flag anomalies; humans review the flags.&lt;/p&gt;&lt;/li&gt;
&lt;/ol&gt;




&lt;p&gt;&lt;em&gt;What observability tools do you use for your agent platform? I’m especially interested in cost monitoring and loop detection approaches. Find me on &lt;a href="https://github.com/sks" rel="noopener noreferrer"&gt;GitHub&lt;/a&gt; or &lt;a href="https://linkedin.com/in/sabithks" rel="noopener noreferrer"&gt;LinkedIn&lt;/a&gt;.&lt;/em&gt;&lt;/p&gt;




&lt;blockquote&gt;
&lt;p&gt;🚀 &lt;strong&gt;We’re building AI-powered SRE at StackGen.&lt;/strong&gt; If you’re tired of 3 AM pages and want AI agents that triage incidents, run diagnostics, and draft RCA reports — check out &lt;a href="https://ai.stackgen.com" rel="noopener noreferrer"&gt;ai.stackgen.com&lt;/a&gt; and try our new SRE offering.&lt;/p&gt;
&lt;/blockquote&gt;

</description>
      <category>observability</category>
      <category>aiagents</category>
      <category>langfuse</category>
      <category>monitoring</category>
    </item>
    <item>
      <title>The HITL Paradox — When Human Approval Makes Agents Worse</title>
      <dc:creator>Sabith KS</dc:creator>
      <pubDate>Wed, 01 Jul 2026 14:00:00 +0000</pubDate>
      <link>https://dev.to/sks/the-hitl-paradox-when-human-approval-makes-agents-worse-1095</link>
      <guid>https://dev.to/sks/the-hitl-paradox-when-human-approval-makes-agents-worse-1095</guid>
      <description>&lt;p&gt;Human-in-the-loop (HITL) is supposed to make agents safer. Put a human between the agent and the dangerous action. Simple.&lt;/p&gt;

&lt;p&gt;In practice, HITL has a paradox: &lt;strong&gt;too much approval kills productivity, too little kills safety, and the wrong amount creates a false sense of security.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;We deployed HITL for our agent runtime and watched three failure modes emerge. Here’s what happened and how we fixed each one.&lt;/p&gt;




&lt;h2&gt;
  
  
  Failure Mode 1: Approval Fatigue
&lt;/h2&gt;

&lt;p&gt;Our first HITL deployment required approval for every tool call. Shell commands, web searches, memory reads — everything needed a human click.&lt;/p&gt;

&lt;p&gt;Within two days, operators were auto-approving everything without reading the details. The approval popup became muscle memory: see popup → click approve → continue.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The data:&lt;/strong&gt; We tracked approval latency. In week 1, operators spent an average of 8 seconds reviewing each request. By week 2, it was under 2 seconds. They weren’t reviewing — they were dismissing.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Why this is worse than no HITL:&lt;/strong&gt; Operators now believe they have a safety net. They don’t. The safety net is a rubber stamp. But everyone — operators, managers, auditors — thinks the system is reviewed because “human approval is required.”&lt;/p&gt;

&lt;h3&gt;
  
  
  The Fix: Risk-Based Classification
&lt;/h3&gt;

&lt;p&gt;We classified tools into three tiers:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight toml"&gt;&lt;code&gt;&lt;span class="nn"&gt;[hitl]&lt;/span&gt;
&lt;span class="c"&gt;# Never needs approval — safe, read-only, or internal&lt;/span&gt;
&lt;span class="py"&gt;always_allowed&lt;/span&gt; &lt;span class="p"&gt;=&lt;/span&gt; &lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="s"&gt;"web_search"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s"&gt;"memory_*"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s"&gt;"read_*"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s"&gt;"discover_skills"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s"&gt;"note"&lt;/span&gt;&lt;span class="p"&gt;]&lt;/span&gt;

&lt;span class="c"&gt;# Needs approval — can modify external state&lt;/span&gt;
&lt;span class="c"&gt;# (This is the default for any tool not in always_allowed)&lt;/span&gt;

&lt;span class="c"&gt;# Never allowed — blocked entirely, regardless of approval&lt;/span&gt;
&lt;span class="py"&gt;denied_tools&lt;/span&gt; &lt;span class="p"&gt;=&lt;/span&gt; &lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="s"&gt;"bash"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s"&gt;"shell_*"&lt;/span&gt;&lt;span class="p"&gt;]&lt;/span&gt;

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;An important subtlety: &lt;strong&gt;these lists match tool names, not shell command strings.&lt;/strong&gt; &lt;code&gt;denied_tools = ["bash"]&lt;/code&gt; blocks the tool named &lt;code&gt;bash&lt;/code&gt; from being invoked at all — it doesn’t do regex matching against command arguments passed to &lt;code&gt;run_shell&lt;/code&gt;. String-level blocklisting on shell primitives (e.g., blocking “rm” as a substring) is fundamentally unsafe — any sufficiently creative LLM can bypass it via base64 encoding, variable interpolation, or aliasing. Instead, the HITL gate operates at the &lt;strong&gt;tool invocation boundary&lt;/strong&gt; : &lt;code&gt;run_shell&lt;/code&gt; as a whole requires human approval, and the human sees the full command in the approval request. If you need granular command-level control, the right approach is typed Go API clients (e.g., a Kubernetes Go client with RBAC) instead of raw shell access.&lt;/p&gt;

&lt;p&gt;Only state-modifying tools require approval. Read-only operations auto-approve. Destructive tool names hard-block regardless of approval.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Result:&lt;/strong&gt; Approval requests dropped by 70%. Operators now see 3-5 requests per task instead of 20+. Each request is meaningful — they actually read them.&lt;/p&gt;




&lt;h2&gt;
  
  
  Failure Mode 2: The &lt;code&gt;always_allowed = ["*"]&lt;/code&gt; Escape Hatch
&lt;/h2&gt;

&lt;p&gt;Some teams set &lt;code&gt;always_allowed = ["*"]&lt;/code&gt; to skip all approvals. They’d been burned by approval fatigue and decided HITL wasn’t worth the friction.&lt;/p&gt;

&lt;p&gt;This defeats the entire purpose of governance. An agent with &lt;code&gt;always_allowed = ["*"]&lt;/code&gt; can execute any tool without review — including shell commands on production servers.&lt;/p&gt;

&lt;h3&gt;
  
  
  The Fix: Guardrails on the Guardrails
&lt;/h3&gt;

&lt;p&gt;We added warnings when &lt;code&gt;always_allowed&lt;/code&gt; contains wildcards:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;⚠️ Warning: always_allowed contains "*" — all tools will bypass 
HITL approval. This includes run_shell, kubectl, and other 
state-modifying tools. Are you sure?

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The real safeguard is the &lt;code&gt;denied_tools&lt;/code&gt; list — even when &lt;code&gt;always_allowed = ["*"]&lt;/code&gt;, any tool in &lt;code&gt;denied_tools&lt;/code&gt; is hard-blocked. So teams that want minimal friction can set &lt;code&gt;always_allowed = ["*"]&lt;/code&gt; while keeping the most dangerous tool names denied:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight toml"&gt;&lt;code&gt;&lt;span class="nn"&gt;[hitl]&lt;/span&gt;
&lt;span class="py"&gt;always_allowed&lt;/span&gt; &lt;span class="p"&gt;=&lt;/span&gt; &lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="s"&gt;"*"&lt;/span&gt;&lt;span class="p"&gt;]&lt;/span&gt;
&lt;span class="py"&gt;denied_tools&lt;/span&gt; &lt;span class="p"&gt;=&lt;/span&gt; &lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="s"&gt;"bash"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s"&gt;"shell_*"&lt;/span&gt;&lt;span class="p"&gt;]&lt;/span&gt;

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;This gives teams the fast workflow they want while maintaining hard gates on the most dangerous operations.&lt;/p&gt;




&lt;h2&gt;
  
  
  Failure Mode 3: Blocking on Approval Halts Everything
&lt;/h2&gt;

&lt;p&gt;Early HITL was synchronous — the agent stopped working and waited for approval. If the operator was in a meeting, the agent sat idle for 45 minutes waiting for a click.&lt;/p&gt;

&lt;p&gt;For a single approval, this is annoying. For a task requiring 5 approvals across different tools, the total wait time could exceed the task’s useful lifetime.&lt;/p&gt;

&lt;h3&gt;
  
  
  The Fix: Asynchronous Approval
&lt;/h3&gt;

&lt;p&gt;HITL approval is now asynchronous:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;Agent encounters a tool that requires approval&lt;/li&gt;
&lt;li&gt;Stores the pending request in the database with a TTL (default: 30 minutes)&lt;/li&gt;
&lt;li&gt;Sends a notification via the event bus (Slack, web UI, AG-UI protocol)&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Continues working on other parts of the task&lt;/strong&gt;&lt;/li&gt;
&lt;li&gt;When approved, the tool executes and results flow back&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;The agent doesn’t block. If it has parallel sub-tasks, it works on those while waiting. If there’s nothing else to do, it waits — but the user sees a clear “waiting for approval” status, not a mysteriously silent agent.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;A note on state drift:&lt;/strong&gt; Asynchronous approval introduces a classic distributed systems risk — the environment state may change between when the agent formulated the tool call and when a human approves it 20 minutes later. We mitigate this with short approval TTLs (stale approvals auto-expire via a background reaper) and session-scoped approval caching that expires entries after 10 minutes, ensuring that long-deferred approvals don’t execute against a drifted environment without the agent re-evaluating.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Batch operations:&lt;/strong&gt; Operators can view multiple pending requests at once via the &lt;code&gt;ListPending&lt;/code&gt; API, grouped by tool name:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Pending approvals (3):
  [✅ Approve All] [❌ Reject All]

  🔧 run_shell: kubectl get pods -n production
  🔧 run_shell: kubectl describe pod api-server-7d8f
  🔧 run_shell: kubectl logs api-server-7d8f --tail=50

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;One important guardrail: the “Approve All” pattern works well for &lt;strong&gt;read-only investigation commands&lt;/strong&gt; like the above. For state-modifying operations, each approval should be reviewed individually — otherwise you recreate the rubber-stamp problem at a higher abstraction level.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;What happens on rejection?&lt;/strong&gt; When a human rejects a tool call (with or without feedback), the middleware returns an &lt;code&gt;ErrToolCallRejected&lt;/code&gt; error to the agent’s context. This isn’t a hard cancellation — the LLM receives the rejection as a tool error and can replan. If the human provided feedback (e.g., “use the staging cluster instead”), the agent sees it and can adjust its approach. This gives operators a conversational override, not just a binary approve/deny gate.&lt;/p&gt;




&lt;h2&gt;
  
  
  The Hidden Bug: HITL Bypass on Sub-Agents
&lt;/h2&gt;

&lt;p&gt;This was a real security issue. When our agent delegated to sub-agents via ReAcTree, the sub-agent’s tools were bound directly from the registry — &lt;strong&gt;without the HITL middleware wrapper&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;A sub-agent could run &lt;code&gt;run_shell&lt;/code&gt; without approval, even though the parent agent required it.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Why it happened:&lt;/strong&gt; The sub-agent tool binding was written before HITL existed. When we added HITL, we wrapped the parent’s tools but forgot the sub-agent delegation path.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The fix:&lt;/strong&gt; All tool binding — parent, sub-agent, plan-step, fallback — goes through the same &lt;code&gt;ToolWrapSvc&lt;/code&gt; middleware chain. One path. One governance stack. No exceptions.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The lesson:&lt;/strong&gt; When you add a governance layer, you must audit every tool execution path. The path you forget is the one that gets exploited.&lt;/p&gt;




&lt;h2&gt;
  
  
  What Good HITL Looks Like
&lt;/h2&gt;

&lt;p&gt;After three iterations, here’s our current model:&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Tool Type&lt;/th&gt;
&lt;th&gt;Behavior&lt;/th&gt;
&lt;th&gt;Example&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;Read-only&lt;/td&gt;
&lt;td&gt;Auto-approve&lt;/td&gt;
&lt;td&gt;
&lt;code&gt;web_search&lt;/code&gt;, &lt;code&gt;memory_search&lt;/code&gt;, &lt;code&gt;read_file&lt;/code&gt;
&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Informational&lt;/td&gt;
&lt;td&gt;Auto-approve&lt;/td&gt;
&lt;td&gt;
&lt;code&gt;discover_skills&lt;/code&gt;, &lt;code&gt;list_pods&lt;/code&gt;
&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;State-modifying&lt;/td&gt;
&lt;td&gt;Require approval&lt;/td&gt;
&lt;td&gt;
&lt;code&gt;run_shell&lt;/code&gt;, &lt;code&gt;commit_code&lt;/code&gt;, &lt;code&gt;create_pr&lt;/code&gt;
&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Destructive&lt;/td&gt;
&lt;td&gt;Hard deny&lt;/td&gt;
&lt;td&gt;
&lt;code&gt;rm -rf&lt;/code&gt;, &lt;code&gt;kubectl delete namespace&lt;/code&gt;, &lt;code&gt;DROP TABLE&lt;/code&gt;
&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Memory writes&lt;/td&gt;
&lt;td&gt;Exempt (not state)&lt;/td&gt;
&lt;td&gt;
&lt;code&gt;memory_manage&lt;/code&gt;, &lt;code&gt;note&lt;/code&gt;
&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;&lt;strong&gt;The exemption for memory writes&lt;/strong&gt; is important. Memory tools modify the agent’s internal state, not external systems. Requiring approval for every &lt;code&gt;memory_manage&lt;/code&gt; call would trigger approval fatigue without adding safety — the agent is only modifying its own notes.&lt;/p&gt;




&lt;h2&gt;
  
  
  Metrics That Matter
&lt;/h2&gt;

&lt;p&gt;Track these to know if your HITL system is working:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;
&lt;strong&gt;Approval latency&lt;/strong&gt; — If it drops below 3 seconds, operators aren’t reading requests&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Approval rate&lt;/strong&gt; — If it’s above 95%, you’re probably approving too aggressively&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Rejection rate&lt;/strong&gt; — If it’s below 1%, either your agent is perfect or nobody is paying attention&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Time-to-abandon&lt;/strong&gt; — How long before operators set &lt;code&gt;always_allowed = ["*"]&lt;/code&gt;
&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;Our current numbers: ~6 second average approval latency, 88% approval rate, 7% rejection rate, 5% auto-expired (operator didn’t respond in time).&lt;/p&gt;




&lt;h2&gt;
  
  
  Lessons Learned
&lt;/h2&gt;

&lt;ol&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Less approval is more safety.&lt;/strong&gt; Fewer, higher-signal approval requests get more attention than constant popups.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Classify tools by risk, not by category.&lt;/strong&gt; Not all shell commands are dangerous. &lt;code&gt;kubectl get pods&lt;/code&gt; is read-only; &lt;code&gt;kubectl delete pod&lt;/code&gt; is not.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Make approval asynchronous.&lt;/strong&gt; Synchronous blocking kills agent productivity and operator patience.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Audit every tool path.&lt;/strong&gt; HITL that applies to 90% of tool calls creates a false sense of security. The 10% that bypasses it is where the risk lives.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Memory tools are not external state.&lt;/strong&gt; Don’t require approval for internal memory operations — it’s noise that drowns out real signals.&lt;/p&gt;&lt;/li&gt;
&lt;/ol&gt;




&lt;p&gt;&lt;em&gt;How does your team handle the approval fatigue problem? I’d love to hear about alternative approaches. Find me on &lt;a href="https://github.com/sks" rel="noopener noreferrer"&gt;GitHub&lt;/a&gt; or &lt;a href="https://linkedin.com/in/sabithks" rel="noopener noreferrer"&gt;LinkedIn&lt;/a&gt;.&lt;/em&gt;&lt;/p&gt;




&lt;blockquote&gt;
&lt;p&gt;🚀 &lt;strong&gt;We’re building AI-powered SRE at StackGen.&lt;/strong&gt; If you’re tired of 3 AM pages and want AI agents that triage incidents, run diagnostics, and draft RCA reports — check out &lt;a href="https://ai.stackgen.com" rel="noopener noreferrer"&gt;ai.stackgen.com&lt;/a&gt; and try our new SRE offering.&lt;/p&gt;
&lt;/blockquote&gt;

</description>
      <category>hitl</category>
      <category>aiagents</category>
      <category>ux</category>
      <category>governance</category>
    </item>
    <item>
      <title>Pensieve — Memory Management for AI Agents That Actually Forget</title>
      <dc:creator>Sabith KS</dc:creator>
      <pubDate>Wed, 01 Jul 2026 12:00:00 +0000</pubDate>
      <link>https://dev.to/sks/pensieve-memory-management-for-ai-agents-that-actually-forget-3d3c</link>
      <guid>https://dev.to/sks/pensieve-memory-management-for-ai-agents-that-actually-forget-3d3c</guid>
      <description>&lt;p&gt;Your agent remembers everything. That’s a bug, not a feature.&lt;/p&gt;

&lt;p&gt;We’ve all seen it: you give an agent a task, it retrieves 40 “relevant” context chunks from a vector store, stuffs them into a 128K context window, and produces a response that’s technically accurate but practically useless — because 35 of those chunks were irrelevant, stale, or contradictory.&lt;/p&gt;

&lt;p&gt;We built a memory system called Pensieve that handles this differently. Instead of “remember everything and search later,” Pensieve manages &lt;strong&gt;four distinct memory types&lt;/strong&gt; , with automatic decay, importance scoring, and self-pruning. This post walks through the architecture, the algorithms, and the production lessons.&lt;/p&gt;




&lt;h2&gt;
  
  
  Why Naive RAG Fails for Agents
&lt;/h2&gt;

&lt;p&gt;RAG (Retrieval-Augmented Generation) works brilliantly for Q&amp;amp;A systems. You have a corpus of documents, you embed them, and you retrieve the most similar chunks for a user question.&lt;/p&gt;

&lt;p&gt;Agents are different:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Agents generate memories at runtime.&lt;/strong&gt; Every task produces new experiences — what worked, what failed, what the user corrected. The corpus grows with every interaction.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Agent memories have temporal relevance.&lt;/strong&gt; “The staging API was down” was true yesterday. Retrieving it today makes the agent avoid an API that’s working fine.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Agent memories have quality variation.&lt;/strong&gt; Successful task completions, failed attempts, hallucinated outputs, and user corrections all go into the same store. Quality varies wildly.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Agents need structured recall, not just similarity.&lt;/strong&gt; “What skills do I have for Kubernetes troubleshooting?” is a different retrieval mode than “find text similar to ‘pod crashloopbackoff’.”&lt;/p&gt;&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;Standard RAG treats all chunks equally — same embedding, same retrieval, same ranking. Agents need &lt;strong&gt;curated, time-aware, quality-gated memory.&lt;/strong&gt;&lt;/p&gt;




&lt;h2&gt;
  
  
  The Four Memory Types
&lt;/h2&gt;

&lt;p&gt;Pensieve manages four distinct memory stores, each with different lifecycle and retrieval semantics:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;┌──────────────────────────────────────────────────────────┐
│ Agent Memory │
├──────────────┬──────────────┬──────────┬─────────────────┤
│ Working │ Episodic │ Notes │ Skills │
│ Memory │ Memory │ │ │
├──────────────┼──────────────┼──────────┼─────────────────┤
│ Session │ Goal-keyed │ Cross- │ Reusable │
│ blackboard │ experiences │ session │ procedures │
│ │ │ facts │ │
├──────────────┼──────────────┼──────────┼─────────────────┤
│ Lifetime: │ Lifetime: │ Lifetime:│ Lifetime: │
│ Single task │ Decays over │ Until │ Until │
│ │ ~2 weeks │ deleted │ deprecated │
├──────────────┼──────────────┼──────────┼─────────────────┤
│ No embedding │ Vector + │ Key- │ Semantic │
│ (key-value) │ weighted │ value │ search │
│ │ retrieval │ lookup │ │
└──────────────┴──────────────┴──────────┴─────────────────┘

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  1. Working Memory — The Session Blackboard
&lt;/h3&gt;

&lt;p&gt;Working memory is a key-value store scoped to a single task execution. When a parent agent delegates to multiple sub-agents via ReAcTree, working memory is the shared blackboard:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Parent: "Investigate production outage"
  ├─ Sub-agent 1: writes working_memory["log_analysis"] = "OOM killer triggered at 14:32"
  ├─ Sub-agent 2: writes working_memory["metric_summary"] = "Memory usage spiked from 2GB to 8GB"
  └─ Parent reads both entries to synthesize RCA

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Working memory dies when the task completes. It’s not persisted. Think of it as function-scoped variables.&lt;/p&gt;

&lt;h3&gt;
  
  
  2. Episodic Memory — Experiences with Expiration Dates
&lt;/h3&gt;

&lt;p&gt;Episodic memory stores &lt;strong&gt;what happened during past tasks&lt;/strong&gt; — the goal, the approach, the outcome, and what was learned. It’s the agent’s autobiography.&lt;/p&gt;

&lt;p&gt;The critical design choice: &lt;strong&gt;not all episodes are worth remembering.&lt;/strong&gt;&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight go"&gt;&lt;code&gt;&lt;span class="c"&gt;// Store episodes with explicit status tagging&lt;/span&gt;
&lt;span class="k"&gt;if&lt;/span&gt; &lt;span class="o"&gt;!&lt;/span&gt;&lt;span class="n"&gt;output&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;looksLikeError&lt;/span&gt;&lt;span class="p"&gt;()&lt;/span&gt; &lt;span class="p"&gt;{&lt;/span&gt;
    &lt;span class="n"&gt;episodic&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Store&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;ctx&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;Episode&lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;
        &lt;span class="n"&gt;Goal&lt;/span&gt;&lt;span class="o"&gt;:&lt;/span&gt; &lt;span class="n"&gt;goal&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
        &lt;span class="n"&gt;Trajectory&lt;/span&gt;&lt;span class="o"&gt;:&lt;/span&gt; &lt;span class="n"&gt;trajectory&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
        &lt;span class="n"&gt;Status&lt;/span&gt;&lt;span class="o"&gt;:&lt;/span&gt; &lt;span class="n"&gt;EpisodePending&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="c"&gt;// promoted to Success on user validation&lt;/span&gt;
        &lt;span class="n"&gt;Importance&lt;/span&gt;&lt;span class="o"&gt;:&lt;/span&gt; &lt;span class="n"&gt;scoreImportance&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;ctx&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;goal&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;trajectory&lt;/span&gt;&lt;span class="p"&gt;),&lt;/span&gt;
    &lt;span class="p"&gt;})&lt;/span&gt;
&lt;span class="p"&gt;}&lt;/span&gt;

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;This is &lt;strong&gt;status-aware storage&lt;/strong&gt;. Episodes start as &lt;code&gt;pending&lt;/code&gt; and are promoted to &lt;code&gt;success&lt;/code&gt; only after user validation (e.g., a 👍 emoji reaction). Failed tasks are stored separately with verbal reflections (more on this below). Raw error traces never pollute the memory — only synthesized lessons.&lt;/p&gt;

&lt;p&gt;&lt;em&gt;(In the &lt;a href="https://dev.to/blog/2026/07/01/reactree-bugs/"&gt;ReAcTree bugs post&lt;/a&gt;, I mentioned that naively storing failures poisoned the agent. That’s true for raw execution traces — saving a 50-line stack trace of a timeout pollutes context. We evolved this: we still drop the raw failure log, but we pass the event to a &lt;code&gt;FailureReflector&lt;/code&gt; to synthesize a concise, one-sentence lesson, which we safely store with a &lt;code&gt;failure&lt;/code&gt; status and a verbal reflection.)&lt;/em&gt;&lt;/p&gt;

&lt;h4&gt;
  
  
  Retrieval Scoring
&lt;/h4&gt;

&lt;p&gt;Episodic memories are ranked using a &lt;strong&gt;three-signal weighted score&lt;/strong&gt; that combines semantic similarity, temporal recency, and importance:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;final_score = 0.4 × cosine_similarity + 0.3 × recency + 0.3 × importance

recency = e^(-0.01 × hours_since_created) // exponential decay, [0,1]
importance = importance_score / 10 // normalized from 1-10 to [0,1]

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;A memory from 1 hour ago has a recency score of ~0.99. A memory from 1 week (168 hours) ago has a recency score of ~0.19. After ~2 weeks, the recency component effectively drops to zero.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Why three signals instead of two?&lt;/strong&gt; With only recency and importance (the naive approach), you get a mathematical problem: if recency dominates (say, 60% weight), then &lt;em&gt;any&lt;/em&gt; memory older than 4-5 days will always be outranked by a completely routine memory from the last hour — even a critical production incident. By adding cosine similarity as the largest component (40%), the system surfaces memories that are &lt;em&gt;semantically relevant to the current task&lt;/em&gt; first, then uses recency and importance as tiebreakers. A week-old production incident retrieves strongly when the current task involves a similar failure mode.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Why this matters:&lt;/strong&gt; An agent that investigated a DNS issue last week shouldn’t treat that experience the same as investigating the same issue 5 minutes ago. But it also shouldn’t forget a critical production incident just because a routine health check happened more recently. The three-way weighting handles both cases.&lt;/p&gt;

&lt;h4&gt;
  
  
  Importance Scoring
&lt;/h4&gt;

&lt;p&gt;When an episode is stored, a lightweight LLM call scores it 1-10:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;"Deployed a hotfix to production under time pressure" → 9/10
"Ran a routine health check, everything was green" → 2/10

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Unscored episodes (importance = 0) receive a neutral 0.5 weight so they don’t dominate or disappear from retrieval.&lt;/p&gt;

&lt;h3&gt;
  
  
  3. Notes — Cross-Session Persistence
&lt;/h3&gt;

&lt;p&gt;Notes are simple key-value pairs that persist across sessions:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;notes["user_preference_timezone"] = "US/Pacific"
notes["team_oncall_rotation"] = "PagerDuty schedule ID: P123ABC"
notes["k8s_cluster_prod"] = "us-east-1, EKS 1.31, 47 nodes"

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Notes don’t decay. They’re for facts that change rarely and apply broadly. The agent manages its own notes — it can create, read, update, and delete them as tool calls.&lt;/p&gt;

&lt;h3&gt;
  
  
  4. Skills — Reusable Procedures
&lt;/h3&gt;

&lt;p&gt;Skills are structured documents that describe &lt;strong&gt;how to do something&lt;/strong&gt; — step-by-step procedures with failure handling:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight markdown"&gt;&lt;code&gt;&lt;span class="gh"&gt;# Skill: Kubernetes Pod Crashloop Triage&lt;/span&gt;

&lt;span class="gu"&gt;## Steps&lt;/span&gt;
&lt;span class="p"&gt;1.&lt;/span&gt; Get pod status: &lt;span class="sb"&gt;`kubectl get pods -n {namespace} | grep CrashLoopBackOff`&lt;/span&gt;
&lt;span class="p"&gt;2.&lt;/span&gt; Check pod events: &lt;span class="sb"&gt;`kubectl describe pod {pod_name} -n {namespace}`&lt;/span&gt;
&lt;span class="p"&gt;3.&lt;/span&gt; Read last 100 log lines: &lt;span class="sb"&gt;`kubectl logs {pod_name} -n {namespace} --tail=100`&lt;/span&gt;
&lt;span class="p"&gt;4.&lt;/span&gt; Check resource limits vs actual usage
&lt;span class="p"&gt;5.&lt;/span&gt; Check if recent deployments changed the image or config

&lt;span class="gu"&gt;## Common Causes&lt;/span&gt;
&lt;span class="p"&gt;-&lt;/span&gt; OOM kills → check memory limits
&lt;span class="p"&gt;-&lt;/span&gt; Missing config/secrets → check configmap/secret mounts
&lt;span class="p"&gt;-&lt;/span&gt; Image pull failures → check registry access

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Skills are stored on the filesystem and indexed in a vector store for semantic discovery. When an agent starts a task, it searches for relevant skills and loads them into context.&lt;/p&gt;




&lt;h2&gt;
  
  
  The Self-Pruning Agent
&lt;/h2&gt;

&lt;p&gt;Here’s where it gets interesting. In most agent frameworks, you (the developer) manage memory — you decide what to store, what to retrieve, how much context to inject.&lt;/p&gt;

&lt;p&gt;In Pensieve, the &lt;strong&gt;agent manages its own memory budget.&lt;/strong&gt; It has memory management tools:&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Tool&lt;/th&gt;
&lt;th&gt;Purpose&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;memory_search&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;Semantic search across episodic memories&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;memory_manage&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;Save, update, or delete memories&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;note&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;Read or write persistent notes&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;read_notes&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;List all notes&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;discover_skills&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;Search for relevant skills&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;load_skill&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;Load a skill into working context&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;The agent’s system prompt includes instructions to manage its context proactively:&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;&lt;em&gt;“Before starting work, search memory for relevant past experiences and available skills. If your context is getting large, offload resolved information to notes and prune completed sub-task context.”&lt;/em&gt;&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;&lt;strong&gt;Why this works:&lt;/strong&gt; The agent knows what information it needs for the current step better than any static retrieval algorithm. By giving it memory management tools, we let it curate its own context window.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Why this is risky:&lt;/strong&gt; An agent can delete useful memories or fail to store important ones. We mitigate this with audit logging — every memory operation is logged to an immutable audit trail, so we can reconstruct what happened. Sub-agents get a restricted view: they can read and search memory, but writes flow through the parent agent’s middleware stack with the same governance controls as any other tool call.&lt;/p&gt;




&lt;h2&gt;
  
  
  The Learning Loop
&lt;/h2&gt;

&lt;p&gt;After every completed task, a background process evaluates whether the experience is worth remembering as a reusable &lt;strong&gt;skill&lt;/strong&gt; :&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Task completes
  → Novelty scoring (LLM rates 1-10)
  → If novelty ≥ 7:
      → Distill into structured skill document
      → Semantic dedup check (similarity ≥ 0.8 → merge or skip)
      → Store to filesystem + vector index

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;This is &lt;strong&gt;post-session skill distillation&lt;/strong&gt;. The agent doesn’t learn during the task — it learns after, asynchronously, without blocking the user.&lt;/p&gt;

&lt;p&gt;Example: An agent successfully triages a Redis connection storm for the first time. The learning loop:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;Scores it 8/10 novelty (agent hasn’t handled Redis issues before)&lt;/li&gt;
&lt;li&gt;Distills the approach into a skill document&lt;/li&gt;
&lt;li&gt;Checks if a similar skill exists (none found)&lt;/li&gt;
&lt;li&gt;Stores it as &lt;code&gt;dynamic_skills/redis_connection_storm_triage.md&lt;/code&gt;
&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;Next time a Redis issue comes up, the agent finds this skill via &lt;code&gt;discover_skills&lt;/code&gt; and follows the documented procedure.&lt;/p&gt;




&lt;h2&gt;
  
  
  Failure Learning
&lt;/h2&gt;

&lt;p&gt;Successes teach you what to do. Failures teach you what to avoid. We capture both.&lt;/p&gt;

&lt;p&gt;When an agent fails a task (timeout, too many errors, explicit failure status), a &lt;code&gt;FailureReflector&lt;/code&gt; generates a verbal reflection:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Goal: "Scale the production database replica set"
Status: Failed
Reflection: "Attempted to modify replica count without checking 
if the cluster was in maintenance mode. The API returned 403 
Forbidden. Next time, verify cluster status before making 
scaling changes."

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;These failure reflections are stored as episodic memories with a ⚠️ prefix. When the agent encounters a similar task, it retrieves both successful experiences and past failures:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight markdown"&gt;&lt;code&gt;&lt;span class="gu"&gt;## Relevant Experience&lt;/span&gt;
✅ Successfully scaled Redis cluster by updating replica count 
   after confirming maintenance window (2 days ago)

⚠️ Failed to scale database replica set — didn't check 
   maintenance mode first, got 403 (5 days ago)

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The agent sees what worked &lt;strong&gt;and&lt;/strong&gt; what didn’t. This is inspired by &lt;a href="https://arxiv.org/abs/2303.11366" rel="noopener noreferrer"&gt;Reflexion&lt;/a&gt; — verbal reinforcement without weight updates.&lt;/p&gt;




&lt;h2&gt;
  
  
  Daily Wisdom Consolidation
&lt;/h2&gt;

&lt;p&gt;Individual episodic memories accumulate. Over time, retrieving them all is expensive and noisy.&lt;/p&gt;

&lt;p&gt;Once per day, an &lt;code&gt;EpisodeConsolidator&lt;/code&gt; reads recent episodes and summarizes them into &lt;strong&gt;wisdom notes&lt;/strong&gt; — concise bullet-point lessons:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight markdown"&gt;&lt;code&gt;&lt;span class="gu"&gt;## Consolidated Lessons (July 1, 2026)&lt;/span&gt;
&lt;span class="p"&gt;
-&lt;/span&gt; When scaling database replicas, always check cluster 
  maintenance mode status first (learned from failed attempt)
&lt;span class="p"&gt;-&lt;/span&gt; Redis connection storms usually indicate connection pool 
  exhaustion in the application, not Redis server issues
&lt;span class="p"&gt;-&lt;/span&gt; PagerDuty incident creation requires the service_id field; 
  use the pd_service_list tool to find it first

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Wisdom notes are injected into the agent’s system prompt as a &lt;code&gt;## Consolidated Lessons&lt;/code&gt; section, capped at the 2-3 most recent notes. They provide distilled experience without the noise of individual episodes.&lt;/p&gt;

&lt;p&gt;To prevent the wisdom section from growing unboundedly, the consolidator naturally limits itself: it retrieves a small, fixed window of recent wisdom notes per prompt injection. Older notes still exist in storage but aren’t injected — they’ve served their purpose by informing the agent during their active window, and the lessons they encode are either still relevant (and get re-learned) or have become stale. The consolidation job also deletes the raw episodes it summarized from the vector store, so the episodic memory stays clean.&lt;/p&gt;




&lt;h2&gt;
  
  
  The PII Problem
&lt;/h2&gt;

&lt;p&gt;Agent memories contain user conversations, tool outputs, API responses — all potentially containing PII (names, emails, IP addresses, tokens).&lt;/p&gt;

&lt;p&gt;We run PII redaction &lt;strong&gt;before&lt;/strong&gt; persisting any memory:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight go"&gt;&lt;code&gt;&lt;span class="c"&gt;// All memory storage goes through PII redaction&lt;/span&gt;
&lt;span class="k"&gt;func&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;s&lt;/span&gt; &lt;span class="o"&gt;*&lt;/span&gt;&lt;span class="n"&gt;Store&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="n"&gt;Save&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;ctx&lt;/span&gt; &lt;span class="n"&gt;context&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Context&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;req&lt;/span&gt; &lt;span class="n"&gt;SaveRequest&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="kt"&gt;error&lt;/span&gt; &lt;span class="p"&gt;{&lt;/span&gt;
    &lt;span class="n"&gt;req&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Content&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;pii&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Redact&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;req&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Content&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
    &lt;span class="n"&gt;req&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Goal&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;pii&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Redact&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;req&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Goal&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
    &lt;span class="k"&gt;return&lt;/span&gt; &lt;span class="n"&gt;s&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;backend&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Save&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;ctx&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;req&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
&lt;span class="p"&gt;}&lt;/span&gt;

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The &lt;code&gt;pii.Redact()&lt;/code&gt; function strips emails, bearer tokens, API keys, and other sensitive patterns before persistence.&lt;/p&gt;

&lt;p&gt;There’s a domain-specific tension here: our product is an SRE copilot, and in SRE contexts, IP addresses and hostnames are critical telemetry. If the agent remembers “the outage was caused by a rogue pod on node [REDACTED],” the memory is functionally useless for future debugging. We handle this by allowlisting internal RFC-1918 subnets and using deterministic tokenization for external addresses — &lt;code&gt;203.0.113.42&lt;/code&gt; consistently becomes &lt;code&gt;[EXT_IP_A]&lt;/code&gt; across memories, so the agent can still learn network topology patterns without storing regulated PII in the vector store.&lt;/p&gt;




&lt;h2&gt;
  
  
  Architecture Summary
&lt;/h2&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;┌─────────────────────────────────────────────────────┐
│ Agent Prompt │
│ ├── Consolidated wisdom (injected automatically) │
│ ├── Episodic memories (retrieved per-goal) │
│ ├── Skills (loaded on demand) │
│ └── Notes (read via tool call) │
├─────────────────────────────────────────────────────┤
│ Memory Management Tools │
│ memory_search, memory_manage, note, │
│ discover_skills, load_skill │
├─────────────────────────────────────────────────────┤
│ Storage Layer │
│ ├── Vector store (Qdrant) — embeddings │
│ ├── Filesystem — skills, notes │
│ └── PII redaction — applied before persist │
├─────────────────────────────────────────────────────┤
│ Background Processes │
│ ├── Learning loop — skill distillation │
│ ├── Wisdom consolidation — daily digest │
│ └── Memory decay — exponential time-based │
└─────────────────────────────────────────────────────┘

&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;






&lt;h2&gt;
  
  
  Lessons Learned
&lt;/h2&gt;

&lt;ol&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Memory is a data quality problem.&lt;/strong&gt; Treat it like a database — validate before insert, enforce schema, handle duplicates.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Temporal decay is essential.&lt;/strong&gt; Agents operate in changing environments. Yesterday’s truths can be today’s hallucinations.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Let agents manage their own context.&lt;/strong&gt; Static retrieval algorithms can’t know what the agent needs at each step. Give it memory tools and let it curate.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Separate memory types for different lifetimes.&lt;/strong&gt; Working memory (seconds), episodic (weeks), notes (permanent), skills (permanent until deprecated). Mixing lifetimes causes stale context pollution.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Failure memories are as valuable as success memories&lt;/strong&gt; — but they must be clearly labeled. An agent should learn “don’t do X” without concluding “X is impossible.”&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;PII redaction is non-negotiable.&lt;/strong&gt; Agents process sensitive data. Memory stores are search targets. Unredacted PII in a vector store is a compliance incident waiting to happen.&lt;/p&gt;&lt;/li&gt;
&lt;/ol&gt;




&lt;h2&gt;
  
  
  The Comparison
&lt;/h2&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Feature&lt;/th&gt;
&lt;th&gt;Standard RAG&lt;/th&gt;
&lt;th&gt;LangGraph Memory&lt;/th&gt;
&lt;th&gt;Mem0&lt;/th&gt;
&lt;th&gt;Pensieve&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;Memory types&lt;/td&gt;
&lt;td&gt;1 (chunks)&lt;/td&gt;
&lt;td&gt;2 (checkpointer + Store)&lt;/td&gt;
&lt;td&gt;Centralized (user/session)&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;4 (working, episodic, notes, skills)&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Temporal decay&lt;/td&gt;
&lt;td&gt;None&lt;/td&gt;
&lt;td&gt;Manual / custom&lt;/td&gt;
&lt;td&gt;Implicit (managed platform)&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;Exponential (configurable λ)&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Quality gates&lt;/td&gt;
&lt;td&gt;None&lt;/td&gt;
&lt;td&gt;Manual / custom&lt;/td&gt;
&lt;td&gt;Manual / custom&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;Status-aware + importance scoring&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Failure learning&lt;/td&gt;
&lt;td&gt;None&lt;/td&gt;
&lt;td&gt;Manual / custom&lt;/td&gt;
&lt;td&gt;None&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;Reflexion-style verbal reflections&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Self-pruning&lt;/td&gt;
&lt;td&gt;None&lt;/td&gt;
&lt;td&gt;Manual / custom&lt;/td&gt;
&lt;td&gt;None&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;Agent-managed via tools&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Skill distillation&lt;/td&gt;
&lt;td&gt;None&lt;/td&gt;
&lt;td&gt;None&lt;/td&gt;
&lt;td&gt;None&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;Post-session, novelty-gated&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;PII redaction&lt;/td&gt;
&lt;td&gt;Manual&lt;/td&gt;
&lt;td&gt;Manual&lt;/td&gt;
&lt;td&gt;Manual&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;Automatic before persist&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;




&lt;p&gt;&lt;em&gt;How does your agent handle memory? I’m especially interested in approaches to temporal decay and memory quality. Find me on &lt;a href="https://github.com/sks" rel="noopener noreferrer"&gt;GitHub&lt;/a&gt; or &lt;a href="https://linkedin.com/in/sabithks" rel="noopener noreferrer"&gt;LinkedIn&lt;/a&gt;.&lt;/em&gt;&lt;/p&gt;




&lt;blockquote&gt;
&lt;p&gt;🚀 &lt;strong&gt;We’re building AI-powered SRE at StackGen.&lt;/strong&gt; If you’re tired of 3 AM pages and want AI agents that triage incidents, run diagnostics, and draft RCA reports — check out &lt;a href="https://ai.stackgen.com" rel="noopener noreferrer"&gt;ai.stackgen.com&lt;/a&gt; and try our new SRE offering.&lt;/p&gt;
&lt;/blockquote&gt;

</description>
      <category>aiagents</category>
      <category>memory</category>
      <category>rag</category>
      <category>architecture</category>
    </item>
    <item>
      <title>Why We Chose Go for Our AI Agent Platform (When Everyone Else Picked Python)</title>
      <dc:creator>Sabith KS</dc:creator>
      <pubDate>Sat, 20 Jun 2026 17:00:00 +0000</pubDate>
      <link>https://dev.to/sks/why-we-chose-go-for-our-ai-agent-platform-when-everyone-else-picked-python-4ok2</link>
      <guid>https://dev.to/sks/why-we-chose-go-for-our-ai-agent-platform-when-everyone-else-picked-python-4ok2</guid>
      <description>&lt;p&gt;Every AI framework is in Python. We built ours in Go. Here's why we'd do it again — and when you shouldn't.&lt;/p&gt;

&lt;p&gt;&lt;em&gt;Originally published at &lt;a href="https://sks.github.io/blog/why-go/" rel="noopener noreferrer"&gt;Production Notes&lt;/a&gt;&lt;/em&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fv53in0x6pv79n3isdsuw.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fv53in0x6pv79n3isdsuw.png" alt=" " width="800" height="800"&gt;&lt;/a&gt;&lt;/p&gt;

</description>
      <category>go</category>
      <category>aiagents</category>
      <category>architecture</category>
      <category>llm</category>
    </item>
  </channel>
</rss>
