<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:dc="http://purl.org/dc/elements/1.1/">
  <channel>
    <title>DEV Community: Divy Yadav</title>
    <description>The latest articles on DEV Community by Divy Yadav (@divy_ai).</description>
    <link>https://dev.to/divy_ai</link>
    <image>
      <url>https://media2.dev.to/dynamic/image/width=90,height=90,fit=cover,gravity=auto,format=auto/https:%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F3845890%2Fc910b35f-87b4-4888-b116-ad2cde414360.png</url>
      <title>DEV Community: Divy Yadav</title>
      <link>https://dev.to/divy_ai</link>
    </image>
    <atom:link rel="self" type="application/rss+xml" href="https://dev.to/feed/divy_ai"/>
    <language>en</language>
    <item>
      <title>[Boost]</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Sun, 30 Aug 2026 20:31:02 +0000</pubDate>
      <link>https://dev.to/divy_ai/-3fg9</link>
      <guid>https://dev.to/divy_ai/-3fg9</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/divy_ai/your-sandbox-shouldnt-keep-its-install-time-network-access-4fn8" class="crayons-story__hidden-navigation-link"&gt;Your Sandbox Shouldn't Keep Its Install-Time Network Access&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/divy_ai" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F3845890%2Fc910b35f-87b4-4888-b116-ad2cde414360.png" alt="divy_ai profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/divy_ai" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Divy Yadav
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Divy Yadav
                
                
              
              &lt;div id="story-author-preview-content-4530851" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/divy_ai" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F3845890%2Fc910b35f-87b4-4888-b116-ad2cde414360.png" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Divy Yadav&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/divy_ai/your-sandbox-shouldnt-keep-its-install-time-network-access-4fn8" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Aug 30&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/divy_ai/your-sandbox-shouldnt-keep-its-install-time-network-access-4fn8" id="article-link-4530851"&gt;
          Your Sandbox Shouldn't Keep Its Install-Time Network Access
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/programming"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;programming&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/devops"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;devops&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/security"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;security&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/divy_ai/your-sandbox-shouldnt-keep-its-install-time-network-access-4fn8" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;1&lt;span class="hidden s:inline"&gt;&amp;nbsp;reaction&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/divy_ai/your-sandbox-shouldnt-keep-its-install-time-network-access-4fn8#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            12 min read
          &lt;/small&gt;
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>Your Sandbox Shouldn't Keep Its Install-Time Network Access</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Sun, 30 Aug 2026 20:29:59 +0000</pubDate>
      <link>https://dev.to/divy_ai/your-sandbox-shouldnt-keep-its-install-time-network-access-4fn8</link>
      <guid>https://dev.to/divy_ai/your-sandbox-shouldnt-keep-its-install-time-network-access-4fn8</guid>
      <description>&lt;p&gt;&lt;strong&gt;Give setup the access it needs, then lock the sandbox down before untrusted code runs, without restarting it.&lt;/strong&gt;&lt;/p&gt;




&lt;p&gt;Your AI agent needed internet access to install a dependency. Why should it still have that access when it starts running code you never wrote?&lt;/p&gt;

&lt;p&gt;Say your setup step needs &lt;code&gt;pypi.org&lt;/code&gt;, &lt;code&gt;files.pythonhosted.org&lt;/code&gt;, and a Git host. The code your model writes next needs one internal API and nothing else. Both phases run in the same sandbox under the same firewall rules, because the network policy was decided before either phase existed.&lt;/p&gt;

&lt;p&gt;If the policy stays unchanged after installation, the untrusted phase inherits the installer's network reach and keeps it until the sandbox dies. The broader requirement wins simply because the narrower one would have broken setup.&lt;/p&gt;

&lt;p&gt;That's the problem with treating network policy as part of the environment spec, alongside CPU, memory, and disk. For workloads where trust level changes during execution, permissions should follow the phase, not the sandbox.&lt;/p&gt;




&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fy9d09p9l8ej6osumoor3.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fy9d09p9l8ej6osumoor3.png" alt="Photo from AI" width="800" height="533"&gt;&lt;/a&gt;&lt;/p&gt;




&lt;h2&gt;
  
  
  The permission window is wider than the work
&lt;/h2&gt;

&lt;p&gt;![Photo from AI]&lt;a href="https://dev-to-uploads.s3.us-east-2.amazonaws.com/uploads/articles/i6dgpjktl7llp7fsi6z8.png" rel="noopener noreferrer"&gt;https://dev-to-uploads.s3.us-east-2.amazonaws.com/uploads/articles/i6dgpjktl7llp7fsi6z8.png&lt;/a&gt;)&lt;/p&gt;

&lt;p&gt;Setting network policy at environment creation feels natural, it's how we usually think about containers and VMs: give the environment a set of resources and permissions, then leave them alone until it goes away.&lt;/p&gt;

&lt;p&gt;But agent workloads don't stay in one trust state. During setup, you run dependencies and tools you explicitly chose. A few minutes later, the same sandbox might execute code generated by a model or submitted by a user. Those phases can need very different access, but they sit behind the same firewall.&lt;/p&gt;

&lt;p&gt;Keep one policy for both, and the broader permission set carries into the less-trusted phase, not because the sandbox requires it, but because the policy was tied to the sandbox instead of to the work happening inside it.&lt;/p&gt;

&lt;p&gt;There are a few obvious workarounds, none particularly attractive:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Separate environments.&lt;/strong&gt; Works, but you move state between them and pay the setup cost twice.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Enforce the policy from inside the sandbox.&lt;/strong&gt; Puts the control mechanism in the same environment you're trying to restrict.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Recreate the sandbox between phases.&lt;/strong&gt; Keeps the security boundary, but throws away the state you just built.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;What I actually want is simpler: keep the sandbox, change its egress policy when the phase changes, and see exactly what policy is active at each point. An egress policy defines where a sandbox can make outbound connections, and those rules don't have to stay the same for the sandbox's entire lifetime. That requires enforcement to live outside the workload, with the transition controlled by the system orchestrating the run.&lt;/p&gt;

&lt;p&gt;This is where &lt;a href="https://tensorlake.ai/?utm_source=medium&amp;amp;utm_medium=sponsored_content&amp;amp;utm_campaign=Divy_aug2026" rel="noopener noreferrer"&gt;Tensorlake's Sandboxes&lt;/a&gt; are useful: the egress policy can be replaced on a running sandbox with a single update call. I'm using Tensorlake here because the behavior is explicit enough in the &lt;a href="https://docs.tensorlake.ai/?utm_source=medium&amp;amp;utm_medium=sponsored_content&amp;amp;utm_campaign=Divy_aug2026" rel="noopener noreferrer"&gt;documentation&lt;/a&gt; and SDK source to verify what actually happens, rather than filling gaps with assumptions.&lt;/p&gt;




&lt;h2&gt;
  
  
  What a live policy change has to guarantee to be worth using
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fq3eh75rumjl0cxj3sldv.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fq3eh75rumjl0cxj3sldv.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Before trusting a mechanism like this in a phased design, I want to know what happens when something goes wrong. Four things matter:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Atomicity&lt;/strong&gt; — no moment where old rules have disappeared but new ones aren't active yet. &lt;strong&gt;Failure containment&lt;/strong&gt; — a bad policy leaves the previous one in place. &lt;strong&gt;An external control path&lt;/strong&gt; — the transition is driven by the orchestrator, not the workload. &lt;strong&gt;Clear replacement semantics&lt;/strong&gt; — updates replace rather than merge, so permissions can't quietly accumulate across phases.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://docs.tensorlake.ai/?utm_source=medium&amp;amp;utm_medium=sponsored_content&amp;amp;utm_campaign=Divy_aug2026" rel="noopener noreferrer"&gt;Tensorlake documents&lt;/a&gt; the first three directly: the swap is atomic with no enforcement gap, sending a policy object replaces the entire policy, and a policy naming an unresolvable hostname is rejected while the previous policy stays enforced.&lt;/p&gt;

&lt;p&gt;The fourth is more about how you design the control path than a guarantee Tensorlake gives you. The update runs through an authenticated API, so keeping that path outside the sandbox is up to how you handle credentials, code inside the sandbox could still call the API if it holds a key with enough permissions. The test is mine. The behavior it checks is not.&lt;/p&gt;




&lt;h2&gt;
  
  
  How the policy swap behaves on a running sandbox
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fzidd6lcnjhgjgn3eg40g.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fzidd6lcnjhgjgn3eg40g.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;You can change a sandbox's egress policy without recreating or suspending it. The new policy applies to the running firewall in a single atomic swap, no gap where old rules are gone and new ones haven't taken effect.&lt;/p&gt;

&lt;p&gt;The important detail is where enforcement happens. In the Python SDK, &lt;code&gt;NetworkConfig&lt;/code&gt; is enforced host-side, per sandbox, not inside the guest's own network stack. Changing routes or firewall rules from inside the sandbox doesn't change the policy; doing that would require calling the same authenticated API the orchestrator uses, which makes credential placement a separate security concern.&lt;/p&gt;

&lt;p&gt;The firewall is also stateful: established and related connections remain permitted, so changing the policy doesn't cut off traffic already in progress.&lt;/p&gt;

&lt;p&gt;The live update surface is deliberately limited to name, exposed ports, unauthenticated access for those ports, and egress policy. CPU, memory, and disk are fixed at creation and require a new sandbox to change. Egress policy is one of the few resource properties you can change while running, and a replacement naming an unresolvable hostname is rejected outright, leaving the sandbox on its prior policy in full. A rejected update leaves you where you already were, never somewhere weaker.&lt;/p&gt;

&lt;p&gt;The &lt;code&gt;network&lt;/code&gt; argument is tri-state:&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;You pass&lt;/th&gt;
&lt;th&gt;Result&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;Nothing (omit &lt;code&gt;network&lt;/code&gt;)&lt;/td&gt;
&lt;td&gt;Current policy unchanged. Update &lt;code&gt;name&lt;/code&gt; or ports freely.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;A policy object&lt;/td&gt;
&lt;td&gt;The entire policy is replaced by what you sent.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;An explicit clear&lt;/td&gt;
&lt;td&gt;The sandbox returns to unrestricted egress.&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;Updates replace rather than merge, so every update has to express the complete intended policy for that phase. Clearing removes restrictions rather than imposing them, so a phase with no network at all is &lt;code&gt;allow_internet_access=false&lt;/code&gt; with an empty &lt;code&gt;allow_out&lt;/code&gt;, not a clear.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight python"&gt;&lt;code&gt;&lt;span class="kn"&gt;from&lt;/span&gt; &lt;span class="n"&gt;tensorlake.sandbox&lt;/span&gt; &lt;span class="kn"&gt;import&lt;/span&gt; &lt;span class="n"&gt;CLEAR_NETWORK_POLICY&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;NetworkConfig&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;Sandbox&lt;/span&gt;

&lt;span class="n"&gt;sandbox&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;Sandbox&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="nf"&gt;connect&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;&amp;lt;sandbox-id&amp;gt;&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;

&lt;span class="c1"&gt;# Replace the whole policy: reach api.example.com and nothing else.
&lt;/span&gt;&lt;span class="n"&gt;sandbox&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="nf"&gt;update&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;
    &lt;span class="n"&gt;network&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="nc"&gt;NetworkConfig&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;
        &lt;span class="n"&gt;allow_internet_access&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="bp"&gt;True&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
        &lt;span class="n"&gt;allow_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;api.example.com&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;],&lt;/span&gt;
    &lt;span class="p"&gt;)&lt;/span&gt;
&lt;span class="p"&gt;)&lt;/span&gt;

&lt;span class="c1"&gt;# Later: return the sandbox to unrestricted egress.
&lt;/span&gt;&lt;span class="n"&gt;sandbox&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="nf"&gt;update&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;network&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="n"&gt;CLEAR_NETWORK_POLICY&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;A few SDK details worth knowing if you're writing a wrapper: Python's &lt;code&gt;None&lt;/code&gt; means &lt;em&gt;keep&lt;/em&gt;, not &lt;em&gt;clear&lt;/em&gt; (the explicit &lt;code&gt;CLEAR_NETWORK_POLICY&lt;/code&gt; sentinel is required to clear), while TypeScript expresses the clear as &lt;code&gt;network: null&lt;/code&gt;. Over HTTP it's a &lt;code&gt;PATCH&lt;/code&gt; to &lt;code&gt;/sandboxes/{sandbox_id}&lt;/code&gt;. The field is named &lt;code&gt;network&lt;/code&gt; in the SDK but &lt;code&gt;network_policy&lt;/code&gt; in the raw HTTP response schema, worth pinning to whichever surface you're parsing.&lt;/p&gt;

&lt;p&gt;&lt;code&gt;allow_internet_access&lt;/code&gt; defaults to &lt;code&gt;true&lt;/code&gt;, which makes a policy object that only lists &lt;code&gt;allow_out&lt;/code&gt; a DNS-permitted allowlist rather than a lockdown. I write all three fields on every update, &lt;code&gt;deny_out=[]&lt;/code&gt; included when I mean none, to remove the guesswork.&lt;/p&gt;

&lt;p&gt;The CLI follows the same replace-or-clear rule:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="c"&gt;# Replace the policy: api.example.com plus DNS to the sandbox's resolvers.&lt;/span&gt;
tl sbx update &amp;lt;sandbox-id-or-name&amp;gt; &lt;span class="nt"&gt;-A&lt;/span&gt; api.example.com

&lt;span class="c"&gt;# Replace the policy with an absolute block on outbound traffic.&lt;/span&gt;
tl sbx update &amp;lt;sandbox-id-or-name&amp;gt; &lt;span class="nt"&gt;--no-internet&lt;/span&gt;

&lt;span class="c"&gt;# Clear the policy and restore unrestricted egress.&lt;/span&gt;
tl sbx update &amp;lt;sandbox-id-or-name&amp;gt; &lt;span class="nt"&gt;--clear-network&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;&lt;code&gt;--no-internet&lt;/code&gt; can't combine with a rule flag, and &lt;code&gt;--clear-network&lt;/code&gt; can't combine with any replacement-policy flag, since each describes a whole policy on its own.&lt;/p&gt;

&lt;p&gt;On error handling: a missing sandbox raises &lt;code&gt;SandboxNotFoundError&lt;/code&gt;; anything else the API rejects arrives as &lt;code&gt;RemoteAPIError&lt;/code&gt; with a status code and message; a failure to reach the API at all raises &lt;code&gt;SandboxConnectionError&lt;/code&gt;. That split matters for retries, since a call that never landed and a policy the server refused are different problems. The one state-specific case is &lt;code&gt;409&lt;/code&gt; ("Sandbox is terminated and cannot be updated"), recoverable via &lt;code&gt;restart&lt;/code&gt; for 48 hours, which restores from the most recent snapshot if one exists or cold-boots otherwise, re-entering &lt;code&gt;Pending&lt;/code&gt; either way, so re-establish the phase policy explicitly rather than assume it survived. The exact status code for an unresolvable-hostname rejection isn't clearly pinned in the schema, so branch on the message too, not just the status code, before treating a policy as permanently bad.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://docs.tensorlake.ai/?utm_source=medium&amp;amp;utm_medium=sponsored_content&amp;amp;utm_campaign=Divy_aug2026" rel="noopener noreferrer"&gt;Tensorlake's documentation&lt;/a&gt; reports two public-cloud checks of the enforcement itself: DNS resolution failing under &lt;code&gt;allow_internet_access=False&lt;/code&gt;, and a &lt;code&gt;deny_out&lt;/code&gt; entry blocking one destination while another stayed reachable. Those are existence proofs for specific destinations, not throughput or general enforcement measurements.&lt;/p&gt;




&lt;h2&gt;
  
  
  What &lt;code&gt;allow_internet_access&lt;/code&gt; actually controls once &lt;code&gt;allow_out&lt;/code&gt; is non-empty
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F261yeldaykhong3r5nq4.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F261yeldaykhong3r5nq4.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Whether this flag opens general egress or only DNS depends on &lt;code&gt;allow_out&lt;/code&gt;. A non-empty allowlist is itself a default-deny rule, which changes what the boolean is doing:&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;&lt;code&gt;allow_internet_access&lt;/code&gt;&lt;/th&gt;
&lt;th&gt;&lt;code&gt;allow_out&lt;/code&gt;&lt;/th&gt;
&lt;th&gt;Outbound behavior&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;true&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;empty&lt;/td&gt;
&lt;td&gt;Everything reachable, minus whatever &lt;code&gt;deny_out&lt;/code&gt; matches.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;true&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;non-empty&lt;/td&gt;
&lt;td&gt;Default-deny: nothing but the listed destinations, DNS to the sandbox's resolvers permitted.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;false&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;non-empty&lt;/td&gt;
&lt;td&gt;Default-deny again, and DNS fails unless the resolver's own IP appears in &lt;code&gt;allow_out&lt;/code&gt;.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;false&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;empty&lt;/td&gt;
&lt;td&gt;Nothing leaves the sandbox, DNS included.&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;Reading &lt;code&gt;allow_internet_access=True, allow_out=[...]&lt;/code&gt; as "internet on, with an allowlist on top" gets the model backwards: you have an allowlist, with DNS permitted. The &lt;code&gt;false&lt;/code&gt; plus non-empty row is where an afternoon disappears, since hostnames can't resolve unless the resolver's IP is also listed, or unless you list only IPv4 addresses and CIDR ranges.&lt;/p&gt;

&lt;p&gt;&lt;code&gt;allow_out&lt;/code&gt; accepts domains, leading-wildcard domains like &lt;code&gt;*.example.com&lt;/code&gt;, IPv4 addresses, and CIDR ranges; &lt;code&gt;deny_out&lt;/code&gt; accepts the same except wildcards. &lt;code&gt;deny_out&lt;/code&gt; takes precedence on any overlap, including against the sandbox's own resolver, which is the failure mode where a correct-looking allowlist suddenly stops resolving anything. A &lt;code&gt;:port&lt;/code&gt; suffix is accepted but doesn't make a rule port-specific. A wildcard covers subdomains but not the apex (&lt;code&gt;*.example.com&lt;/code&gt; reaches &lt;code&gt;api.example.com&lt;/code&gt; but not &lt;code&gt;example.com&lt;/code&gt; itself), and hostname rules follow DNS changes, so an allowlisted CDN-backed domain keeps working as its IPs rotate.&lt;/p&gt;




&lt;h2&gt;
  
  
  A policy state machine for a phased agent run
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Feb9nu9wlkukk9zcncjaj.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Feb9nu9wlkukk9zcncjaj.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Treat the run as a state machine whose states are policies: &lt;code&gt;install&lt;/code&gt; → &lt;code&gt;fetch&lt;/code&gt; → &lt;code&gt;execute&lt;/code&gt; → &lt;code&gt;deliver&lt;/code&gt; → &lt;code&gt;sealed&lt;/code&gt;, one &lt;code&gt;update&lt;/code&gt; call per arrow, a complete policy statement in every state. The phase model is mine, not a documented &lt;a href="https://tensorlake.ai/?utm_source=medium&amp;amp;utm_medium=sponsored_content&amp;amp;utm_campaign=Divy_aug2026" rel="noopener noreferrer"&gt;Tensorlake&lt;/a&gt; pattern, though every mechanic it relies on is &lt;a href="https://docs.tensorlake.ai/?utm_source=medium&amp;amp;utm_medium=sponsored_content&amp;amp;utm_campaign=Divy_aug2026" rel="noopener noreferrer"&gt;documented&lt;/a&gt;.&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Phase&lt;/th&gt;
&lt;th&gt;&lt;code&gt;allow_internet_access&lt;/code&gt;&lt;/th&gt;
&lt;th&gt;&lt;code&gt;allow_out&lt;/code&gt;&lt;/th&gt;
&lt;th&gt;Purpose&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;install&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;&lt;code&gt;true&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;Registries and source hosts&lt;/td&gt;
&lt;td&gt;Dependency installation under default-deny&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;fetch&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;&lt;code&gt;true&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;Input data hosts only&lt;/td&gt;
&lt;td&gt;Retrieve inputs, install-time reach already retired&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;execute&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;
&lt;code&gt;true&lt;/code&gt; (allowlist) or &lt;code&gt;false&lt;/code&gt; if no network needed&lt;/td&gt;
&lt;td&gt;One endpoint, or empty&lt;/td&gt;
&lt;td&gt;Smallest surface for least-trusted code&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;deliver&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;&lt;code&gt;true&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;Upload/result endpoint only&lt;/td&gt;
&lt;td&gt;Output path separate from execute&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;sealed&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;&lt;code&gt;false&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;empty&lt;/td&gt;
&lt;td&gt;No egress at all, DNS included&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;The swap is atomic, so no window exists where egress goes unfiltered, and a failed transition (unresolvable hostname) leaves the previous state enforced rather than stranding the sandbox somewhere wider. What the swap doesn't do is close what's already open: established connections survive it, so a connection opened during &lt;code&gt;install&lt;/code&gt; is still usable in &lt;code&gt;execute&lt;/code&gt;. Narrowing &lt;code&gt;allow_out&lt;/code&gt; shrinks what's reachable going forward; it doesn't retroactively make survivors safe.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight python"&gt;&lt;code&gt;&lt;span class="c1"&gt;# Setup finished. Retire the install-time surface before running generated code.
&lt;/span&gt;&lt;span class="n"&gt;sandbox&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="nf"&gt;update&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;
    &lt;span class="n"&gt;network&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="nc"&gt;NetworkConfig&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;
        &lt;span class="n"&gt;allow_internet_access&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="bp"&gt;True&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
        &lt;span class="n"&gt;allow_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;api.internal.example.com&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;],&lt;/span&gt;
    &lt;span class="p"&gt;)&lt;/span&gt;
&lt;span class="p"&gt;)&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Past two phases, I'd stop writing update calls by hand and keep the policy next to the phase definition instead, so the two can't drift apart:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight python"&gt;&lt;code&gt;&lt;span class="kn"&gt;from&lt;/span&gt; &lt;span class="n"&gt;tensorlake.sandbox&lt;/span&gt; &lt;span class="kn"&gt;import&lt;/span&gt; &lt;span class="n"&gt;NetworkConfig&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;Sandbox&lt;/span&gt;

&lt;span class="n"&gt;PHASE_ORDER&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;install&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;fetch&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;execute&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;deliver&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;sealed&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;

&lt;span class="n"&gt;PHASE_POLICY&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="p"&gt;{&lt;/span&gt;
    &lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;install&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt; &lt;span class="nc"&gt;NetworkConfig&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;
        &lt;span class="n"&gt;allow_internet_access&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="bp"&gt;True&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
        &lt;span class="n"&gt;allow_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;pypi.org&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;files.pythonhosted.org&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;github.com&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;],&lt;/span&gt;
        &lt;span class="n"&gt;deny_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[],&lt;/span&gt;
    &lt;span class="p"&gt;),&lt;/span&gt;
    &lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;fetch&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt; &lt;span class="nc"&gt;NetworkConfig&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;
        &lt;span class="n"&gt;allow_internet_access&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="bp"&gt;True&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
        &lt;span class="n"&gt;allow_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;inputs.internal.example.com&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;],&lt;/span&gt;
        &lt;span class="n"&gt;deny_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[],&lt;/span&gt;
    &lt;span class="p"&gt;),&lt;/span&gt;
    &lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;execute&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt; &lt;span class="nc"&gt;NetworkConfig&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;
        &lt;span class="n"&gt;allow_internet_access&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="bp"&gt;True&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
        &lt;span class="n"&gt;allow_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;api.internal.example.com&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;],&lt;/span&gt;
        &lt;span class="n"&gt;deny_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[],&lt;/span&gt;
    &lt;span class="p"&gt;),&lt;/span&gt;
    &lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;deliver&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt; &lt;span class="nc"&gt;NetworkConfig&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;
        &lt;span class="n"&gt;allow_internet_access&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="bp"&gt;True&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
        &lt;span class="n"&gt;allow_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;results.internal.example.com&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;],&lt;/span&gt;
        &lt;span class="n"&gt;deny_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[],&lt;/span&gt;
    &lt;span class="p"&gt;),&lt;/span&gt;
    &lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;sealed&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt; &lt;span class="nc"&gt;NetworkConfig&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;
        &lt;span class="n"&gt;allow_internet_access&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="bp"&gt;False&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
        &lt;span class="n"&gt;allow_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[],&lt;/span&gt;
        &lt;span class="n"&gt;deny_out&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="p"&gt;[],&lt;/span&gt;
    &lt;span class="p"&gt;),&lt;/span&gt;
&lt;span class="p"&gt;}&lt;/span&gt;


&lt;span class="k"&gt;def&lt;/span&gt; &lt;span class="nf"&gt;enter_phase&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;sandbox&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt; &lt;span class="n"&gt;Sandbox&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;phase&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt; &lt;span class="nb"&gt;str&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;current&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt; &lt;span class="nb"&gt;str&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="bp"&gt;None&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="bp"&gt;None&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="o"&gt;-&amp;gt;&lt;/span&gt; &lt;span class="nb"&gt;str&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;
    &lt;span class="k"&gt;if&lt;/span&gt; &lt;span class="n"&gt;current&lt;/span&gt; &lt;span class="ow"&gt;is&lt;/span&gt; &lt;span class="ow"&gt;not&lt;/span&gt; &lt;span class="bp"&gt;None&lt;/span&gt; &lt;span class="ow"&gt;and&lt;/span&gt; &lt;span class="n"&gt;PHASE_ORDER&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="nf"&gt;index&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;phase&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="o"&gt;&amp;lt;=&lt;/span&gt; &lt;span class="n"&gt;PHASE_ORDER&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="nf"&gt;index&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;current&lt;/span&gt;&lt;span class="p"&gt;):&lt;/span&gt;
        &lt;span class="k"&gt;raise&lt;/span&gt; &lt;span class="nc"&gt;RuntimeError&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="sa"&gt;f&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;refusing to move from &lt;/span&gt;&lt;span class="si"&gt;{&lt;/span&gt;&lt;span class="n"&gt;current&lt;/span&gt;&lt;span class="si"&gt;}&lt;/span&gt;&lt;span class="s"&gt; back to &lt;/span&gt;&lt;span class="si"&gt;{&lt;/span&gt;&lt;span class="n"&gt;phase&lt;/span&gt;&lt;span class="si"&gt;}&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
    &lt;span class="n"&gt;intended&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;PHASE_POLICY&lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="n"&gt;phase&lt;/span&gt;&lt;span class="p"&gt;]&lt;/span&gt;
    &lt;span class="n"&gt;info&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;sandbox&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="nf"&gt;update&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;network&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="n"&gt;intended&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
    &lt;span class="n"&gt;effective&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;info&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;network&lt;/span&gt;
    &lt;span class="nf"&gt;if &lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;
        &lt;span class="n"&gt;effective&lt;/span&gt; &lt;span class="ow"&gt;is&lt;/span&gt; &lt;span class="bp"&gt;None&lt;/span&gt;
        &lt;span class="ow"&gt;or&lt;/span&gt; &lt;span class="n"&gt;effective&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;allow_internet_access&lt;/span&gt; &lt;span class="o"&gt;!=&lt;/span&gt; &lt;span class="n"&gt;intended&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;allow_internet_access&lt;/span&gt;
        &lt;span class="ow"&gt;or&lt;/span&gt; &lt;span class="nf"&gt;set&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;effective&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;allow_out&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="o"&gt;!=&lt;/span&gt; &lt;span class="nf"&gt;set&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;intended&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;allow_out&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
        &lt;span class="ow"&gt;or&lt;/span&gt; &lt;span class="nf"&gt;set&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;effective&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;deny_out&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="o"&gt;!=&lt;/span&gt; &lt;span class="nf"&gt;set&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;intended&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;deny_out&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
    &lt;span class="p"&gt;):&lt;/span&gt;
        &lt;span class="k"&gt;raise&lt;/span&gt; &lt;span class="nc"&gt;RuntimeError&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="sa"&gt;f&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="si"&gt;{&lt;/span&gt;&lt;span class="n"&gt;phase&lt;/span&gt;&lt;span class="si"&gt;}&lt;/span&gt;&lt;span class="s"&gt;: policy did not take effect, got &lt;/span&gt;&lt;span class="si"&gt;{&lt;/span&gt;&lt;span class="n"&gt;effective&lt;/span&gt;&lt;span class="si"&gt;}&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
    &lt;span class="k"&gt;return&lt;/span&gt; &lt;span class="n"&gt;phase&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Every entry spells out all three fields, since leaning on the &lt;code&gt;allow_internet_access&lt;/code&gt; default is how an intended lockdown becomes an allowlist. The check compares all three too, so a policy whose boolean or deny list drifted from the phase definition still fails loudly rather than passing on an &lt;code&gt;allow_out&lt;/code&gt; match alone. I want that assertion because if the phase table and the sandbox ever disagree, the run should stop rather than continue under a policy nobody chose. The ordering guard enforces the declared sequence, it doesn't prove each policy is a strict subset of the one before it, but it's the cheap version of the property I actually want.&lt;/p&gt;

&lt;p&gt;The &lt;code&gt;install&lt;/code&gt; entry is where the real work hides. Enumerating what your build actually touches is harder than it looks, a dependency resolver can reach a CDN you never named, so the honest approach is to start strict and widen from the failures rather than guess wide and never revisit it.&lt;/p&gt;




&lt;h2&gt;
  
  
  Each transition returns a record you can keep
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fhwap84a8x9etgr4k8g4m.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fhwap84a8x9etgr4k8g4m.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Every phase change goes through an authenticated control-plane call, and on success, the response gives you the sandbox record as it stands after the update, including the current policy. I'd keep those responses: they let you answer "what was this sandbox allowed to reach at 14:32?" later, without instrumenting the workload itself.&lt;/p&gt;

&lt;p&gt;The security benefit is straightforward, but the operational benefit is what I find more useful. After an incident, knowing a policy existed isn't enough, you need to know which policy was actually in place at that point. A create-time configuration can't give you that history for a sandbox that's been running for six hours. It also gives you a boundary you can test in CI: run against a real sandbox and verify the setup-time hosts are unreachable before the untrusted phase starts.&lt;/p&gt;




&lt;h2&gt;
  
  
  What the swap does not do for you
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F62j4eo5n4ivaj0uweh5t.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F62j4eo5n4ivaj0uweh5t.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Connection survival&lt;/strong&gt; is the detail I'd pay the most attention to. Tightening the policy isn't the same as cutting off everything already connected. If your threat model requires setup-time connections to disappear before the next phase, close them yourself.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Egress and ingress are separate.&lt;/strong&gt; Changing one doesn't change the other. Inbound access runs through &lt;code&gt;exposed_ports&lt;/code&gt; and &lt;code&gt;allow_unauthenticated_access&lt;/code&gt;, while the management port &lt;code&gt;9501&lt;/code&gt; always requires authentication. They clear differently too: an empty &lt;code&gt;exposed_ports&lt;/code&gt; array still leaves the management port available, while clearing the egress policy requires an explicit &lt;code&gt;null&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The API key is part of this boundary.&lt;/strong&gt; Updating the policy is an authenticated operation whether through SDK, CLI, or &lt;code&gt;PATCH&lt;/code&gt;. &lt;a href="https://docs.tensorlake.ai/?utm_source=medium&amp;amp;utm_medium=sponsored_content&amp;amp;utm_campaign=Divy_aug2026" rel="noopener noreferrer"&gt;Tensorlake API keys&lt;/a&gt; are project-scoped with project-member permissions, so a key that can update the sandbox can widen its network policy, which is why I'd keep the credential in the orchestrator rather than inside the environment it controls.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Idle timeout is a separate failure mode.&lt;/strong&gt; &lt;code&gt;timeout_secs&lt;/code&gt; is an idle threshold, not a wall-clock lifetime, measured against traffic through the sandbox proxy and defaulting to 600 seconds. A named sandbox suspends and can resume under the same ID; an ephemeral one terminates. Naming a sandbox is worth deciding before a phased run, since it makes suspend/resume available, and the same update call can set &lt;code&gt;name&lt;/code&gt; alongside the policy. I wouldn't assume a policy update resets the idle clock: the update request has no &lt;code&gt;timeout_secs&lt;/code&gt; field, and whether a control-plane &lt;code&gt;PATCH&lt;/code&gt; counts as the "traffic through the proxy" the timeout is defined against isn't obvious, so I'd measure that directly rather than assume it. I'd also resume a suspended sandbox explicitly before updating it rather than assume the update handles that state transition; &lt;code&gt;Sandbox.connect()&lt;/code&gt; doesn't auto-resume, though a request through an exposed port can auto-resume a suspended named sandbox.&lt;/p&gt;




&lt;h2&gt;
  
  
  When this is the wrong tool
&lt;/h2&gt;

&lt;p&gt;If the sandbox stays at one trust level start to finish, or every phase genuinely needs the same broad access, set the policy once and leave it. Switching policies at that point just adds something to manage without reducing the sandbox's actual reach.&lt;/p&gt;

&lt;p&gt;If untrusted code needs no network at all, skip the transitions entirely: create the sandbox with &lt;code&gt;allow_internet_access=False&lt;/code&gt; and an empty &lt;code&gt;allow_out&lt;/code&gt; from the start.&lt;/p&gt;

&lt;p&gt;And egress policy doesn't solve a filesystem isolation problem. If untrusted code must never share an environment with credentials or artifacts left over from setup, tightening network access isn't enough, the policy controls where the sandbox connects, not what's already sitting inside it. Use separate environments for that instead.&lt;/p&gt;




&lt;h2&gt;
  
  
  The takeaway
&lt;/h2&gt;

&lt;p&gt;Permission lifetime is a design choice. Tie it to the sandbox's lifetime, and you carry the combined network access every phase needs for as long as the sandbox exists.&lt;/p&gt;

&lt;p&gt;Once you can change the policy while running, the egress policy can match the phase instead. List what setup actually needs rather than opening access broadly, replace that policy before untrusted code runs, and check the returned policy at each transition.&lt;/p&gt;

&lt;p&gt;The version I'd actually use is simple: keep policies in a phase dictionary, one update per boundary, refuse to move back to a broader phase, and fail loudly if the policy that comes back isn't the one you expected.&lt;/p&gt;

&lt;p&gt;It isn't much code, but it changes the security model. The sandbox no longer carries the installer's network reach for its entire lifetime. It only has it while the installer needs it.&lt;/p&gt;




&lt;h2&gt;
  
  
  References
&lt;/h2&gt;

&lt;ul&gt;
&lt;li&gt;&lt;a href="https://tensorlake.ai/?utm_source=medium&amp;amp;utm_medium=sponsored_content&amp;amp;utm_campaign=Divy_aug2026" rel="noopener noreferrer"&gt;Tensorlake Website&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://docs.tensorlake.ai/?utm_source=medium&amp;amp;utm_medium=sponsored_content&amp;amp;utm_campaign=Divy_aug2026" rel="noopener noreferrer"&gt;Tensorlake Documentation&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://docs.tensorlake.ai/?utm_source=medium&amp;amp;utm_medium=sponsored_content&amp;amp;utm_campaign=Divy_aug2026" rel="noopener noreferrer"&gt;Tensorlake Sandboxes &amp;amp; Networking Guides&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;

</description>
      <category>ai</category>
      <category>programming</category>
      <category>devops</category>
      <category>security</category>
    </item>
    <item>
      <title>Where Sandbox Ingress Speed Actually Comes From</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Sat, 22 Aug 2026 20:17:34 +0000</pubDate>
      <link>https://dev.to/divy_ai/where-sandbox-ingress-speed-actually-comes-from-24gm</link>
      <guid>https://dev.to/divy_ai/where-sandbox-ingress-speed-actually-comes-from-24gm</guid>
      <description>&lt;p&gt;A proxy chain into an isolated sandbox usually has more than one hop, and each one deserves its own answer to the same question: &lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;&lt;strong&gt;Does this specific layer need to understand the application protocol, or is it mainly moving bytes between two points that already trust each other?&lt;/strong&gt;&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;Many production proxy chains combine L7 and L4 functions at different hops. The interesting engineering work is in separating the cost of parsing and buffering at L7 from the cost of moving bytes at L4.&lt;/p&gt;

&lt;p&gt;Get that separation wrong, and a technique like kernel Transport Layer Security (kTLS) can end up credited with a win that actually came from removing a parser.&lt;/p&gt;

&lt;p&gt;A recent ingress rebuild provides a concrete test of that separation. The team moved one dataplane hop from a full L7 reverse proxy to an L4 forwarder using kTLS and &lt;code&gt;splice(2)&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;They measured each change on its own rather than only comparing the old system to the new one. Most of the CPU saving in that staged comparison came from removing the parsing layer; kTLS and &lt;code&gt;splice(2)&lt;/code&gt; added a smaller, separate throughput gain on top. &lt;/p&gt;

&lt;p&gt;Tensorlake’s rebuild is the worked example; the underlying decision applies to proxy chains more broadly.&lt;/p&gt;

&lt;p&gt;The questions underneath it, when an L7 hop earns its cost, when L4 becomes attractive, what disappears when a hop stops parsing, what has to be rebuilt at L4, and how to test whether removing L7 actually matters, apply to any proxy chain, not just this one.&lt;/p&gt;




&lt;h2&gt;
  
  
  The real question: which hop needs to understand the protocol
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fsx7p1dcra9hwl7gwyci1.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fsx7p1dcra9hwl7gwyci1.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;The path into a Tensorlake sandbox runs through an edge gateway and a separate dataplane-side hop. The edge gateway, built on Cloudflare's Pingora, terminates the client's Transport Layer Security (TLS) connection, accepts HTTP/S, WebSocket, and gRPC traffic, and extracts sandbox-routing information from the request for those protocols. SSH uses a different extraction path from the connection.&lt;/p&gt;

&lt;p&gt;That hop is still L7 today, and stayed that way through the entire redesign. What changed sits behind it: a dataplane-side hop that used to run a full L7 reverse proxy speaking HTTP/2 over mutual TLS (mTLS), and now runs an L4 forwarder. &lt;/p&gt;

&lt;p&gt;That's the shape of the actual decision, and it's made per hop, not once for the whole path: which layer needs to read the application protocol, and which layers only need to move authenticated bytes?&lt;/p&gt;

&lt;p&gt;Keep L7 where a hop needs request-aware behavior, such as protocol-aware routing, application-aware retry policy, request deadlines, or synthesized responses. L4 becomes a candidate when an earlier hop has already made those decisions and the next hop only needs to move bytes over an authenticated channel.&lt;/p&gt;

&lt;p&gt;The team also considered lower-level alternatives for that dataplane hop, including VXLAN, eBPF-steered routing, a CNI plugin, and host-level encapsulation. The engineering post says the team chose a non-invasive, application-layer mechanism instead, because the platform runs across AWS, GCP, and planned GPU neoclouds.&lt;/p&gt;

&lt;p&gt;The data path shouldn't depend on any one provider's networking primitives. L4 forwarding won out over L7 for this one hop; the choice against those lower-level options was a separate portability decision. &lt;/p&gt;

&lt;p&gt;The change comes down to one hop:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;OLD - two L7 hops between the client and the app

+--------+      +--------------+      +----------------+      +-------------+
| Client | --&amp;gt;  | Edge gateway | --&amp;gt;  |    L7 proxy    | --&amp;gt;  | Sandbox app |
|        |      |  TLS + auth  |      |  mTLS+HTTP/2   |      |             |
+--------+      +--------------+      +----------------+      +-------------+


NEW - one L4 hop between the client and the app

+--------+      +--------------+      +----------------+      +-------------+
| Client | --&amp;gt;  | Edge gateway | --&amp;gt;  |  L4 forwarder  | --&amp;gt;  | Sandbox app |
|        |      |  TLS + auth  |      | kTLS+splice(2) |      |  plaintext  |
+--------+      +--------------+      +----------------+      +-------------+

Only the third box changes. The edge gateway stays L7 in both paths.
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The public-facing side of that edge layer is documented separately, in the networking docs: the proxy preserves the request path and query string, supports WebSocket upgrades, and forwards gRPC over HTTP/2. Those are request- and protocol-aware behaviors, the kind of work that generally belongs at an L7 boundary. &lt;/p&gt;




&lt;h2&gt;
  
  
  When L7 is worth keeping
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Flmqtaqu7ktz2xxpzwros.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Flmqtaqu7ktz2xxpzwros.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;An L7 hop earns its cost by reading the request. Because it can see what's actually in flight, it can support:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;routing on a path, header, or other application-level field&lt;/li&gt;
&lt;li&gt;an application-aware retry policy&lt;/li&gt;
&lt;li&gt;a per-request timeout or deadline&lt;/li&gt;
&lt;li&gt;transforming a request or response in flight&lt;/li&gt;
&lt;li&gt;a synthesized, application-level error when something downstream misbehaves&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;The edge gateway here is a clean example of a hop where that trade still holds. For HTTP/S, WebSocket, and gRPC, it extracts routing information from the request itself and authenticates the user; SSH uses a different extraction path from the connection. &lt;/p&gt;

&lt;p&gt;An L4 hop cannot perform those application-protocol-aware functions without parsing the traffic, although it can authenticate its own transport channel using mechanisms such as mTLS.&lt;/p&gt;

&lt;p&gt;Nothing about the dataplane redesign touched that layer, because the argument for dropping L7 never applied to it. &lt;/p&gt;




&lt;h2&gt;
  
  
  When L4 starts looking attractive
&lt;/h2&gt;

&lt;p&gt;A hop becomes a real candidate for L4 once its cost is dominated by parsing and buffering bytes it never needs to act on, and that cost scales with data volume rather than request count. That's a different profile from a hop handling many small, distinct calls, where the request-level work is the entire reason the hop exists.&lt;/p&gt;

&lt;p&gt;The dataplane hop here fit that profile closely. Traffic included thousands of small calls to start, stop, and observe sandboxes, alongside file uploads and downloads that moved substantially more data.&lt;/p&gt;

&lt;p&gt;The engineering post says the old dataplane proxy parsed and buffered that bulk traffic even though the hop never needed to interpret it, since routing had already happened at the edge. &lt;/p&gt;

&lt;p&gt;There was also a separate, structural problem layered on top. That proxy shared a binary with the sandbox orchestrator, so a routine orchestrator deploy could interrupt live connections that had nothing to do with the update.&lt;/p&gt;

&lt;p&gt;That's a separate issue from the performance question: a team facing only the deploy-interruption problem could address it by decoupling the proxy's lifecycle from the orchestrator's, without changing the network layer at all. &lt;/p&gt;

&lt;p&gt;&lt;a href="/Users/divyyadav/article/second_article/images/02_when_l7_vs_l4_matters.png" class="article-body-image-wrapper"&gt;&lt;img src="/Users/divyyadav/article/second_article/images/02_when_l7_vs_l4_matters.png" alt="When L7 is Worth Keeping vs When L4 Looks Attractive"&gt;&lt;/a&gt;&lt;/p&gt;




&lt;h2&gt;
  
  
  What you give up when a hop drops to L4
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fc2onawq8zoqgqj322b3n.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fc2onawq8zoqgqj322b3n.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Nothing about moving a hop to L4 is free. An L4 proxy that does not parse the application protocol can forward bytes but cannot make request-semantic decisions, so the removed L7 hop no longer provides:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;a different routing decision for each request on the same connection&lt;/li&gt;
&lt;li&gt;intelligent, application-aware retries&lt;/li&gt;
&lt;li&gt;a real, synthesized error response when something downstream fails&lt;/li&gt;
&lt;li&gt;watching individual requests to know a connection is still active&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Any capability that depended on reading traffic has to move somewhere else or get rebuilt on a signal the L4 layer can actually observe. This forwarder takes on two of those: routing and liveness.&lt;/p&gt;

&lt;p&gt;Routing information, which used to be obtained from each request, now arrives before tenant data in a short, length-bounded preamble naming the sandbox ID and the sandbox’s private target address (&lt;code&gt;ip:port&lt;/code&gt;). The gateway learned that target from the scheduler.&lt;/p&gt;

&lt;p&gt;That happens over a connection that's mutually authenticated in both directions: the forwarder presents a certificate that the gateway pins, and the gateway presents a client certificate whose identity the forwarder checks against an allowlist. &lt;/p&gt;

&lt;p&gt;Liveness, which used to come from watching requests, is rebuilt in this implementation by metering bytes moving through each connection and reporting those byte counts to the layer that owns the sandbox’s idle timer.&lt;/p&gt;

&lt;p&gt;In Tensorlake's implementation, that idle timer runs on a threshold rather than a fixed lifetime: the forwarder meters bytes and reports them to the dataplane, which resets the sandbox's idle timer as long as traffic keeps flowing, and only lets it time out once nothing has moved for the configured period.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Both routing and liveness took real engineering work. An L4 hop doesn't produce either one on its own.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="/Users/divyyadav/article/second_article/images/03_rebuilding_l4_features.png" class="article-body-image-wrapper"&gt;&lt;img src="/Users/divyyadav/article/second_article/images/03_rebuilding_l4_features.png" alt="What You Give Up at L4 and How It Was Rebuilt"&gt;&lt;/a&gt;&lt;/p&gt;




&lt;h2&gt;
  
  
  Why kernel TLS enables the zero-copy path, but isn't where most of the performance gain comes from
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fdmc0l14syl4xipva8ndz.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fdmc0l14syl4xipva8ndz.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The forwarder still terminates mutual TLS on the gateway connection, which is what makes the routing preamble trustworthy in the first place.&lt;/strong&gt; A plain L4 forwarder that does this the ordinary way, decrypting into a userspace buffer and writing the plaintext back out, &lt;strong&gt;already gets most of the benefit of dropping the L7 hop.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;This setup measured 2.07 GB/s and 0.50 CPU-seconds per GB at exactly that stage, before kTLS entered the picture at all. Removing the request-parsing layer is what did that.&lt;/p&gt;

&lt;p&gt;The team's own expectation going in was that kTLS would be responsible for most of the savings, that eliminating two userspace copies was where the CPU went. The isolated measurement said otherwise.&lt;/p&gt;

&lt;p&gt;What kTLS adds on top is narrower than it first sounds, and it's worth being precise about what kind of narrower. In a conventional userspace TLS termination path, encrypted bytes cross into userspace, are decrypted there, and the resulting plaintext is then written onward, creating the userspace staging and boundary-crossing work that this design aims to avoid.&lt;/p&gt;

&lt;p&gt;Kernel TLS moves symmetric TLS record processing into the kernel after the userspace handshake and key installation, so application data is decrypted on receive and encrypted on transmit by the kernel. In this socket-to-pipe-to-socket implementation, that enables &lt;code&gt;splice(2)&lt;/code&gt; to avoid the ordinary userspace staging path for tenant payloads; it should not be read as a guarantee that every possible internal copy is eliminated in every kTLS configuration.&lt;/p&gt;

&lt;p&gt;The plain userspace-copy stage above still had to copy every decrypted byte through the application; it just did that without also parsing a request first. That alone is what produced most of the measured gain in this staged test, before kTLS entered the picture.&lt;/p&gt;

&lt;p&gt;What kTLS specifically supplies is a way for a hop to keep terminating TLS, which the routing preamble depends on, while still avoiding the ordinary userspace staging path in this implementation, instead of requiring a userspace copy for every decrypted byte because the hop needs to inspect the TLS record layer.&lt;/p&gt;

&lt;p&gt;Tensorlake reports requiring Linux with &lt;code&gt;CONFIG_TLS&lt;/code&gt; and kTLS receive support and using Linux 6.x in production; it reports Linux 5.1 or newer for its TLS 1.3 path. Actual availability depends on the kernel build and configuration, distribution, TLS-library integration, cipher support, and the socket path.&lt;/p&gt;

&lt;p&gt;This implementation uses software kTLS, with cryptography on the CPU rather than required NIC offload.&lt;/p&gt;

&lt;p&gt;It's also fail-closed by design: the forwarder daemon holds no sandbox state of its own and refuses to start at all if the kernel can't attach the TLS ULP, rather than silently falling back to a slower, userspace-copying path. &lt;/p&gt;

&lt;p&gt;On the CPU that's left, the encryption itself is a minority cost.&lt;/p&gt;

&lt;p&gt;AES-256-GCM ran at roughly 7.9 GB/s per core. At 0.49 CPU-s/GB, that implies approximately 0.127 CPU-s/GB for the cryptographic work, or about 26% of the measured forwarder CPU. The remainder was attributed to TCP handling and syscalls in that measurement; this should not be treated as a portable AES performance constant.&lt;/p&gt;

&lt;p&gt;&lt;a href="/Users/divyyadav/article/second_article/images/04_ktls_splice_zero_copy_and_cpu.png" class="article-body-image-wrapper"&gt;&lt;img src="/Users/divyyadav/article/second_article/images/04_ktls_splice_zero_copy_and_cpu.png" alt="How Kernel TLS and splice(2) Achieve Zero-Copy"&gt;&lt;/a&gt;&lt;/p&gt;




&lt;h2&gt;
  
  
  How to know whether removing L7 actually matters for you
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F6r6kbb7honz8ms02nhih.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F6r6kbb7honz8ms02nhih.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Comparing only the old system to the new one won't tell you what caused the gain.&lt;/p&gt;

&lt;p&gt;The useful version of this test has three stages: the existing L7 path, an L4 path that still does a plain userspace copy, and the L4 path with kTLS and &lt;code&gt;splice(2)&lt;/code&gt; added on top. The middle stage helps isolate the effect of removing the parser and buffering layer from the incremental effect of the kTLS-plus-&lt;code&gt;splice(2)&lt;/code&gt; path, provided the workload and measurement conditions remain controlled.&lt;/p&gt;

&lt;p&gt;The staged measurements:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;STAGED MEASUREMENT: isolate each change before crediting it

+--------------------+      +--------------------+      +--------------------+
|    L7, two hops    |      | L4, userspace copy |      | L4, kTLS+splice(2) |
|     1.12 GB/s      | --&amp;gt;  |     2.07 GB/s      | --&amp;gt;  |     2.50 GB/s      |
|   0.90 CPU-s/GB    |      |   0.50 CPU-s/GB    |      |   0.49 CPU-s/GB    |
+--------------------+      +--------------------+      +--------------------+

removing L7: +0.95 GB/s, -0.40 CPU-s/GB
adding kTLS + splice(2): +0.43 GB/s, -0.01 CPU-s/GB
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;&lt;a href="/Users/divyyadav/article/second_article/images/05_staged_benchmark_isolation.png" class="article-body-image-wrapper"&gt;&lt;img src="/Users/divyyadav/article/second_article/images/05_staged_benchmark_isolation.png" alt="Staged Measurement: Isolating the Real Win"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Measuring those as separate steps is what makes it possible to say that removing the L7 hop accounted for most of the throughput gain and nearly all of the CPU saving, with kTLS contributing the smaller remainder.&lt;/p&gt;

&lt;p&gt;That conclusion describes this one test. It isn't a claim that every L7-to-L4 migration will split the same way. &lt;/p&gt;

&lt;p&gt;The test also didn't stop at a single connection, and it was direct about the limits of what it measured. Under concurrency, eight tunnels on that same host aggregated to 8.67 GB/s, at which point the bottleneck was serialization in the splice loop rather than CPU.&lt;/p&gt;

&lt;p&gt;The headline numbers come from one connection, one direction, over loopback, on a single host, stated plainly as a bound on what this specific test can tell you rather than a production capacity claim. &lt;/p&gt;

&lt;p&gt;A benchmark that changes routing, authentication, and workload shape all at once, and only reports one before/after number, won't tell you which change actually caused the result.&lt;/p&gt;




&lt;h2&gt;
  
  
  When this doesn't matter at all
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fo38mqqzl8tll2tx1x7d5.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fo38mqqzl8tll2tx1x7d5.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;This throughput optimization is unlikely to materially improve a hop dominated by short, frequent calls.&lt;/p&gt;

&lt;p&gt;Start, stop, and observe calls were never bandwidth-bound, and kTLS adds little to them, since the handshake stays in userspace and dominates a short connection.&lt;/p&gt;

&lt;p&gt;No end-to-end latency numbers have been published for the new path, so this piece doesn't infer any. &lt;/p&gt;

&lt;p&gt;Those calls can still see a different failure model, unrelated to throughput.&lt;/p&gt;

&lt;p&gt;A forwarder that never parses a response cannot synthesize an application-level status code. In Tensorlake’s implementation, a half-close is forwarded as a half-close and an upstream reset reaches the client as a reset, rather than being converted into a proxy-manufactured response or timeout. TCP defines the underlying half-close and reset signals, but transparent propagation is a property of this forwarder’s implementation.&lt;/p&gt;

&lt;p&gt;A bandwidth gain and a latency gain are two different claims. A workload bound by connection setup rather than data volume can see close to nothing from this change, even while bulk transfers improve substantially.&lt;/p&gt;

&lt;p&gt;&lt;a href="/Users/divyyadav/article/second_article/images/06_workload_impact_comparison.png" class="article-body-image-wrapper"&gt;&lt;img src="/Users/divyyadav/article/second_article/images/06_workload_impact_comparison.png" alt="When L4 Optimization Matters vs When It Does Not"&gt;&lt;/a&gt;&lt;/p&gt;




&lt;h2&gt;
  
  
  The takeaway
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fdfourbqo3ezvt9oo6q71.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fdfourbqo3ezvt9oo6q71.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Use this sequence on your own ingress path, independent of what any one platform did:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;
&lt;strong&gt;Find the hop that's actually costing you something&lt;/strong&gt;, and check whether the cost is per-byte, per-connection, or per-request. Bulk transfers expose parsing, buffering, and copy costs. Short calls stay dominated by connection setup and the handshake.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Name what that hop currently does that depends on reading traffic&lt;/strong&gt;: routing, retries, deadlines, liveness, honest errors. All of it has to move elsewhere or get rebuilt on a signal the lower layer can actually see.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;If the hop still needs to authenticate its traffic&lt;/strong&gt;, work out separately whether you can keep that authentication without paying for a userspace copy on every byte. That's a distinct problem from removing the parsing layer, and conflating the two is how a technique like kTLS ends up credited with a win it didn't cause.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Benchmark the removal in three stages under concurrency&lt;/strong&gt;, before you believe the number.&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;If the hop in question is mostly handling small, frequent calls, this is often a lower-priority throughput optimization unless measurements show a meaningful per-byte or proxy-CPU bottleneck. The redesign may still be worthwhile for lifecycle isolation or a cleaner failure model.&lt;/p&gt;

&lt;p&gt;&lt;a href="/Users/divyyadav/article/second_article/images/07_4step_ingress_checklist_roadmap.png" class="article-body-image-wrapper"&gt;&lt;img src="/Users/divyyadav/article/second_article/images/07_4step_ingress_checklist_roadmap.png" alt="4-Step Checklist for Optimizing Ingress Proxy Chains"&gt;&lt;/a&gt;&lt;/p&gt;




&lt;h2&gt;
  
  
  References
&lt;/h2&gt;

&lt;ul&gt;
&lt;li&gt;&lt;a href="https://www.tensorlake.ai/blog/near-zero-overhead-sandbox-networking" rel="noopener noreferrer"&gt;Tensorlake Blog: Near Zero-Overhead Sandbox Networking&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://docs.tensorlake.ai/sandboxes/networking" rel="noopener noreferrer"&gt;Tensorlake Docs: Sandbox Networking&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://docs.tensorlake.ai/sandboxes/lifecycle" rel="noopener noreferrer"&gt;Tensorlake Docs: Sandbox Lifecycle&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://docs.kernel.org/networking/tls.html" rel="noopener noreferrer"&gt;Linux kernel documentation: Kernel TLS&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://man7.org/linux/man-pages/man2/splice.2.html" rel="noopener noreferrer"&gt;Linux manual page: splice(2)&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://datatracker.ietf.org/doc/html/rfc8446" rel="noopener noreferrer"&gt;IETF RFC 8446: The Transport Layer Security (TLS) Protocol Version 1.3&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://datatracker.ietf.org/doc/html/rfc9113" rel="noopener noreferrer"&gt;IETF RFC 9113: HTTP/2&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://datatracker.ietf.org/doc/html/rfc9293" rel="noopener noreferrer"&gt;IETF RFC 9293: Transmission Control Protocol (TCP)&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;

</description>
      <category>ai</category>
      <category>programming</category>
      <category>webdev</category>
      <category>architecture</category>
    </item>
    <item>
      <title>[Boost]</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Fri, 21 Aug 2026 17:29:04 +0000</pubDate>
      <link>https://dev.to/divy_ai/-3pm0</link>
      <guid>https://dev.to/divy_ai/-3pm0</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/divy_ai/best-ai-document-processing-tools-in-2026-top-5-compared-for-developers-and-engineering-teams-m5n" class="crayons-story__hidden-navigation-link"&gt;Best AI Document Processing Tools in 2026: Top 5 Compared for Developers and Engineering Teams&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/divy_ai" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F3845890%2Fc910b35f-87b4-4888-b116-ad2cde414360.png" alt="divy_ai profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/divy_ai" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Divy Yadav
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Divy Yadav
                
                
              
              &lt;div id="story-author-preview-content-4455660" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/divy_ai" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F3845890%2Fc910b35f-87b4-4888-b116-ad2cde414360.png" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Divy Yadav&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/divy_ai/best-ai-document-processing-tools-in-2026-top-5-compared-for-developers-and-engineering-teams-m5n" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Aug 21&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/divy_ai/best-ai-document-processing-tools-in-2026-top-5-compared-for-developers-and-engineering-teams-m5n" id="article-link-4455660"&gt;
          Best AI Document Processing Tools in 2026: Top 5 Compared for Developers and Engineering Teams
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/programming"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;programming&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/machinelearning"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;machinelearning&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/datascience"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;datascience&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/divy_ai/best-ai-document-processing-tools-in-2026-top-5-compared-for-developers-and-engineering-teams-m5n" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;1&lt;span class="hidden s:inline"&gt;&amp;nbsp;reaction&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/divy_ai/best-ai-document-processing-tools-in-2026-top-5-compared-for-developers-and-engineering-teams-m5n#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            15 min read
          &lt;/small&gt;
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>Best AI Document Processing Tools in 2026: Top 5 Compared for Developers and Engineering Teams</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Fri, 21 Aug 2026 17:09:56 +0000</pubDate>
      <link>https://dev.to/divy_ai/best-ai-document-processing-tools-in-2026-top-5-compared-for-developers-and-engineering-teams-m5n</link>
      <guid>https://dev.to/divy_ai/best-ai-document-processing-tools-in-2026-top-5-compared-for-developers-and-engineering-teams-m5n</guid>
      <description>&lt;p&gt;&lt;strong&gt;Every vendor on this list claims "99% accuracy." None of them are lying, and none of them are telling you the whole story.&lt;/strong&gt;&lt;/p&gt;




&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fk81fxn8879vyuuno3e3p.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fk81fxn8879vyuuno3e3p.png" alt="Photo from AI" width="" height=""&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;I pulled up an invoice while writing this. Nothing exotic. &lt;/p&gt;

&lt;p&gt;A print shop billing a camera store for three line items, a 20% discount, tax, a total due.&lt;/p&gt;

&lt;p&gt;The kind of document any tool on this list would claim to handle without blinking.&lt;/p&gt;

&lt;p&gt;Except it has four payment-mode checkboxes with only one ticked. &lt;/p&gt;

&lt;p&gt;A "past dues inclusive" checkbox sitting next to the issue date. A discount buried inside a line label instead of its own field.&lt;/p&gt;

&lt;p&gt;And a stray line of placeholder text, "Describe your item," left behind by whatever invoice template made the file. Sitting exactly where a lazy parser would scoop it up as if it were real content.&lt;/p&gt;

&lt;p&gt;None of that is unusual. It's a Tuesday. It's what your accounts-payable inbox actually looks like once you get past the one vendor you tested against during the demo.&lt;/p&gt;

&lt;p&gt;*&lt;em&gt;That's the whole test for any AI document processing software. Not whether a tool can read a clean sample PDF, but whether it survives the messy, half-filled, inconsistent documents your real vendors send you every day.&lt;br&gt;
*&lt;/em&gt;&lt;br&gt;
I spent the last several days pulling apart the architecture, the pricing, and the actual verified capabilities of five tools built to handle exactly that. Here's what I found, and where each one quietly falls short.&lt;/p&gt;




&lt;h2&gt;
  
  
  What Actually Separates These Tools
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ftg8y12fkwrkr1ryx09z3.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ftg8y12fkwrkr1ryx09z3.png" alt="Photo from AI" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;"Accuracy" is the number every vendor leads with. It's also the least useful number for comparing tools, because each vendor tests it on their own benchmark, their own documents, under their own definition of "correct."&lt;/p&gt;

&lt;p&gt;If you're evaluating ai for document processing, what actually decides whether it survives production is less flashy than a benchmark score:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Layout preservation.&lt;/strong&gt; Does the parsing layer keep table structure, multi-column text, and checkbox state intact, or does it flatten everything into a stream of text that destroys the spatial relationships a human reader relies on?&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Schema and prompt automation.&lt;/strong&gt; Do you manually define every field and write every extraction rule, or does the platform infer structure from sample documents?&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Validation and confidence.&lt;/strong&gt; Does a wrong extraction look identical to a right one, or does the system flag uncertainty before bad data reaches your database?&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Deployment flexibility.&lt;/strong&gt; Can this run in your VPC, on-prem, or only in the vendor's cloud? That answer alone rules tools in or out for regulated industries.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Pricing you can actually model.&lt;/strong&gt; Per-page, per-credit, or "talk to sales"? The difference changes whether a finance team can approve this without a quarter-long procurement cycle.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;I evaluated the shortlist against these five criteria, not just benchmark scores. Here's how they stack up.&lt;/p&gt;




&lt;h2&gt;
  
  
  1. Unstract: Best Overall for Production-Grade Document Automation
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ff08gqx14l3la0gtd9jxe.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ff08gqx14l3la0gtd9jxe.png" alt="Photo from AI" width="" height=""&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Overview
&lt;/h3&gt;

&lt;p&gt;&lt;a href="https://unstract.com" rel="noopener noreferrer"&gt;Unstract&lt;/a&gt; is an open-source, LLM-native document processing platform built by Zipstack. It ships in three forms: a free, self-hosted open-source edition under AGPL-3.0, a managed Cloud edition, and an On-Premise Enterprise edition for regulated environments.&lt;/p&gt;

&lt;p&gt;The &lt;a href="https://github.com/Zipstack/unstract" rel="noopener noreferrer"&gt;GitHub repository&lt;/a&gt; sits at 6.6k stars and 629 forks, with 1,604 commits and 547 releases behind it. That's a decent sign of a project still being actively built, not one abandoned after a launch blog post.&lt;/p&gt;

&lt;p&gt;It's built for teams who want document extraction to work like software they control, not a black box they rent. You define what to extract using prompts instead of training custom models per document type, and it deploys as an API endpoint or an ETL pipeline into your existing data stack.&lt;/p&gt;

&lt;h3&gt;
  
  
  Key Capabilities
&lt;/h3&gt;

&lt;p&gt;Unstract's core differentiator is that it treats extraction as a pipeline with distinct, inspectable stages rather than a single opaque call.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Layout-aware parsing&lt;/strong&gt; comes from &lt;a href="https://unstract.com/llmwhisperer/" rel="noopener noreferrer"&gt;LLMWhisperer&lt;/a&gt;, Unstract's own OCR layer. Where standard OCR reads a page left to right and flattens a multi-column table into a jumbled text stream, LLMWhisperer preserves the spatial structure so the extraction LLM is reasoning over an intact table instead of noise. On that sample invoice, this is the difference between correctly separating the "Discount(20%)" label from the subtotal line, versus merging them into unusable text.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Schema and prompt automation&lt;/strong&gt; happens through the &lt;a href="https://unstract.com/agentic-prompt-studio/" rel="noopener noreferrer"&gt;Agentic Prompt Studio&lt;/a&gt;, a pipeline of six AI agents split into two stages.&lt;/p&gt;

&lt;p&gt;Three agents build the schema. A Summarizer reads each sample document on its own. A Uniformer merges duplicate fields across documents. A Finalizer turns the result into a clean, standard-compliant JSON schema.&lt;/p&gt;

&lt;p&gt;Three more agents build and test the extraction prompts. A Pattern Miner finds the labels and formatting clues behind each field. A Prompt Architect writes the actual extraction instructions. A Critic Dry-Runner stress-tests the prompt against the schema before it ever touches a real document.&lt;/p&gt;

&lt;p&gt;In practice, you feed it sample invoices, bank statements, or claims forms. It builds and checks the extraction logic. You don't write it by hand.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Confidence validation&lt;/strong&gt; comes from &lt;a href="https://unstract.com/llmchallenge/" rel="noopener noreferrer"&gt;LLMChallenge&lt;/a&gt;, which runs two LLMs in parallel on the same extraction, an extractor and a challenger, and only returns a value if both agree. Mismatches return &lt;code&gt;NULL&lt;/code&gt; instead of a silently wrong guess, and the full comparison log is available via API for debugging. This is available on Cloud and Enterprise plans, not the open-source edition.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Human-in-the-loop review&lt;/strong&gt; routes low-confidence extractions to a human reviewer with document highlighting, so a person corrects the field with the source document visible, not a blind form. This is included on Cloud plans starting at Starter.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Deployment and integration&lt;/strong&gt; is where Unstract's flexibility shows. It connects to nine LLM providers, including OpenAI, Anthropic, Azure OpenAI, AWS Bedrock, Google Gemini, and Mistral, plus Ollama for local, offline inference. It also connects to five vector databases.&lt;/p&gt;

&lt;p&gt;On the data side, it pulls from or pushes to a long list of destinations: Snowflake, Amazon Redshift, Google BigQuery, PostgreSQL, MySQL, S3, Dropbox, Google Drive, and more. It ships an MCP server and an n8n node too, so it slots into agent workflows and existing automation tools without custom glue code.&lt;/p&gt;

&lt;h3&gt;
  
  
  Strengths &amp;amp; Limitations
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Strengths:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Automated schema generation, dual-LLM validation, and native human review, in one integrated pipeline. None of the other four tools on this list offer all three together.&lt;/li&gt;
&lt;li&gt;LLM-agnostic. Bring your own OpenAI, Anthropic, or Bedrock key, so you're never locked into one model vendor's pricing or quality ceiling.&lt;/li&gt;
&lt;li&gt;The open-source edition is a real, usable product, not a crippled trial meant to push you toward a paid plan.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Limitations:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;LLM-agnostic pricing means your total cost includes LLM API usage on top of the platform fee. That takes more modeling than an all-in-one flat rate.&lt;/li&gt;
&lt;li&gt;LLMChallenge and the token-saving SinglePass and Summarized Extraction modes are Cloud and Enterprise features, not part of the free self-hosted edition.&lt;/li&gt;
&lt;li&gt;More flexibility means more surface area to configure. Multiple LLMs, multiple vector databases, multiple deployment modes, it's not a narrow, single-purpose tool you can set up in five minutes.&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  Ideal Use Case
&lt;/h3&gt;

&lt;p&gt;Engineering teams and automation leads who need document variety handled without a template-per-vendor maintenance burden, especially in finance, insurance, healthcare, and KYC workflows where both accuracy and auditability matter. It's also the strongest fit if you specifically need self-hosted or air-gapped deployment for compliance reasons, since that's a first-class option, not an enterprise afterthought.&lt;/p&gt;

&lt;h3&gt;
  
  
  Pricing Model
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Open Source:&lt;/strong&gt; Free, self-hosted via Docker Compose, AGPL-3.0 licensed.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Cloud Starter:&lt;/strong&gt; $499/month (or $416/month billed annually) for 5,000 pages/month, $0.10/page overage.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Cloud Growth:&lt;/strong&gt; $2,249/month (or $1,874/month billed annually) for 25,000 pages/month, $0.09/page overage.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Enterprise (Cloud or On-Prem):&lt;/strong&gt; custom pricing, adds SOC 2, ISO 27001, HIPAA, GDPR compliance, SSO/SAML, and a dedicated support SLA.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Free tier:&lt;/strong&gt; 100 pages/day, no credit card required, plus a 14-day full trial with a pre-configured LLM stack.&lt;/li&gt;
&lt;li&gt;LLMWhisperer is also sold standalone, from $1 per 1,000 pages for native-text PDFs up to $15 per 1,000 pages for forms and tables with checkbox detection.&lt;/li&gt;
&lt;/ul&gt;




&lt;h2&gt;
  
  
  2. Reducto: Best for Developer-First RAG and LLM Pipelines
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ff5tho54vxqub974b8p3a.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ff5tho54vxqub974b8p3a.png" alt="Photo from AI" width="" height=""&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Overview
&lt;/h3&gt;

&lt;p&gt;Where Unstract is built for teams who want a full pipeline, &lt;a href="https://reducto.ai" rel="noopener noreferrer"&gt;Reducto&lt;/a&gt; is built for one job done extremely well: turning messy documents into clean structured data for engineers feeding RAG systems, AI agents, and LLM applications. It's raised $108M in total funding, including a $75M Series B led by a16z, and reports having processed billions of pages for customers that include Harvey, Scale AI, and Vanta.&lt;/p&gt;

&lt;p&gt;Reducto treats documents as visual objects rather than raw text, which is the architectural choice behind its accuracy claims on messy, real-world files.&lt;/p&gt;

&lt;h3&gt;
  
  
  Key Capabilities
&lt;/h3&gt;

&lt;p&gt;Reducto's core engine combines computer vision with vision-language models in a multi-pass pipeline it calls Agentic OCR. A VLM agent reviews the baseline OCR output and corrects errors, similar to a human proofreading a first draft.&lt;/p&gt;

&lt;p&gt;Every extracted value comes back with a bounding box and a confidence score, so you can trace any output back to its exact spot on the source page.&lt;/p&gt;

&lt;p&gt;The platform ships five APIs: Parse, Extract, Classify, Split, and Edit, across 30+ file types including PDFs, spreadsheets, and slide decks. It reports a 0.90 score on the public RD-TableBench benchmark and claims roughly 20 percentage points of accuracy advantage over major cloud document APIs, on its own internal benchmarks. Worth checking against your own documents before you take that number at face value, as with any vendor-run benchmark.&lt;/p&gt;

&lt;h3&gt;
  
  
  Strengths &amp;amp; Limitations
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Strengths:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Strong performance on messy, non-templated documents, multi-column layouts, and complex financial tables. This is Reducto's core reputation.&lt;/li&gt;
&lt;li&gt;Bounding-box citations make it well suited for compliance-sensitive review work.&lt;/li&gt;
&lt;li&gt;Deployment options extend to VPC, on-prem, and air-gapped for enterprise customers.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Limitations:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;It's a parsing and extraction API, not a full workflow platform. You define your own schema by hand; there's no equivalent to Unstract's agent-driven schema generation.&lt;/li&gt;
&lt;li&gt;Native human review and correction workflows aren't a headline feature, so teams generally build that layer themselves.&lt;/li&gt;
&lt;li&gt;A signed BAA and zero data retention require the Growth tier, not the entry-level Standard plan.&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  Ideal Use Case
&lt;/h3&gt;

&lt;p&gt;Engineering teams building RAG pipelines or AI agents who need top-tier parsing on real, messy documents, and who have the engineering bandwidth to build their own schema definitions, review workflows, and downstream integrations around the API.&lt;/p&gt;

&lt;h3&gt;
  
  
  Pricing Model
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Standard:&lt;/strong&gt; first 15,000 credits free, then $0.015/credit; a standard parse costs 1 credit per page.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Growth:&lt;/strong&gt; custom pricing, required for Studio evaluations, zero data retention, a signed BAA, premium rate limits, and data residency endpoints.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Enterprise:&lt;/strong&gt; custom pricing for SSO/SAML, dedicated SLAs, and air-gapped deployment.&lt;/li&gt;
&lt;li&gt;No published free-forever tier beyond the initial credit grant.&lt;/li&gt;
&lt;/ul&gt;




&lt;h2&gt;
  
  
  3. Rossum: Best for High-Volume Transactional Invoice Processing
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ftxasjddy94gz8mfc4zvz.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ftxasjddy94gz8mfc4zvz.png" alt="Photo from AI" width="" height=""&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Overview
&lt;/h3&gt;

&lt;p&gt;&lt;a href="https://rossum.ai" rel="noopener noreferrer"&gt;Rossum&lt;/a&gt; picks a narrower fight than the other tools on this list. It's built specifically around transactional business documents, invoices, purchase orders, delivery notes, sales orders, and explicitly stays out of contracts or legal documents. That narrower scope is a deliberate trade, and it's sharpened what Rossum is good at. Over 450 organizations use it, including Bosch, Siemens, Panasonic, and Flexport.&lt;/p&gt;

&lt;p&gt;Worth knowing before you evaluate it: Coupa, the spend-management platform, acquired Rossum in May 2026. Rossum's product and pricing are unchanged as of this writing, but it's no longer an independent company. If long-term product direction matters to your decision, factor in that Rossum's roadmap now sits inside Coupa's.&lt;/p&gt;

&lt;h3&gt;
  
  
  Key Capabilities
&lt;/h3&gt;

&lt;p&gt;Rossum's extraction engine, called Aurora, is a proprietary AI model trained entirely in-house on millions of transactional documents. That's different from Unstract or Reducto, which call out to third-party LLM APIs. Rossum says this cuts the risk of your data leaking to an outside model provider. The trade-off: you can't swap in a different model if Rossum's own model underperforms on your specific document type.&lt;/p&gt;

&lt;p&gt;The platform supports 276 languages, including handwriting, and checks extracted data against your ERP master data automatically. It also includes what Rossum calls Loop Validation, a built-in human review step that feeds corrections back into the model over time.&lt;/p&gt;

&lt;p&gt;Rossum reports reaching 90%+ accuracy within 10 to 20 processed documents for a given layout, with a named case study (the Port of Rotterdam) cited at that level.&lt;/p&gt;

&lt;h3&gt;
  
  
  Strengths &amp;amp; Limitations
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Strengths:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Strong accuracy reputation for its specific niche, accounts payable and other transactional documents, well documented across independent reviews.&lt;/li&gt;
&lt;li&gt;ERP integrations (SAP, Coupa, Workday, Oracle, NetSuite, QuickBooks Online) are deep, not superficial.&lt;/li&gt;
&lt;li&gt;The built-in human review loop means less engineering work to stand up a review process, compared to a bare API.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Limitations:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Pricing is the clearest constraint. There's no public per-page rate, no free tier, and independent sources put the Starter plan floor around $18,000 a year.&lt;/li&gt;
&lt;li&gt;That Starter plan excludes custom business logic, master data matching, and most ERP integrations. Those need Business or Enterprise tiers, both quote-only.&lt;/li&gt;
&lt;li&gt;The proprietary, closed model means no flexibility to choose or swap the underlying AI. Real user reviews on G2 and Capterra consistently flag weaker accuracy on unusual layouts and non-English documents.&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  Ideal Use Case
&lt;/h3&gt;

&lt;p&gt;Mid-market to enterprise accounts payable teams processing high volumes of standardized transactional documents, where the budget supports a five-figure annual commitment and the document types stay within Rossum's transactional focus rather than expanding into contracts, claims, or general document variety.&lt;/p&gt;

&lt;h3&gt;
  
  
  Pricing Model
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Starter:&lt;/strong&gt; approximately $18,000/year minimum (published floor, per multiple independent sources), includes unlimited seats, email/API ingestion, and 12 months of document archive.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Business / Enterprise:&lt;/strong&gt; custom, quote-based, required for SSO, custom business logic, and most ERP connectors beyond the basics.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;SAP Marketplace reference tiers:&lt;/strong&gt; Silver at $40,000/year for 100,000 pages, Gold at $70,000/year for 250,000 pages.&lt;/li&gt;
&lt;li&gt;No free tier; trials and pilots are available through the sales process.&lt;/li&gt;
&lt;/ul&gt;




&lt;h2&gt;
  
  
  4. Nanonets: Best for No-Code Workflow Automation on a Budget
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ft57ovstzrfih8zt00esa.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ft57ovstzrfih8zt00esa.png" alt="Photo from AI" width="" height=""&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Overview
&lt;/h3&gt;

&lt;p&gt;&lt;a href="https://nanonets.com" rel="noopener noreferrer"&gt;Nanonets&lt;/a&gt; is an AI-powered IDP platform built around a no-code, block-based workflow builder rather than a developer-first API. It's designed so operations teams, not just engineers, can assemble a document automation workflow by chaining together blocks: extraction, classification, formatting, routing to an integration.&lt;/p&gt;

&lt;h3&gt;
  
  
  Key Capabilities
&lt;/h3&gt;

&lt;p&gt;Nanonets' extraction runs on its own OCR model, Nanonets-OCR-s, built on the Qwen2.5-VL vision-language model and trained on 250,000+ documents. It converts documents to structured markdown rather than raw text.&lt;/p&gt;

&lt;p&gt;The AI Guidelines feature lets you describe extraction logic in plain language instead of touching a rules engine, things like applying jurisdiction-specific VAT rules or pulling only certain pages from a mixed document.&lt;/p&gt;

&lt;p&gt;Pricing runs on a credit-per-block model. A document typically runs through four to six blocks (extraction, classification, formatting, integration) at $0.02 to $0.30 per block depending on complexity. Nanonets' own documentation estimates this lands under $2 per invoice end to end.&lt;/p&gt;

&lt;p&gt;It connects to ERP systems including SAP, Oracle, and Salesforce, and supports SOC 2, HIPAA, GDPR, and ISO 27001 compliance with private cloud and on-prem deployment options.&lt;/p&gt;

&lt;h3&gt;
  
  
  Strengths &amp;amp; Limitations
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Strengths:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;The free tier is genuinely usable: $200 in credits that never expire, enough to process thousands of documents before paying anything.&lt;/li&gt;
&lt;li&gt;The no-code block builder lowers the barrier for non-developers to own a workflow.&lt;/li&gt;
&lt;li&gt;Broad ERP connector support and multi-region data residency (US/EU/APAC) make it viable for teams outside pure engineering.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Limitations:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Per-block, per-run pricing is metered at every step. A document that touches five blocks bills five times, and a rework loop bills again, which makes total cost harder to predict at real scale than a flat per-page rate.&lt;/li&gt;
&lt;li&gt;Independent reviews note that advanced custom extraction can still need real setup time, and sometimes model training, despite the "template-free" positioning.&lt;/li&gt;
&lt;li&gt;No option to bring your own foundation LLM. You're using Nanonets' own model, unlike Unstract.&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  Ideal Use Case
&lt;/h3&gt;

&lt;p&gt;Operations-heavy teams, particularly finance and accounts payable groups, who want a workflow automation tool that non-engineers can configure and maintain, and whose document volume is moderate enough that per-block credit costs stay predictable.&lt;/p&gt;

&lt;h3&gt;
  
  
  Pricing Model
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Starter:&lt;/strong&gt; free to start with $200 in credits, credits never expire.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Usage-based blocks:&lt;/strong&gt; $0.02/run for simple operations, $0.10/run for standard AI blocks, $0.30/run for complex AI extraction.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Growth / Enterprise:&lt;/strong&gt; custom, quote-based, with volume discounts reported up to 40%.&lt;/li&gt;
&lt;li&gt;No flat monthly platform fee on the entry tier; you pay only for blocks run.&lt;/li&gt;
&lt;/ul&gt;




&lt;h2&gt;
  
  
  5. Google Document AI: Best for Teams Already Built on Google Cloud
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fcarjs4jd6qvz48rqezt6.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fcarjs4jd6qvz48rqezt6.png" alt="Photo from AI" width="" height=""&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Overview
&lt;/h3&gt;

&lt;p&gt;If your infrastructure already lives on Google Cloud, &lt;a href="https://cloud.google.com/document-ai" rel="noopener noreferrer"&gt;Google Document AI&lt;/a&gt; is probably already on your shortlist by default. It's Google Cloud's API-based document processing service, built on Vertex AI and organized around "processors": pre-trained or custom models for specific tasks, from general OCR to prebuilt invoice and tax-form parsers to generative custom extractors you can bootstrap with as few as 10 to 50 labeled examples.&lt;/p&gt;

&lt;p&gt;It's consumed entirely through REST/gRPC APIs and client libraries, with Document AI Workbench as the console for labeling, training, and evaluating processors, not a turnkey business-user interface.&lt;/p&gt;

&lt;h3&gt;
  
  
  Key Capabilities
&lt;/h3&gt;

&lt;p&gt;Document AI's processors split into three families: digitize (Enterprise Document OCR), extract (Form Parser, Layout Parser, Custom Extractor, and prebuilt parsers for invoices, receipts, W-2s, and other tax forms), and classify (custom splitters and classifiers for routing mixed document batches).&lt;/p&gt;

&lt;p&gt;Recent updates have introduced Gemini 3 Flash and Gemini 3 Pro-powered versions of the Layout Parser and Custom Extractor, currently in preview. That's a signal Google is folding its newer models into the processor stack, not leaving it as a static OCR product.&lt;/p&gt;

&lt;p&gt;Synchronous requests are capped at 10 pages per call. Anything larger needs the asynchronous batch API, which accepts up to 200 pages per file. Output includes bounding-box coordinates for every extracted value, so you can trace any field back to its exact source location.&lt;/p&gt;

&lt;h3&gt;
  
  
  Strengths &amp;amp; Limitations
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Strengths:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Integrates natively with BigQuery, Cloud Storage, and the rest of the GCP stack, no separate vendor relationship needed if you're already there.&lt;/li&gt;
&lt;li&gt;Per-page pricing is genuinely transparent and published, unlike two of the other tools on this list.&lt;/li&gt;
&lt;li&gt;The generative Custom Extractor can bootstrap from a small labeled set, a real strength for niche document types.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Limitations:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Costs scale in a way that's easy to underestimate. Enterprise OCR is cheap at $1.50 per 1,000 pages, but Custom Extractor and Form Parser run $30 per 1,000 pages, and real G2 reviews consistently flag pricing as a pain point once volume grows.&lt;/li&gt;
&lt;li&gt;No built-in workflow layer, no native human review interface, and no automated schema generation from sample documents. You're assembling a pipeline out of API calls, not adopting a packaged platform.&lt;/li&gt;
&lt;li&gt;GCP-native by design, so teams on AWS or Azure take on real integration overhead to use it.&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  Ideal Use Case
&lt;/h3&gt;

&lt;p&gt;Teams already standardized on Google Cloud who need specific pre-built processors (invoices, tax forms, identity documents) and are comfortable building the orchestration, review, and workflow layer themselves in exchange for tight GCP integration and transparent per-page pricing.&lt;/p&gt;

&lt;h3&gt;
  
  
  Pricing Model
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Enterprise Document OCR:&lt;/strong&gt; $1.50 per 1,000 pages (1 to 5 million pages/month), dropping to $0.60 per 1,000 pages above that.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Custom Extractor / Form Parser:&lt;/strong&gt; $30 per 1,000 pages (up to 1 million pages/month), dropping to $20 per 1,000 pages above that.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Layout Parser:&lt;/strong&gt; $10 per 1,000 pages, flat.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Custom splitter / classifier:&lt;/strong&gt; $5 per 1,000 pages, dropping to $3 per 1,000 above 1 million pages/month.&lt;/li&gt;
&lt;li&gt;No dedicated free tier beyond standard Google Cloud trial credits; custom processor hosting adds $0.05 per deployed version-hour.&lt;/li&gt;
&lt;/ul&gt;




&lt;h2&gt;
  
  
  Side-by-Side Comparison
&lt;/h2&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;&lt;strong&gt;Criteria&lt;/strong&gt;&lt;/th&gt;
&lt;th&gt;&lt;strong&gt;Unstract&lt;/strong&gt;&lt;/th&gt;
&lt;th&gt;&lt;strong&gt;Reducto&lt;/strong&gt;&lt;/th&gt;
&lt;th&gt;&lt;strong&gt;Rossum&lt;/strong&gt;&lt;/th&gt;
&lt;th&gt;&lt;strong&gt;Nanonets&lt;/strong&gt;&lt;/th&gt;
&lt;th&gt;&lt;strong&gt;Google Document AI&lt;/strong&gt;&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Architecture&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;LLM-agnostic, agentic pipeline&lt;/td&gt;
&lt;td&gt;CV + VLM, Agentic OCR&lt;/td&gt;
&lt;td&gt;Proprietary in-house model&lt;/td&gt;
&lt;td&gt;Own VLM-based OCR model&lt;/td&gt;
&lt;td&gt;Pretrained + generative processors&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Schema automation&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;Yes, six-agent auto schema + prompt generation&lt;/td&gt;
&lt;td&gt;No, manual schema definition&lt;/td&gt;
&lt;td&gt;No, pretrained for transactional docs&lt;/td&gt;
&lt;td&gt;Partial, via plain-language AI Guidelines&lt;/td&gt;
&lt;td&gt;No, manual per processor&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Native HITL review&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;Yes, on Cloud plans&lt;/td&gt;
&lt;td&gt;No, build your own&lt;/td&gt;
&lt;td&gt;Yes, Loop Validation&lt;/td&gt;
&lt;td&gt;Via workflow blocks&lt;/td&gt;
&lt;td&gt;No&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Confidence validation&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;Yes, dual-LLM LLMChallenge (Cloud/Ent.)&lt;/td&gt;
&lt;td&gt;Per-field confidence scores&lt;/td&gt;
&lt;td&gt;Per-field confidence scores&lt;/td&gt;
&lt;td&gt;Per-block confidence&lt;/td&gt;
&lt;td&gt;Per-field confidence scores&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Deployment options&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;OSS, Cloud, On-Prem/VPC/air-gapped&lt;/td&gt;
&lt;td&gt;Cloud, VPC, on-prem, air-gapped (Ent.)&lt;/td&gt;
&lt;td&gt;Cloud only&lt;/td&gt;
&lt;td&gt;Cloud, private cloud, on-prem&lt;/td&gt;
&lt;td&gt;Google Cloud only&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Entry pricing&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;Free (OSS) / $499/mo&lt;/td&gt;
&lt;td&gt;Free 15K credits / $0.015/credit&lt;/td&gt;
&lt;td&gt;~$18,000/year, no free tier&lt;/td&gt;
&lt;td&gt;Free ($200 credits)&lt;/td&gt;
&lt;td&gt;Pay-per-use, no platform fee&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Best fit&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;General-purpose production IDP&lt;/td&gt;
&lt;td&gt;RAG/LLM pipeline parsing&lt;/td&gt;
&lt;td&gt;High-volume transactional AP&lt;/td&gt;
&lt;td&gt;No-code ops workflows&lt;/td&gt;
&lt;td&gt;GCP-native teams&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;




&lt;h2&gt;
  
  
  Decision Matrix: Which Tool Fits Your Situation
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fy71lmg3c5j01cb2zpx4z.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fy71lmg3c5j01cb2zpx4z.png" alt="Photo from AI" width="" height=""&gt;&lt;/a&gt;&lt;/p&gt;




&lt;h2&gt;
  
  
  When Not to Buy Any of These
&lt;/h2&gt;

&lt;p&gt;Every tool on this list earns its cost when document variety is high, volume is real, and layouts change often enough that a hand-built parser becomes a maintenance job in itself. That's the honest threshold for when AI document processing automation pays for itself instead of adding a new vendor bill for no real gain.&lt;/p&gt;

&lt;p&gt;If you're processing invoices from one vendor with a layout that hasn't changed in three years, none of this infrastructure pays for itself. A basic PDF text extraction script and a handful of regex patterns will outperform any of these tools on cost, and you can build it in an afternoon.&lt;/p&gt;

&lt;p&gt;The moment that calculus flips is the moment a second vendor's layout doesn't match the first, or a third document type shows up that your rules engine can't handle without another round of hand-tuning. That's the point where automated schema generation, layout-aware parsing, and confidence-based routing stop being nice-to-haves and start being the thing standing between you and a quietly broken production pipeline.&lt;/p&gt;




&lt;h2&gt;
  
  
  The Real Test
&lt;/h2&gt;

&lt;p&gt;Before picking a tool off this list, run your worst 20% of documents, not your cleanest ones, through whichever platform you're evaluating. The invoice with checkboxes nobody filled in consistently. The scan with a coffee stain. The vendor who changed their template last quarter and didn't tell you.&lt;/p&gt;

&lt;p&gt;Vendor benchmarks measure best-case documents on someone else's dataset. Your production pipeline will run on your worst-case documents, every day, indefinitely. The tool that survives that test is the one worth paying for.&lt;/p&gt;

</description>
      <category>ai</category>
      <category>programming</category>
      <category>machinelearning</category>
      <category>datascience</category>
    </item>
    <item>
      <title>[Boost]</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Sat, 15 Aug 2026 12:07:07 +0000</pubDate>
      <link>https://dev.to/divy_ai/-3299</link>
      <guid>https://dev.to/divy_ai/-3299</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc" class="crayons-story__hidden-navigation-link"&gt;3 Statement Descriptor Mistakes That Quietly Turn Real Purchases Into Chargebacks&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/divy_ai" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F3845890%2Fc910b35f-87b4-4888-b116-ad2cde414360.png" alt="divy_ai profile" class="crayons-avatar__image" width="800" height="800"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/divy_ai" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Divy Yadav
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Divy Yadav
                
                
              
              &lt;div id="story-author-preview-content-4395892" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/divy_ai" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F3845890%2Fc910b35f-87b4-4888-b116-ad2cde414360.png" class="crayons-avatar__image" alt="" width="800" height="800"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Divy Yadav&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Aug 14&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc" id="article-link-4395892"&gt;
          3 Statement Descriptor Mistakes That Quietly Turn Real Purchases Into Chargebacks
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag crayons-tag--filled  " href="/t/news"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;news&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/fintech"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;fintech&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/productivity"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;productivity&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="24" height="24"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;1&lt;span class="hidden s:inline"&gt;&amp;nbsp;reaction&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            4 min read
          &lt;/small&gt;
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>[Boost]</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Fri, 14 Aug 2026 10:35:27 +0000</pubDate>
      <link>https://dev.to/divy_ai/-477p</link>
      <guid>https://dev.to/divy_ai/-477p</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc" class="crayons-story__hidden-navigation-link"&gt;3 Statement Descriptor Mistakes That Quietly Turn Real Purchases Into Chargebacks&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/divy_ai" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F3845890%2Fc910b35f-87b4-4888-b116-ad2cde414360.png" alt="divy_ai profile" class="crayons-avatar__image" width="800" height="800"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/divy_ai" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Divy Yadav
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Divy Yadav
                
                
              
              &lt;div id="story-author-preview-content-4395892" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/divy_ai" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F3845890%2Fc910b35f-87b4-4888-b116-ad2cde414360.png" class="crayons-avatar__image" alt="" width="800" height="800"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Divy Yadav&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Aug 14&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc" id="article-link-4395892"&gt;
          3 Statement Descriptor Mistakes That Quietly Turn Real Purchases Into Chargebacks
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag crayons-tag--filled  " href="/t/news"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;news&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/fintech"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;fintech&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/productivity"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;productivity&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="24" height="24"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;1&lt;span class="hidden s:inline"&gt;&amp;nbsp;reaction&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            4 min read
          &lt;/small&gt;
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>3 Statement Descriptor Mistakes That Quietly Turn Real Purchases Into Chargebacks</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Fri, 14 Aug 2026 10:34:28 +0000</pubDate>
      <link>https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc</link>
      <guid>https://dev.to/divy_ai/3-statement-descriptor-mistakes-that-quietly-turn-real-purchases-into-chargebacks-2pfc</guid>
      <description>&lt;p&gt;I spent a week chasing what looked like a fraud spike. It wasn't fraud. It was 22 characters of bad UX that I'd shipped myself.&lt;/p&gt;

&lt;p&gt;A support ticket queue that's usually quiet started filling up with the same complaint: "I didn't make this charge." Different customers, different amounts, same pattern.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;My first instinct was fraud&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Rotate keys, check for a leaked webhook secret, the usual on-call checklist. None of it panned out. Every single "disputed" charge turned out to be a real purchase, made by the actual customer, on the actual date.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The problem wasn't the payment&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;It was what the payment looked like on a bank statement.&lt;/p&gt;




&lt;h2&gt;
  
  
  The 22-character problem nobody warns you about
&lt;/h2&gt;

&lt;p&gt;Card networks give you roughly 22 characters to explain a charge to your customer. Not 22 words. Twenty-two characters, including spaces.&lt;/p&gt;

&lt;p&gt;If you're on Stripe, this isn't a soft guideline, it's enforced. Statement descriptors are capped at 22 characters, and you can't use &lt;code&gt;&amp;lt;&lt;/code&gt;, &lt;code&gt;&amp;gt;&lt;/code&gt;, &lt;code&gt;'&lt;/code&gt;, &lt;code&gt;"&lt;/code&gt;, or &lt;code&gt;*&lt;/code&gt;, and the descriptor can't be only numbers.&lt;/p&gt;

&lt;p&gt;That limit is tight enough that your actual product name often doesn't fit. "RunClub Premium Monthly" doesn't fit in 22 characters. Neither does most of what a real business is called.&lt;/p&gt;

&lt;p&gt;So it gets cut. And a cut-off name looks nothing like the thing your customer remembers buying.&lt;/p&gt;

&lt;h2&gt;
  
  
  Where dynamic descriptors make it worse
&lt;/h2&gt;

&lt;p&gt;If you're setting a dynamic suffix so each transaction shows different context, the math gets tighter, not looser. Here's a basic example using Stripe's Payment Intents API:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;curl https://api.stripe.com/v1/payment_intents &lt;span class="se"&gt;\&lt;/span&gt;
  &lt;span class="nt"&gt;-u&lt;/span&gt; &lt;span class="s2"&gt;"sk_test_yourkey:"&lt;/span&gt; &lt;span class="se"&gt;\&lt;/span&gt;
  &lt;span class="nt"&gt;-d&lt;/span&gt; &lt;span class="nv"&gt;amount&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;1099 &lt;span class="se"&gt;\&lt;/span&gt;
  &lt;span class="nt"&gt;-d&lt;/span&gt; &lt;span class="nv"&gt;currency&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;usd &lt;span class="se"&gt;\&lt;/span&gt;
  &lt;span class="nt"&gt;-d&lt;/span&gt; &lt;span class="s2"&gt;"statement_descriptor_suffix=RUN CLUB APR"&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Your account-level prefix, an asterisk, a space, and your suffix all share that same 22-character budget. Stripe allots roughly 10 characters for the dynamic part by default. Go over, and Stripe truncates it, sometimes into something that reads as gibberish or, worse, as a duplicate of your own business name stacked twice.&lt;/p&gt;

&lt;p&gt;I've seen this exact failure mode described in payment platform support docs: a merchant's static prefix and their dynamic suffix both containing a version of the brand name, so the customer sees the company name twice in a row and assumes something's broken. Nothing was broken. It was just two systems both trying to be helpful in the same 22 characters.&lt;/p&gt;

&lt;h2&gt;
  
  
  Testing this before it reaches a real customer
&lt;/h2&gt;

&lt;p&gt;Here's the part most teams skip: nobody on the engineering side actually checks what the final descriptor renders as on a real bank statement, because you can't easily see that in a sandbox environment.&lt;/p&gt;

&lt;p&gt;What I started doing instead: after setting a new descriptor pattern, I paste the exact string I'm about to ship into &lt;a href="https://unknowncharges.com/tools/decode" rel="noopener noreferrer"&gt;UnknownCharges' descriptor decoder&lt;/a&gt;, the same tool a confused customer would use if they searched their statement. If a decoder built for confused consumers can't make sense of what I just shipped, neither will they.&lt;/p&gt;

&lt;p&gt;It's a five-second sanity check, and it catches the "wait, this reads as nonsense" problem before a support ticket does.&lt;/p&gt;

&lt;h2&gt;
  
  
  Understanding processor prefixes as the person who set them
&lt;/h2&gt;

&lt;p&gt;If your payment stack routes through a processor like Square, Stripe Connect, or Toast instead of settling directly, your customers see the processor's name before they see yours. That's not a bug in your integration. It's how card networks handle sub-merchants.&lt;/p&gt;

&lt;p&gt;&lt;code&gt;SQ *YOURBUSINESS&lt;/code&gt; is a completely normal, expected pattern for Square-routed charges. Customers who don't know that read it as a stranger's name attached to their statement. UnknownCharges keeps a running &lt;a href="https://unknowncharges.com/prefix/sq" rel="noopener noreferrer"&gt;reference page for the SQ* prefix&lt;/a&gt; specifically because this exact confusion generates enough support volume across enough businesses to be worth documenting on its own.&lt;/p&gt;

&lt;p&gt;If you're building on top of a processor, it's worth reading how your specific prefix looks to an end user who has zero context on your backend architecture, because that's the only context your customer actually has.&lt;/p&gt;

&lt;h2&gt;
  
  
  What happens after the customer still disputes it
&lt;/h2&gt;

&lt;p&gt;Even with a clean descriptor, some percentage of customers will still file a dispute instead of contacting you first. That's a habit, not a fixable bug, and it's worth understanding what happens on their end so your support responses actually match reality.&lt;/p&gt;

&lt;p&gt;Credit card disputes in the US run under the Fair Credit Billing Act, with a 60-day window from the statement date. Debit card disputes run under Regulation E instead, with a much tighter two-business-day window before liability caps start rising. If your support team is telling customers "just call your bank" without knowing which rule applies to which card type, you're giving advice that's sometimes wrong.&lt;/p&gt;

&lt;p&gt;I've pointed support-facing teammates at &lt;a href="https://unknowncharges.com/guide" rel="noopener noreferrer"&gt;UnknownCharges' dispute playbook&lt;/a&gt; when writing our own canned responses, mainly because it lays out the credit-versus-debit distinction more clearly than the two paragraphs buried in most payment processor docs.&lt;/p&gt;




&lt;h2&gt;
  
  
  A quick checklist before you ship a descriptor change
&lt;/h2&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Situation&lt;/th&gt;
&lt;th&gt;What to check&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;Setting a static prefix&lt;/td&gt;
&lt;td&gt;Does it fit in 22 characters, no truncation guesswork?&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Adding a dynamic suffix&lt;/td&gt;
&lt;td&gt;Does prefix + &lt;code&gt;*&lt;/code&gt; + space + suffix stay under 22 total?&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Routing through Square, Stripe Connect, or similar&lt;/td&gt;
&lt;td&gt;Does the resulting &lt;code&gt;PROCESSOR *NAME&lt;/code&gt; pattern make sense out of context?&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Multi-line-item or subscription billing&lt;/td&gt;
&lt;td&gt;Is the descriptor different enough per product to avoid support confusion?&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Before shipping any descriptor change&lt;/td&gt;
&lt;td&gt;Paste the exact rendered string into a decoder and read it like a confused customer would&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;




&lt;p&gt;None of this is complicated engineering. It's just the one part of the payment flow that has zero automated tests, because the failure mode isn't a 500 error. It's a confused human, three weeks later, staring at their banking app.&lt;/p&gt;

&lt;p&gt;Statement descriptors are the last UI your product shows a customer, and it's the one team most engineers never design on purpose.&lt;/p&gt;

</description>
      <category>fintech</category>
      <category>ai</category>
      <category>productivity</category>
      <category>news</category>
    </item>
    <item>
      <title>PDF Structural Extraction: Why Your AI Has Never Read a Single Page</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Sat, 01 Aug 2026 18:09:55 +0000</pubDate>
      <link>https://dev.to/divy_ai/pdf-structural-extraction-why-your-ai-has-never-read-a-single-page-2f4k</link>
      <guid>https://dev.to/divy_ai/pdf-structural-extraction-why-your-ai-has-never-read-a-single-page-2f4k</guid>
      <description>&lt;p&gt;Ask an AI tool to read a scanned contract, and something strange happens: it tells you it understood the document.&lt;/p&gt;

&lt;p&gt;It didn't.&lt;/p&gt;

&lt;p&gt;I've watched this go wrong enough times that it stopped surprising me. Most PDF tools don't read a PDF at all. &lt;/p&gt;

&lt;p&gt;They copy whatever characters happen to sit on the page, in whatever order the page happens to store them, and call that "extraction." A table becomes a wall of numbers with no rows. A form becomes text with no fields. &lt;/p&gt;

&lt;p&gt;A scanned page becomes nothing, because there was never any text there to copy, just a picture of text pretending to be text.&lt;/p&gt;

&lt;p&gt;If you've ever asked an AI assistant a question about a PDF and gotten a confidently wrong answer, this is very likely why. Here's what's really happening, and what it takes to fix it, including &lt;a href="https://developer-api.foxit.com/developer-blogs/api-guides-tutorials/pdf-structural-extraction-engine/" rel="noopener noreferrer"&gt;the exact engineering breakdown&lt;/a&gt; this piece is pulling from.&lt;/p&gt;




&lt;h2&gt;
  
  
  What you see is not what's stored
&lt;/h2&gt;

&lt;p&gt;Open a PDF and you see a clean page: a heading, a paragraph, a tidy little table. None of that is actually in the file. Not really.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;WHAT YOU SEE ON THE PAGE            WHAT THE PDF ACTUALLY STORES

  Q3 Report                           "Q" at (72, 40)
  ─────────                           "3" at (84, 40)
  Revenue    Costs                    " " at (96, 40)
  142,500    88,200                   "R" at (72, 58)
                                       "e" at (81, 58)
  (a clean heading                    "v" at (90, 58)
   and a tidy table)                  ... character by character,
                                       position by position

                                     no row. no column. no table.
                                     just ink, and where it goes.
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;I didn't quite believe this the first time someone explained it to me either. A PDF, underneath the page you see, is closer to a set of typesetting instructions: put this exact letter here, at this exact spot, in this font, at this size. It never says "this is a table." It never says "this is a heading." It just says where the ink goes.&lt;/p&gt;

&lt;p&gt;That's fine for printing a page. It falls apart the moment you want a computer to pull meaning out of that page instead of just displaying it. &lt;a href="https://developer-api.foxit.com/developer-blogs/api-guides-tutorials/pdf-structural-extraction-engine/" rel="noopener noreferrer"&gt;Foxit's technical breakdown&lt;/a&gt; walks through this exact storage model in far more depth than a Medium post has room for.&lt;/p&gt;




&lt;h2&gt;
  
  
  Copying text isn't the same as understanding it
&lt;/h2&gt;

&lt;p&gt;I used to assume "PDF extraction" meant a computer read the document roughly the way I would. It doesn't. Not by default.&lt;/p&gt;

&lt;p&gt;Most extraction tools do exactly one thing well: they read those typesetting instructions and hand you back the characters, roughly in the order they were drawn. Engineers call this content serialization. You don't need the term. You just need to see where it breaks.&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;A table with mismatched row heights, extremely common in financial statements, gets its rows silently merged into the wrong place.&lt;/li&gt;
&lt;li&gt;A signature, a stamp, or a logo isn't text at all, so it never gets copied, even when it matters for a real decision downstream.&lt;/li&gt;
&lt;li&gt;A scanned page has no underlying characters whatsoever. It's a photograph. Ask a basic tool to "extract" it and you get back nothing.&lt;/li&gt;
&lt;li&gt;A form field can hold a value stored completely separately from what's visually printed on the page. Copy the visible text, and you can lose the actual data entirely.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;None of this is a bug in any one tool. It's the ceiling of what "copy the characters" was ever going to be able to do.&lt;/p&gt;




&lt;h2&gt;
  
  
  What actually understanding a document looks like
&lt;/h2&gt;

&lt;p&gt;Real &lt;a href="https://developer-api.foxit.com/developer-blogs/api-guides-tutorials/pdf-structural-extraction-engine/" rel="noopener noreferrer"&gt;structural extraction&lt;/a&gt; asks a different question entirely. Not "what characters are here," but "what is this, and how does it relate to everything around it."&lt;/p&gt;

&lt;p&gt;In practice, that's a few concrete things happening at once:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Two blocks of text sitting side by side get recognized as two separate columns, not one garbled sentence.&lt;/li&gt;
&lt;li&gt;A cluster of numbers gets tied to a specific row of a specific table, not left floating as a page full of digits.&lt;/li&gt;
&lt;li&gt;A scanned page runs through actual character recognition first, so a photo of text becomes real, usable text.&lt;/li&gt;
&lt;li&gt;A form field's stored value gets read directly, instead of guessed at from whatever happens to be visually printed nearby.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;A properly built system doesn't stop at "text" versus "not text" either. It typically recognizes something like a dozen distinct kinds of content on a page: titles, headings, ordinary paragraphs, tables, images, footnotes, hyperlinks, form fields, even stamped annotations, each one tagged for what it actually is.&lt;/p&gt;

&lt;p&gt;Get all of that right, and a PDF stops being a picture a computer merely displays. It becomes structured data a computer can use: feed to an AI, drop into a spreadsheet, route into a CRM.&lt;/p&gt;




&lt;h2&gt;
  
  
  A quick before-and-after
&lt;/h2&gt;

&lt;p&gt;Here's the same table, seen two ways. I find this comparison does more work than any explanation I could write around it.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;BASIC EXTRACTION                      STRUCTURAL EXTRACTION

"Q3 Revenue Total   142,500           { "type": "table",
Q3 Costs    88,200   Net    54,300"     "rows": [
                                          ["Q3 Revenue", "142,500"],
one wall of characters,                  ["Q3 Costs", "88,200"],
no rows, no columns,                     ["Net", "54,300"]
no idea what belongs                   ]}
where

                                       every value knows exactly
                                       which row and column
                                       it belongs to
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Left side, a machine can display it. Right side, a machine can actually use it. This is the exact transformation &lt;a href="https://developer-api.foxit.com/developer-blogs/api-guides-tutorials/pdf-structural-extraction-engine/" rel="noopener noreferrer"&gt;a properly built structural extraction engine&lt;/a&gt; is responsible for.&lt;/p&gt;




&lt;h2&gt;
  
  
  How this works, behind the scenes
&lt;/h2&gt;

&lt;p&gt;You don't need to write any code to understand the shape of it. A structural extraction system like this usually runs in &lt;a href="https://developer-api.foxit.com/developer-blogs/api-guides-tutorials/pdf-structural-extraction-engine/" rel="noopener noreferrer"&gt;four plain steps&lt;/a&gt;.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;UPLOAD          ANALYZE            CHECK STATUS         DOWNLOAD
(get back a  ──▶ (runs in the  ──▶  (poll every    ──▶  (get the finished,
 ticket for       background,        couple of            structured file)
 the file)        not instantly)     seconds)
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;You hand over the document, and the system stores it and gives you back a ticket number for that file. You ask it to analyze the document, and because genuinely reading a messy page takes real processing time, it starts the job in the background instead of answering instantly. You check on that job every couple of seconds until it reports back done. Then you download the finished result: a structured file listing every table, form, heading, paragraph, and image the system found, each one tagged with exactly what it is and exactly where it sits on the page.&lt;/p&gt;

&lt;p&gt;The first time I watched one of these jobs run, I kept refreshing before I realized the wait was the whole point, not a glitch. There's no single instant "read this PDF" button, because genuinely reading a messy real-world document was never going to be an instant operation. It's a careful pass through the whole page, once, done properly.&lt;/p&gt;




&lt;h2&gt;
  
  
  The proof isn't just anecdotal
&lt;/h2&gt;

&lt;p&gt;This isn't just an opinion floating around engineering teams. NVIDIA ran its own comparison in 2025, testing a purpose-built extraction pipeline against a general-purpose AI vision model on real financial filings and reports. The dedicated extraction approach came out roughly 7% more accurate at pulling back the right information, and processed pages about 32 times faster.&lt;/p&gt;

&lt;p&gt;I'll admit the size of that speed gap surprised me. The general-purpose model wasn't bad at describing a page in broad strokes. It was worse specifically at the boring, structural part: keeping every number tied to the correct row, every field tied to the correct label, consistently, every single time. &lt;/p&gt;

&lt;p&gt;That consistency is exactly what a real business process needs. A rough guess can't reliably provide it, no matter how confident it sounds. It's exactly this gap that &lt;a href="https://developer-api.foxit.com/developer-blogs/api-guides-tutorials/pdf-structural-extraction-engine/" rel="noopener noreferrer"&gt;a dedicated structural extraction engine&lt;/a&gt; is built to close.&lt;/p&gt;




&lt;h2&gt;
  
  
  The part that quietly wrecks AI assistants specifically
&lt;/h2&gt;

&lt;p&gt;If you've built or used any tool that lets an AI answer questions from your own documents, this is worth sitting with for a second.&lt;/p&gt;

&lt;p&gt;That kind of tool works by chopping documents into chunks, then handing the AI the most relevant chunks for a given question. &lt;/p&gt;

&lt;p&gt;If the reading order feeding those chunks is scrambled, say two columns merged into one garbled paragraph, or a table's numbers floating with no row labels, then no amount of clever search logic layered on top can recover the original meaning. You can't retrieve your way out of a bad starting chunk.&lt;/p&gt;

&lt;p&gt;Correct structure has to happen before retrieval, not after.&lt;/p&gt;




&lt;h2&gt;
  
  
  It's not just AI assistants that get burned
&lt;/h2&gt;

&lt;p&gt;RAG pipelines get most of the attention here, but they're not the only place this breaks down. I only started noticing this pattern once I stopped thinking of it as an "AI problem" at all.&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;A finance team pulling numbers into a dashboard needs every value tied to the correct row label, or the dashboard is just confidently wrong.&lt;/li&gt;
&lt;li&gt;A sales team auto-filling a CRM from scanned contracts needs the actual form field values, not a best guess at whatever text happened to sit near a checkbox.&lt;/li&gt;
&lt;li&gt;A compliance team building an audit trail needs to know which stamps, signatures, and annotations exist on a document, not just the paragraph text.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Different teams, different tools, the same root cause every time: something upstream copied characters instead of understanding structure.&lt;/p&gt;




&lt;h2&gt;
  
  
  Signs your own pipeline already has this problem
&lt;/h2&gt;

&lt;p&gt;A few blunt questions, if you're dealing with PDFs at any real volume. I've asked variations of these in enough conversations to know the honest answer usually comes with a pause first.&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Do your table exports ever have the right numbers sitting in the wrong row?&lt;/li&gt;
&lt;li&gt;Does anything break specifically on scanned or faxed documents, but work fine on ones typed directly into a PDF?&lt;/li&gt;
&lt;li&gt;Do form values ever come back blank even though the form clearly has data filled in?&lt;/li&gt;
&lt;li&gt;Does your AI assistant's answer quality quietly drop on longer, multi-column, or older documents specifically?&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;A yes to any of these usually traces back to the same root cause: something in the pipeline is copying characters, not reading structure.&lt;/p&gt;




&lt;h2&gt;
  
  
  Where this gets solved
&lt;/h2&gt;

&lt;p&gt;Companies building document-heavy products don't solve this by hoping the AI figures it out. They solve it at the extraction layer, before the AI ever sees the document, using tools purpose-built to recognize tables, forms, scanned pages, and headers as the distinct things they actually are, not just characters on a grid.&lt;/p&gt;

&lt;p&gt;Foxit's engineering team recently published a detailed technical walkthrough of exactly how this works under the hood, including the specific failure points in popular extraction libraries and how a properly built structural extraction engine avoids them. If you're building anything that touches PDFs at scale, invoices, contracts, scanned forms, it's worth the read: &lt;a href="https://developer-api.foxit.com/developer-blogs/api-guides-tutorials/pdf-structural-extraction-engine/" rel="noopener noreferrer"&gt;Inside Foxit's PDF Structural Extraction Engine&lt;/a&gt;.&lt;/p&gt;




&lt;h2&gt;
  
  
  The takeaway
&lt;/h2&gt;

&lt;p&gt;Next time an AI hands you a confidently wrong answer about a document, look upstream before blaming the model. Somewhere before the AI ever got involved, a table probably lost its rows, or a scanned page never got read at all.&lt;/p&gt;

&lt;p&gt;I think about this every time I see a demo where an AI "reads" a document flawlessly on stage. Stage demos use clean PDFs. Production doesn't. The gap between the two is exactly the gap this piece has been describing: copying characters versus genuinely understanding structure. Now you know which question to ask before you trust either one.&lt;/p&gt;




&lt;h2&gt;
  
  
  If you're building anything document-heavy
&lt;/h2&gt;

&lt;p&gt;Foxit's developer team publishes breakdowns like this one regularly: PDF APIs, document automation, agentic document workflows. If "here's what's actually happening under the hood" is useful to you, &lt;a href="https://www.linkedin.com/showcase/foxit-developer-solutions/" rel="noopener noreferrer"&gt;Foxit Developer Solutions on LinkedIn&lt;/a&gt; is where they post it first, and the original technical write-up this piece is based on is worth bookmarking directly: &lt;a href="https://developer-api.foxit.com/developer-blogs/api-guides-tutorials/pdf-structural-extraction-engine/" rel="noopener noreferrer"&gt;Inside Foxit's PDF Structural Extraction Engine&lt;/a&gt;.&lt;/p&gt;

&lt;p&gt;You can also follow &lt;a href="https://www.linkedin.com/company/foxit-corporation" rel="noopener noreferrer"&gt;Foxit's main LinkedIn page&lt;/a&gt; for the wider product news, outside of the developer-specific posts.&lt;/p&gt;




&lt;h2&gt;
  
  
  References and further reading
&lt;/h2&gt;

&lt;ul&gt;
&lt;li&gt;Foxit, &lt;a href="https://developer-api.foxit.com/developer-blogs/api-guides-tutorials/pdf-structural-extraction-engine/" rel="noopener noreferrer"&gt;"Inside Foxit's PDF Structural Extraction Engine"&lt;/a&gt;
&lt;/li&gt;
&lt;li&gt;LinkedIn: "&lt;a href="https://www.linkedin.com/company/foxit-corporation" rel="noopener noreferrer"&gt;https://www.linkedin.com/company/foxit-corporation&lt;/a&gt;"&lt;/li&gt;
&lt;/ul&gt;

</description>
      <category>ai</category>
      <category>webdev</category>
      <category>programming</category>
      <category>productivity</category>
    </item>
    <item>
      <title>OpenCode Is Powerful. That's Exactly the Problem.</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Tue, 21 Jul 2026 12:31:40 +0000</pubDate>
      <link>https://dev.to/divy_ai/opencode-is-powerful-thats-exactly-the-problem-3po0</link>
      <guid>https://dev.to/divy_ai/opencode-is-powerful-thats-exactly-the-problem-3po0</guid>
      <description>&lt;p&gt;&lt;em&gt;OpenCode is the free, open source alternative to Claude Code, with full shell access to match. Here's how I ran it safely using a sandbox instead of my laptop.&lt;/em&gt;&lt;/p&gt;




&lt;p&gt;OpenCode is an open-source coding agent with a workflow similar to Claude Code.&lt;/p&gt;

&lt;p&gt;You type what you want, it reads your codebase, edits files, runs shell commands on its own. Same basic idea as Claude Code. Free though, open source, and you're not locked into one model.&lt;/p&gt;

&lt;p&gt;Which is great, until you actually sit with what "runs shell commands on its own" means.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The first time you hand a terminal agent that kind of access, there's one question you can't quite shake:&lt;/strong&gt; what happens the moment it runs something you didn't expect?&lt;/p&gt;

&lt;p&gt;Most people pick a folder they don't care about and hope for the best. Or they don't think about it at all until something breaks.&lt;/p&gt;

&lt;p&gt;I didn't love either option, so before I let OpenCode near anything that mattered, I went looking for a third one.&lt;/p&gt;




&lt;p&gt;That's how I ended up moving the agent's shell into a disposable sandbox from &lt;a href="https://www.tensorlake.ai/" rel="noopener noreferrer"&gt;Tensorlake&lt;/a&gt;, that runs these as a cloud service, while leaving everything else right where it was on my laptop.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Your Laptop
      │
      ▼
   OpenCode
      │
      ▼
Tensorlake Plugin
      │
      ▼
 Disposable Linux Sandbox
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Here's what that looked like, and what actually surprised me once I had it running.&lt;/p&gt;




&lt;h2&gt;
  
  
  What I Kept Picturing Before I Typed Anything
&lt;/h2&gt;

&lt;p&gt;A coding agent's real risk has nothing to do with intelligence. Most of the time it gets things right. What it doesn't do is pause. It doesn't stop to double-check whether the command it's about to fire off is the one you actually meant to approve.&lt;/p&gt;

&lt;p&gt;One scenario kept nagging at me before I typed anything. I ask it to clean up build artifacts, and it runs &lt;code&gt;rm -rf ./build&lt;/code&gt;. Except &lt;code&gt;./build&lt;/code&gt; turns out to be a symlink into somewhere I still needed, the kind of thing an agent skims right past, and honestly, the kind of thing I don't always catch either when I'm moving fast.&lt;/p&gt;

&lt;p&gt;Or I ask it to install a dependency. &lt;code&gt;npm install&lt;/code&gt; fires a postinstall script. That script rewrites a config file I never agreed to touch. I've seen postinstall scripts do weirder things than that with a human sitting right there watching.&lt;/p&gt;

&lt;p&gt;Neither one is a bug to be honest. &lt;/p&gt;

&lt;p&gt;That's just what shell access does when nothing stands between a command and your machine.&lt;/p&gt;

&lt;p&gt;Then I thought bigger than my own laptop. Same agent, but now it's refactoring a production monorepo that ten other engineers are actively pushing commits to. Same misread symlink. Same rogue postinstall script. Except now it's not just my afternoon on the line.&lt;/p&gt;

&lt;p&gt;&lt;code&gt;bash&lt;/code&gt;, &lt;code&gt;write&lt;/code&gt;, &lt;code&gt;edit&lt;/code&gt;. These aren't agent-specific features. They're shell and filesystem access, handed to a process making its own calls about what to run next.&lt;/p&gt;

&lt;p&gt;The instinct is to fix this with more caution. Review every diff. Approve every command. That works right up until it doesn't, because the entire point of an autonomous agent is that you eventually stop reviewing every single step.&lt;/p&gt;

&lt;p&gt;The real fix turned out to be a different blast radius. &lt;/p&gt;

&lt;p&gt;Put the agent's commands somewhere disposable instead of on my machine, and now a bad command won't harm my system.&lt;/p&gt;

&lt;p&gt;Running locally isn't wrong, to be clear. For a throwaway project, or a workflow you already trust, it's exactly what you want. What changes the math is pointing autonomous shell access at something a mistake would actually cost you.&lt;/p&gt;




&lt;h2&gt;
  
  
  Brain Local, Hands in a Sandbox
&lt;/h2&gt;

&lt;p&gt;One table covers the whole shift. The wiring behind it takes a little longer to explain.&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;&lt;/th&gt;
&lt;th&gt;&lt;strong&gt;Local OpenCode&lt;/strong&gt;&lt;/th&gt;
&lt;th&gt;&lt;strong&gt;OpenCode + Tensorlake&lt;/strong&gt;&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;Commands run&lt;/td&gt;
&lt;td&gt;On your laptop&lt;/td&gt;
&lt;td&gt;Remotely, in a sandbox&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Filesystem&lt;/td&gt;
&lt;td&gt;Your actual filesystem&lt;/td&gt;
&lt;td&gt;Isolated, disposable&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Dependencies&lt;/td&gt;
&lt;td&gt;Whatever's on your machine&lt;/td&gt;
&lt;td&gt;Disposable environment&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;A bad command affects&lt;/td&gt;
&lt;td&gt;Your machine&lt;/td&gt;
&lt;td&gt;The sandbox, not you&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;That's the outcome. The wiring is simpler than it sounds.&lt;/p&gt;

&lt;p&gt;Tensorlake ships a plugin called &lt;a href="https://www.npmjs.com/package/tensorlake-opencode" rel="noopener noreferrer"&gt;&lt;code&gt;tensorlake-opencode&lt;/code&gt;&lt;/a&gt;. The plugin doesn't replace OpenCode.It intercepts specific tool calls and reroutes them. Once I actually understood that distinction, the idea turned out simpler than I expected.&lt;/p&gt;

&lt;p&gt;OpenCode on your laptop, tools in a sandbox, that's the actual shape of it. The OpenCode harness itself, the interface, your session, all of that keeps running on your machine exactly like before.&lt;/p&gt;

&lt;p&gt;The model call still goes out to whichever provider you've configured, Anthropic, OpenAI, whoever, the same as it always did; that part was never local to begin with, and this plugin doesn't change it.&lt;/p&gt;

&lt;p&gt;What changed was where each individual tool call landed:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;                Your Laptop
                     │
                     ▼
          OpenCode Harness
                     │
      ┌──────────────┴──────────────┐
      ▼                             ▼
Model Provider              Tensorlake Sandbox
(Claude/OpenAI/etc.)      Shell + Filesystem
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;blockquote&gt;
&lt;p&gt;The model call still goes wherever you configured your provider. Only the tool calls, the hands, move into the sandbox.&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;&lt;code&gt;webfetch&lt;/code&gt; and &lt;code&gt;websearch&lt;/code&gt; are the two exceptions that stay local, since neither touches a filesystem. &lt;code&gt;bash&lt;/code&gt;, &lt;code&gt;write&lt;/code&gt;, &lt;code&gt;edit&lt;/code&gt;, &lt;code&gt;read&lt;/code&gt;, &lt;code&gt;ls&lt;/code&gt;, &lt;code&gt;glob&lt;/code&gt;, and &lt;code&gt;grep&lt;/code&gt; are the ones that get rerouted.&lt;/p&gt;

&lt;p&gt;Every intercepted command now makes a network round trip instead of running instantly on my machine. Tensorlake's documentation says a sandbox starts up in a few seconds, with the underlying VM image itself booting in hundreds of milliseconds. Their GitHub page and product site separately claim resume from a suspended state also lands under a second. My own first sandbox, the one the plugin spun up automatically on that &lt;code&gt;uname -a&lt;/code&gt; call, took 2.3 seconds end to end, per the timestamp the plugin logged, which fits comfortably inside what the docs describe. That's likely the plugin's own provisioning and connection overhead stacked on top of the raw VM boot, not just the VM starting up. Either way, it's not something you sit around waiting on.&lt;/p&gt;




&lt;h2&gt;
  
  
  Setting It Up
&lt;/h2&gt;

&lt;p&gt;The whole setup turned out to be one config entry and one environment variable, typed in this order.&lt;/p&gt;

&lt;p&gt;First, the plugin, added to &lt;code&gt;~/.config/opencode/opencode.json&lt;/code&gt;:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight json"&gt;&lt;code&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;"$schema"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"https://opencode.ai/config.json"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;"plugin"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="s2"&gt;"tensorlake-opencode"&lt;/span&gt;&lt;span class="p"&gt;]&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;OpenCode installs it automatically, no separate &lt;code&gt;npm install&lt;/code&gt; needed. Then the API key, exported in the same shell I was about to launch OpenCode from:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="nb"&gt;export &lt;/span&gt;&lt;span class="nv"&gt;TENSORLAKE_API_KEY&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;your_api_key_here
opencode
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Nothing happened. No sandbox spun up on launch, which threw me for a second. First instinct was that I'd messed up the config somehow, and I almost went back to check the JSON for a typo before actually rereading what I'd just set up: lazy creation, not eager. Tailing &lt;code&gt;~/.local/share/opencode/log/tensorlake.log&lt;/code&gt; just confirmed the plugin had loaded, nothing more, until I actually asked the agent to do something that needed a sandbox.&lt;/p&gt;




&lt;h2&gt;
  
  
  The First Real Test
&lt;/h2&gt;

&lt;p&gt;Sandbox creation in this plugin is lazy. It waits for the first tool call that actually needs the filesystem or a shell, not the moment you launch the session.&lt;/p&gt;

&lt;p&gt;I didn't want to point it at anything that mattered yet, so the first command I actually gave it was about as low-stakes as they come:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Run: uname -a
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;That single &lt;code&gt;bash&lt;/code&gt; call was what triggered everything. A "Sandbox created" toast showed up in the terminal, and I had the log tailing in a second window the whole time. It read almost exactly what the docs describe: a line saying a new sandbox was being created for the session, then a second line confirming it was live, at 2.3 seconds.&lt;/p&gt;

&lt;p&gt;The output said Linux. My laptop runs macOS. Not going to lie, that one word convinced me faster than any architecture diagram would have.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Next I asked it to prove the filesystem side too:&lt;/strong&gt;&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Write "Hello Tensorlake" to /tmp/workspace/test.txt, then read it back.
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The write and read both happened entirely inside the sandbox, confirming that filesystem operations were isolated from my machine.&lt;/p&gt;

&lt;p&gt;&lt;code&gt;/tmp/workspace&lt;/code&gt; was the agent's working directory inside the sandbox, not a folder anywhere on my disk. Nothing in either test would have cost me anything if it had gone wrong.&lt;/p&gt;

&lt;p&gt;With the sandbox working, I wanted to move beyond smoke tests and verify the full developer workflow on a real repository.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Clone https://github.com/benjaminp/six.git, find the ensure_str function, 
improve its TypeError message so it names the function it came from, 
then run the test suite and tell me if anything broke.
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;This was intentionally a tiny change—the goal wasn't to contribute a feature, but to verify that OpenCode could complete the same edit–test loop I'd expect during normal development.&lt;/p&gt;

&lt;p&gt;&lt;code&gt;six&lt;/code&gt; is a small, well-known Python compatibility library: one source file, a straightforward test suite, and just enough structure to exercise a realistic edit–test workflow. &lt;/p&gt;

&lt;p&gt;OpenCode cloned the repository, located &lt;code&gt;ensure_str&lt;/code&gt;, modified the function, and ran the test suite entirely inside the sandbox.&lt;/p&gt;

&lt;p&gt;The original implementation raised &lt;code&gt;TypeError("not expecting type '%s'" % type(s))&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;The edit made it &lt;code&gt;TypeError("ensure_str: not expecting type '%s'" % type(s))&lt;/code&gt;—a deliberately small edit that was easy to verify with the existing test suite. Because the tests only verify that a &lt;code&gt;TypeError&lt;/code&gt; is raised rather than checking the exact message text, this made for a safe, minimal edit, and the run confirmed it: 184 passed, 16 skipped, 0 failed. &lt;/p&gt;

&lt;p&gt;Same as before the change.&lt;/p&gt;

&lt;p&gt;That's the part that mattered to me. Not the specific repo or the specific one-line fix, but that a clone, a real edit, and a full test run all happened inside the sandbox, on an actual project, without me once worrying about what would happen to my own machine if something in that chain went sideways.&lt;/p&gt;




&lt;h2&gt;
  
  
  Three Things That Clicked Once It Was Running
&lt;/h2&gt;

&lt;p&gt;I expected the isolation. The other two took me by surprise.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Isolation.&lt;/strong&gt; A bad command genuinely has nowhere real to land. A runaway install, an &lt;code&gt;rm&lt;/code&gt; aimed at the wrong path, a dependency that half-installs and leaves things broken. All of it happens in a sandbox I can throw away, not my actual working tree.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Reproducibility&lt;/strong&gt; mattered more once I pictured someone else on the team running this same setup. Whatever's actually installed on my laptop stops being relevant, because the agent isn't running on my laptop anymore. Register one image with the right toolchain baked in, point every future session at it, and the "works on my machine" conversation just stops happening.&lt;/p&gt;

&lt;p&gt;Then there's &lt;strong&gt;persistence&lt;/strong&gt;, the one I hadn't planned for. A named sandbox doesn't vanish when OpenCode restarts. It just sits there, parked. Come back later and the same working directory, the same installed packages, the same warm caches are all still exactly where I left them. Nothing rebuilt, nothing reinstalled.&lt;/p&gt;

&lt;p&gt;A disposable CI container vanishes the second a run finishes. This one was still there the next day, same as I'd left it.&lt;/p&gt;

&lt;p&gt;Took me an hour of lost installed state before that actually registered.&lt;/p&gt;




&lt;h2&gt;
  
  
  Configuring the Sandbox for Real Work
&lt;/h2&gt;

&lt;p&gt;Once I trusted it with something more than a test file, I went back and actually sized it properly. The plugin reads a small set of environment variables once, at the moment the first sandbox gets created, so they need to be set before you launch OpenCode, not after:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="nb"&gt;export &lt;/span&gt;&lt;span class="nv"&gt;TENSORLAKE_CPUS&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;4
&lt;span class="nb"&gt;export &lt;/span&gt;&lt;span class="nv"&gt;TENSORLAKE_MEMORY_MB&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;8192
&lt;span class="nb"&gt;export &lt;/span&gt;&lt;span class="nv"&gt;TENSORLAKE_DISK_MB&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;20480
&lt;span class="nb"&gt;export &lt;/span&gt;&lt;span class="nv"&gt;TENSORLAKE_IMAGE&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;my-custom-image
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;&lt;strong&gt;Variable&lt;/strong&gt;&lt;/th&gt;
&lt;th&gt;&lt;strong&gt;Default&lt;/strong&gt;&lt;/th&gt;
&lt;th&gt;&lt;strong&gt;Controls&lt;/strong&gt;&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;TENSORLAKE_CPUS&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;2&lt;/td&gt;
&lt;td&gt;vCPUs&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;TENSORLAKE_MEMORY_MB&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;4096&lt;/td&gt;
&lt;td&gt;RAM in MB&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;TENSORLAKE_DISK_MB&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;10240&lt;/td&gt;
&lt;td&gt;Disk in MB&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;TENSORLAKE_IMAGE&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;platform default&lt;/td&gt;
&lt;td&gt;The image the sandbox boots from&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;If you're using a Personal Access Token instead of a project-scoped key, you'll also need &lt;code&gt;TENSORLAKE_ORGANIZATION_ID&lt;/code&gt; and &lt;code&gt;TENSORLAKE_PROJECT_ID&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;&lt;code&gt;TENSORLAKE_IMAGE&lt;/code&gt; is the one actually worth using. Bake your language runtime, system packages, and project dependencies into a registered image once:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;tl sbx image create Dockerfile &lt;span class="nt"&gt;--registered-name&lt;/span&gt; my-custom-image
&lt;span class="nb"&gt;export &lt;/span&gt;&lt;span class="nv"&gt;TENSORLAKE_IMAGE&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;my-custom-image
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Every session after that starts already warm. I stopped watching the agent reinstall my stack from scratch every time I opened a new session.&lt;/p&gt;

&lt;p&gt;&lt;em&gt;Note: &lt;code&gt;export&lt;/code&gt; only lasts for the current shell, so I added these to my shell profile once I knew I'd be using this setup regularly.&lt;/em&gt;&lt;/p&gt;




&lt;h2&gt;
  
  
  The Operational Questions I Actually Had
&lt;/h2&gt;

&lt;p&gt;Before trusting this with anything real, I wanted answers to a few things that don't come up in a quick demo. Worth being upfront about one thing here: most of what follows is documented at the Tensorlake SDK and platform level. I haven't confirmed that the OpenCode plugin specifically exposes or manages each of these the same way, only that the underlying sandboxes it creates support them. Where that distinction matters, I've called it out below.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;What if I just walk away mid-session?&lt;/strong&gt; Named sandboxes auto-suspend after their idle timeout rather than terminating, at the platform level. Tensorlake's product site says the meter stops the moment it suspends, and their GitHub page puts resume at under a second, with filesystem, memory, and running processes exactly where you left them, not rebuilt. I haven't independently timed a resume myself, and I haven't confirmed whether the OpenCode plugin surfaces any control over this behavior or just inherits the platform default, so take the specific number as Tensorlake's claim about the platform, not a tested claim about this plugin.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;What if the agent kicks off something long-running, like a build or a test suite?&lt;/strong&gt; The underlying Tensorlake SDK supports starting background processes inside a sandbox that outlive the single command that launched them. I haven't tested whether the OpenCode plugin's &lt;code&gt;bash&lt;/code&gt; interception uses this pattern specifically, or whether a long-running command inside OpenCode just holds the tool call open for the duration. Either way, the sandbox itself isn't the bottleneck.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;What if my connection drops mid-command?&lt;/strong&gt; Tensorlake has written publicly about this exact failure mode at the platform level: when the transport hiccups mid-run, they retry and reap orphaned sandboxes so a flaky connection doesn't cost you the whole task. Again, this is a platform-level guarantee. I haven't tested how the plugin itself behaves if your local connection drops mid-tool-call.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;What if the sandbox itself crashes outright, not just an idle timeout?&lt;/strong&gt; Here's where I'll be straight with you: I didn't find documentation covering a hard crash mid-command specifically, at either the platform or plugin level, and I didn't manage to force one during testing either. If you're planning to run this against something you really can't afford to lose, that's worth confirming directly with Tensorlake rather than taking my word for it.&lt;/p&gt;




&lt;h2&gt;
  
  
  Where I'd Actually Use This
&lt;/h2&gt;

&lt;p&gt;By the end I had a rough rule for myself. Picture a coding agent refactoring a production monorepo while another engineer is pushing commits at the same time. The question was never whether the model was smart enough. It was whether I wanted its shell commands landing on my own workstation.&lt;/p&gt;

&lt;p&gt;This isn't the right setup for every OpenCode session, and I don't pretend otherwise:&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;&lt;strong&gt;My situation&lt;/strong&gt;&lt;/th&gt;
&lt;th&gt;&lt;strong&gt;What I'd do&lt;/strong&gt;&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;Personal throwaway project&lt;/td&gt;
&lt;td&gt;Run locally&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Client repository&lt;/td&gt;
&lt;td&gt;Use Tensorlake&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Production codebase&lt;/td&gt;
&lt;td&gt;Use Tensorlake&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Letting the agent experiment freely with shell commands&lt;/td&gt;
&lt;td&gt;Use Tensorlake&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Shared engineering environment&lt;/td&gt;
&lt;td&gt;Use Tensorlake&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;The sandbox earns its place anywhere the blast radius of a wrong command actually matters. If I'm the only person who'll ever touch the repo and I'd shrug off losing it, I'm solving a problem I don't have yet.&lt;/p&gt;

&lt;p&gt;Worth a quick word on why this isn't just Docker or a Codespace with extra steps. A Docker container shares your host's kernel, which is fine for packaging an app but a thinner isolation boundary than a full VM if you're worried about what an autonomous agent might run. &lt;/p&gt;

&lt;p&gt;A local VM gives you the stronger boundary but is heavy and slow to spin up and tear down for something you might want to throw away every few minutes. GitHub Codespaces and Dev Containers solve a different problem: a consistent, persistent dev environment tied to a specific repo, not a fast, disposable environment built to be created and discarded on every session. &lt;/p&gt;

&lt;p&gt;The MicroVM approach here is closer to a VM's isolation with something closer to a container's boot speed, and it's built specifically to be thrown away.&lt;/p&gt;




&lt;h2&gt;
  
  
  If You Want to Try This Yourself
&lt;/h2&gt;

&lt;p&gt;Here's the order I'd do it in, knowing what I know now.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Step 1:&lt;/strong&gt; Install OpenCode, add &lt;code&gt;tensorlake-opencode&lt;/code&gt; to &lt;code&gt;opencode.json&lt;/code&gt;. This just gets the plugin loaded, nothing else happens yet.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Step 2:&lt;/strong&gt; Export &lt;code&gt;TENSORLAKE_API_KEY&lt;/code&gt;, launch OpenCode, confirm the plugin loaded via the log file. This is the "is it actually on" check.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Step 3:&lt;/strong&gt; Ask it to run something trivial, like &lt;code&gt;uname -a&lt;/code&gt;. This is what triggers sandbox creation and gives you visible proof the command ran remotely.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Step 4:&lt;/strong&gt; Set &lt;code&gt;TENSORLAKE_CPUS&lt;/code&gt;, &lt;code&gt;TENSORLAKE_MEMORY_MB&lt;/code&gt;, and &lt;code&gt;TENSORLAKE_DISK_MB&lt;/code&gt; to match your real workload, before your next session starts, not mid-session.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Later:&lt;/strong&gt; Build a custom image with your actual toolchain baked in, once the default image starts feeling like it's missing things you keep reinstalling.&lt;/p&gt;




&lt;h2&gt;
  
  
  Closing
&lt;/h2&gt;

&lt;p&gt;I went looking for a third option: an agent with full autonomy, running somewhere that wasn't my laptop. Got that part working fast. What actually surprised me was how little I had to think about afterward.&lt;/p&gt;

&lt;p&gt;I stopped reading every command before approving it once this was running. Wasn't carelessness. Just nothing left on my machine for a bad command to reach.&lt;/p&gt;

&lt;p&gt;Turns out the part that mattered was never whether the agent could run commands. It was where they landed. Not stopping mistakes. Just making sure they happen somewhere disposable instead of somewhere that costs me.&lt;/p&gt;

&lt;p&gt;If you've started letting an autonomous coding agent anywhere near your shell, moving that shell execution into a disposable sandbox is one of the highest-leverage changes you can make, and Tensorlake's own numbers put the setup at a few minutes, not an afternoon. The config entry and the environment variable up above are the whole thing.&lt;/p&gt;




&lt;h2&gt;
  
  
  References
&lt;/h2&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;a href="https://docs.tensorlake.ai/sandboxes/opencode" rel="noopener noreferrer"&gt;Tensorlake OpenCode Integration&lt;/a&gt;: Full setup, configuration, and lazy sandbox creation model&lt;/li&gt;
&lt;li&gt;
&lt;a href="https://opencode.ai" rel="noopener noreferrer"&gt;OpenCode&lt;/a&gt;: The open source, terminal-first coding agent, MIT-licensed and model-agnostic&lt;/li&gt;
&lt;li&gt;
&lt;a href="https://www.npmjs.com/package/tensorlake-opencode" rel="noopener noreferrer"&gt;tensorlake-opencode on npm&lt;/a&gt;: The plugin package referenced in this article&lt;/li&gt;
&lt;li&gt;
&lt;a href="https://github.com/tensorlakeai/opencode-tensorlake-plugin" rel="noopener noreferrer"&gt;Plugin source on GitHub&lt;/a&gt;: Tool interceptors, session manager, lifecycle handling&lt;/li&gt;
&lt;li&gt;
&lt;a href="https://docs.tensorlake.ai/sandboxes/lifecycle" rel="noopener noreferrer"&gt;Tensorlake Sandbox Lifecycle&lt;/a&gt;: The suspend, resume, and snapshot model underneath this integration&lt;/li&gt;
&lt;li&gt;
&lt;a href="https://docs.tensorlake.ai/sandboxes/images" rel="noopener noreferrer"&gt;Tensorlake Sandbox Images&lt;/a&gt;: Building and registering a custom image&lt;/li&gt;
&lt;/ul&gt;




</description>
      <category>ai</category>
      <category>programming</category>
      <category>beginners</category>
      <category>python</category>
    </item>
    <item>
      <title>Master These 6 AI Concepts to Become an AI Engineer ( A Visual Explanation)</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Wed, 15 Jul 2026 09:27:36 +0000</pubDate>
      <link>https://dev.to/divy_ai/master-these-6-ai-concepts-to-become-an-ai-engineer-a-visual-explanation-3589</link>
      <guid>https://dev.to/divy_ai/master-these-6-ai-concepts-to-become-an-ai-engineer-a-visual-explanation-3589</guid>
      <description>&lt;p&gt;Job postings asking for AI skills are up 143% in a single year. The people filling those roles didn't spend years on advanced math. They learned six ideas, in the right order, and started building.&lt;/p&gt;

&lt;p&gt;Most people think you need a math degree, or years of computer science training, to become a serious AI developer. &lt;/p&gt;

&lt;p&gt;&lt;strong&gt;You don't.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The gap between someone who can barely get a chatbot to behave and someone building real, production AI systems isn't intelligence. It isn't years of study. It's six specific concepts, learned in an order that actually makes sense, instead of scattered across a hundred confusing tutorials.&lt;/p&gt;

&lt;p&gt;Here they are, explained assuming you know nothing yet.&lt;/p&gt;




&lt;h3&gt;
  
  
  A quick honest note before we start
&lt;/h3&gt;

&lt;p&gt;This roadmap covers how to build AI-powered applications, chatbots, assistants, and tools that use existing AI models well. That's different from becoming a machine learning researcher who builds new models from scratch, which genuinely does need heavy math and years of study.&lt;/p&gt;

&lt;p&gt;Most people who say "I want to become an AI developer" mean the first path. That's also where almost all the current job demand is. AI-related job postings are up 143% year over year, and the engineers filling those roles are, overwhelmingly, application builders, not research scientists.&lt;/p&gt;

&lt;p&gt;This is that path.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fxu94tkepdx20ceasyuiq.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fxu94tkepdx20ceasyuiq.png" alt="Photo from AI" width="800" height="938"&gt;&lt;/a&gt;&lt;/p&gt;




&lt;h2&gt;
  
  
  Level 1: Beginner
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fecfswhyrpl6d69bl3ljn.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fecfswhyrpl6d69bl3ljn.png" alt="Photo from AI" width="800" height="533"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  1. The Context Window
&lt;/h3&gt;

&lt;p&gt;Here's the concept that trips up almost every beginner, because it isn't obvious until something breaks.&lt;/p&gt;

&lt;p&gt;An AI model can only "see" a limited amount of text at once. That limit is called the &lt;strong&gt;context window&lt;/strong&gt;. Picture a whiteboard in a meeting room. You can write a lot on it, but once it's full, anything written past the edge simply isn't there anymore. It's not that the AI forgot. It never saw it.&lt;/p&gt;

&lt;p&gt;This matters the moment you try to have a long conversation, or feed the AI a huge document. Past a certain point, older parts of the conversation quietly fall off the edge of the whiteboard, and the AI starts responding as if they never happened.&lt;/p&gt;

&lt;p&gt;Understanding this one limit explains most of the "why did the AI suddenly get confused" moments beginners run into.&lt;/p&gt;




&lt;h2&gt;
  
  
  Level 2: Intermediate
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F37d32ib6ob0yv6qh5d5t.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F37d32ib6ob0yv6qh5d5t.png" alt="Photo from AI" width="800" height="1200"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  2. RAG (Retrieval-Augmented Generation)
&lt;/h3&gt;

&lt;p&gt;An AI model only knows what it learned during training. Ask it about your company's internal policies, and it has nothing, because it never read them. &lt;strong&gt;RAG fixes this.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Think of it as the difference between a closed-book exam and an open-book one. Without RAG, the AI answers purely from memory. With RAG, it's allowed to flip open a specific book—your documents, your database, your knowledge base—and check before answering.&lt;/p&gt;

&lt;p&gt;In practice, this happens in three steps:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;
&lt;strong&gt;Chunking:&lt;/strong&gt; Your documents get broken into smaller pieces and stored in a searchable format.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Retrieval:&lt;/strong&gt; The AI searches that storage for the pieces most relevant to the question.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Augmentation and Generation:&lt;/strong&gt; It answers using both what it already knew and what it just found.&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;Virtually every AI chatbot that answers questions about a specific company's internal information is running RAG under the hood. It is arguably the single most valuable skill on this entire list, because it is the difference between a generic AI and one that actually knows your business.&lt;/p&gt;

&lt;h3&gt;
  
  
  3. Fine-Tuning
&lt;/h3&gt;

&lt;p&gt;Prompting hands the AI a fresh set of instructions every single time. Fine-tuning is different: it actually reshapes how the model behaves, permanently, by training it further on a narrow, specific set of examples.&lt;/p&gt;

&lt;p&gt;Think of the difference between handing someone a manual before every task versus sending them through months of specialized training. The manual works for most jobs. Specialized training changes how someone thinks about the job itself.&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;&lt;strong&gt;Pro Tip:&lt;/strong&gt; Most developers should reach for RAG first. Fine-tuning costs more, takes longer, and is usually only worth it when you need a very specific style, format, or behavior that no amount of careful instruction alone can reliably produce.&lt;/p&gt;
&lt;/blockquote&gt;

&lt;h3&gt;
  
  
  4. AI Agents
&lt;/h3&gt;

&lt;p&gt;Everything so far has been about the AI answering questions. &lt;strong&gt;An agent is about the AI actually doing things.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Picture the difference between a consultant who gives you advice and an assistant who actually goes and books the flight. A consultant-style AI just answers. An agent actually takes real action instead of just describing what action you should take: searching the web, running code, sending an email, or updating a database.&lt;/p&gt;

&lt;p&gt;The AI does this by deciding, on its own, which step to take next based on what it just learned. Ask it to fix a bug, and it can read the error, try a fix, check if that fix worked, and try something else if it didn't—all without you typing a new instruction after every single step.&lt;/p&gt;

&lt;p&gt;This is where AI development starts feeling genuinely powerful, and also where it starts requiring real care, which is exactly why the next two concepts exist.&lt;/p&gt;




&lt;h2&gt;
  
  
  Level 3: Advanced
&lt;/h2&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fukdy9crf9liikg83fj71.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fukdy9crf9liikg83fj71.png" alt="Photo from AI" width="800" height="533"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  5. MCP (Model Context Protocol)
&lt;/h3&gt;

&lt;p&gt;Imagine every country having a different shape of electrical socket, and every single appliance needing its own custom adapter just to work when you travel. That was the old way AI agents connected to outside tools: every connection was custom-built, one at a time, for every tool and every AI model.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;MCP works like a universal socket standard.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Once a tool speaks MCP, any AI agent that understands MCP can plug into it directly. No custom adapter needed. Anthropic introduced this standard, and it has quickly become the common way agents connect to databases, files, and other services, regardless of which AI model is actually doing the work. If you're building an agent that needs to reach outside tools, MCP is very likely how that connection gets made.&lt;/p&gt;

&lt;h3&gt;
  
  
  6. Harness Engineering
&lt;/h3&gt;

&lt;p&gt;This is one of the most critical operational concepts for production engineering. Picture a stunt performer on a film set. They're genuinely skilled. But nobody lets them attempt a dangerous stunt without a safety harness, a spotter, a hard limit on how many takes they get, and a director watching every single shot.&lt;/p&gt;

&lt;p&gt;The performer's skill isn't what keeps them safe on set. The equipment and process wrapped around them is.&lt;/p&gt;

&lt;p&gt;An AI agent works the same way. The model is the skilled performer. &lt;strong&gt;The harness is everything wrapped around it:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;A hard limit on how much it's allowed to spend before it has to stop.&lt;/li&gt;
&lt;li&gt;Checkpoints that save progress so a crash doesn't waste hours of work.&lt;/li&gt;
&lt;li&gt;Guardrails on which actions it's actually allowed to execute.&lt;/li&gt;
&lt;li&gt;Logs so a person can audit exactly what happened if something goes wrong.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;This concept exists because of one striking number: &lt;strong&gt;88% of AI agent projects never make it into real production use.&lt;/strong&gt; Not because the models were bad, but because nobody built the harness around them.&lt;/p&gt;




&lt;h3&gt;
  
  
  How these six AI concepts fit together
&lt;/h3&gt;

&lt;p&gt;None of these exist alone in a real system. A production AI assistant, the kind companies actually pay for, typically uses several of these at once: &lt;strong&gt;RAG&lt;/strong&gt; to pull in company-specific knowledge, an &lt;strong&gt;agent&lt;/strong&gt; to actually take action, &lt;strong&gt;MCP&lt;/strong&gt; so that agent can reach real tools, and a &lt;strong&gt;harness&lt;/strong&gt; watching the whole thing to make sure nothing goes wrong.&lt;/p&gt;

&lt;p&gt;Learning them one at a time makes sense. Using only one at a time, in a real product, almost never does.&lt;/p&gt;




&lt;h3&gt;
  
  
  Key Takeaways
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;The context window&lt;/strong&gt; is the AI's limited working memory. Understanding its limit explains most confusing AI mistakes.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;RAG&lt;/strong&gt; lets an AI answer from your own documents instead of just what it learned in training. It is the most valuable skill on this list.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Fine-tuning&lt;/strong&gt; permanently reshapes a model's behavior. Reach for it only when instructions genuinely can't get you there.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;An AI Agent&lt;/strong&gt; takes real action instead of just answering, deciding its own next step based on what it just learned.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;MCP&lt;/strong&gt; is the universal standard that lets agents plug into outside tools without a custom connection built for every single one.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Harness engineering&lt;/strong&gt; is what turns a smart model into a reliable system. The missing harness is usually why agent projects fail to reach production.&lt;/li&gt;
&lt;/ul&gt;




&lt;h3&gt;
  
  
  The part worth remembering
&lt;/h3&gt;

&lt;p&gt;Nobody starts as an advanced AI developer. Every person building serious production AI systems today started exactly where you might be starting now—watching a conversation quietly lose track of what was said three messages ago and wondering why.&lt;/p&gt;

&lt;p&gt;What separates a beginner from someone advanced isn't talent. It's this list, learned in order, applied to something real. The math and the deep model architecture that everyone assumes is required? Most working AI developers never touch it. They learned six ideas, built things with them, and kept going.&lt;/p&gt;

&lt;p&gt;You can start with the context window today. Nothing on this list requires anything you don't already have.&lt;/p&gt;




&lt;h3&gt;
  
  
  References
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;&lt;a href="https://hackernoon.com/5-must-have-ai-skills-for-developers-in-2026-and-how-to-learn-them" rel="noopener noreferrer"&gt;5 Must-Have AI Skills for Developers (HackerNoon)&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://doit.software/blog/ai-developer-skills" rel="noopener noreferrer"&gt;Top 13 AI Developer Skills (DOIT Software)&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://uploadarticle.com/ai-developer-roadmap-2026/" rel="noopener noreferrer"&gt;From Beginner to AI Developer: A Practical Roadmap&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href="https://medium.com/@adnanmasood/agent-harness-engineering-the-rise-of-the-ai-control-plane-938ead884b1d" rel="noopener noreferrer"&gt;Agent Harness Engineering: The Rise of the AI Control Plane (Medium)&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;

</description>
      <category>ai</category>
      <category>webdev</category>
      <category>programming</category>
      <category>beginners</category>
    </item>
    <item>
      <title>[Boost]</title>
      <dc:creator>Divy Yadav</dc:creator>
      <pubDate>Fri, 10 Jul 2026 12:39:27 +0000</pubDate>
      <link>https://dev.to/divy_ai/-2g20</link>
      <guid>https://dev.to/divy_ai/-2g20</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/divy_ai/why-your-ai-experiments-keep-starting-from-scratch-and-how-tensorlake-fixes-it-4gbo" class="crayons-story__hidden-navigation-link"&gt;Why Your AI Experiments Keep Starting From Scratch (And How Tensorlake Fixes It)&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/divy_ai" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F3845890%2Fc910b35f-87b4-4888-b116-ad2cde414360.png" alt="divy_ai profile" class="crayons-avatar__image" width="800" height="800"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/divy_ai" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Divy Yadav
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Divy Yadav
                
              
              &lt;div id="story-author-preview-content-4111884" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/divy_ai" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F3845890%2Fc910b35f-87b4-4888-b116-ad2cde414360.png" class="crayons-avatar__image" alt="" width="800" height="800"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Divy Yadav&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/divy_ai/why-your-ai-experiments-keep-starting-from-scratch-and-how-tensorlake-fixes-it-4gbo" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Jul 10&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/divy_ai/why-your-ai-experiments-keep-starting-from-scratch-and-how-tensorlake-fixes-it-4gbo" id="article-link-4111884"&gt;
          Why Your AI Experiments Keep Starting From Scratch (And How Tensorlake Fixes It)
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/webdev"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;webdev&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/programming"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;programming&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/productivity"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;productivity&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/divy_ai/why-your-ai-experiments-keep-starting-from-scratch-and-how-tensorlake-fixes-it-4gbo" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="24" height="24"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;1&lt;span class="hidden s:inline"&gt;&amp;nbsp;reaction&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/divy_ai/why-your-ai-experiments-keep-starting-from-scratch-and-how-tensorlake-fixes-it-4gbo#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            12 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
  </channel>
</rss>
