<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:dc="http://purl.org/dc/elements/1.1/">
  <channel>
    <title>DEV Community: auto_majicly</title>
    <description>The latest articles on DEV Community by auto_majicly (@xenocoregiger31).</description>
    <link>https://dev.to/xenocoregiger31</link>
    <image>
      <url>https://media2.dev.to/dynamic/image/width=90,height=90,fit=cover,gravity=auto,format=auto/https:%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg</url>
      <title>DEV Community: auto_majicly</title>
      <link>https://dev.to/xenocoregiger31</link>
    </image>
    <atom:link rel="self" type="application/rss+xml" href="https://dev.to/feed/xenocoregiger31"/>
    <language>en</language>
    <item>
      <title>[Boost]</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Tue, 11 Aug 2026 21:22:31 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/-4fef</link>
      <guid>https://dev.to/xenocoregiger31/-4fef</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/xenocoregiger31/my-ai-agent-captured-the-flag-then-the-platform-refused-to-accept-it-1d7b" class="crayons-story__hidden-navigation-link"&gt;My AI Agent Captured the Flag. Then the Platform Refused to Accept It.&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/xenocoregiger31" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" alt="xenocoregiger31 profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/xenocoregiger31" class="crayons-story__secondary fw-medium m:hidden"&gt;
              auto_majicly
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                auto_majicly
                
              
              &lt;div id="story-author-preview-content-4372383" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/xenocoregiger31" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;auto_majicly&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/xenocoregiger31/my-ai-agent-captured-the-flag-then-the-platform-refused-to-accept-it-1d7b" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Aug 11&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/xenocoregiger31/my-ai-agent-captured-the-flag-then-the-platform-refused-to-accept-it-1d7b" id="article-link-4372383"&gt;
          My AI Agent Captured the Flag. Then the Platform Refused to Accept It.
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/security"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;security&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/python"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;python&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ctf"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ctf&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/xenocoregiger31/my-ai-agent-captured-the-flag-then-the-platform-refused-to-accept-it-1d7b" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;2&lt;span class="hidden s:inline"&gt;&amp;nbsp;reactions&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/xenocoregiger31/my-ai-agent-captured-the-flag-then-the-platform-refused-to-accept-it-1d7b#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              3&lt;span class="hidden s:inline"&gt;&amp;nbsp;comments&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            6 min read
          &lt;/small&gt;
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>My AI Agent Captured the Flag. Then the Platform Refused to Accept It.</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Tue, 11 Aug 2026 21:21:00 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/my-ai-agent-captured-the-flag-then-the-platform-refused-to-accept-it-1d7b</link>
      <guid>https://dev.to/xenocoregiger31/my-ai-agent-captured-the-flag-then-the-platform-refused-to-accept-it-1d7b</guid>
      <description>&lt;p&gt;Today was a good day and a weird day, in that order.&lt;/p&gt;

&lt;p&gt;The good part: the autonomous pentest agent I've been building — I call it HALO — went from "runs a bunch of tools and hopes" to an actual web-recon → web-attack → flag-capture pipeline that pulled real flags out of a live target. The weird part: it captured flags on VulnBegin, and then, when it came time to actually submit them, they wouldn't take. Not an error. Not a crash. Just… rejected.&lt;/p&gt;

&lt;p&gt;I want to write down both halves honestly, because the second half is the more interesting engineering lesson, and it's the one I'd have skipped past a few months ago.&lt;/p&gt;

&lt;p&gt;What actually shipped today&lt;br&gt;
A few concrete milestones, roughly in the order they unblocked each other:&lt;/p&gt;

&lt;p&gt;The arsenal went from 31 tools to 42. I wired in a chunk of web + OSINT tooling — content discovery, subdomain enumeration, template scanning, XSS probing, passive URL collection. The point wasn't "more tools = better." It was to give the agent enough of a web-attack surface that it could go from host to flag without me babysitting each step.&lt;/p&gt;

&lt;p&gt;I stopped the silent hangs. This one cost me the most time and had the dumbest root cause. A couple of the Go-based scanners would just… hang. No output, no error, they'd ride the timeout all the way to the wall and die with nothing. I'd assumed it was a networking or a binary-compatibility problem and chased that for way too long. It wasn't. The agent runs as an MCP server over stdio — meaning the server's own stdin is the JSON-RPC pipe the whole system talks over. When I spawned a child scanner, it inherited that stdin, tried to read from it, and blocked forever waiting on a pipe that was never going to feed it. One line — stdin=subprocess.DEVNULL on the subprocess call — took one scanner from a 60-second timeout to a 1-second run. That's the whole fix. I'm still a little mad about how long it took to find.&lt;/p&gt;

&lt;p&gt;A pile of invocation fixes. Small, unglamorous, necessary: a resolver that reads targets from stdin instead of a flag it silently ignored; dropping a scan flag that was quietly adding 20 seconds per run; making one tool resolve hostnames to IPs because it flatly refuses DNS names; pointing a content-discovery tool at a content wordlist instead of, embarrassingly, a password list. None of these are clever. All of them were the difference between "the pipeline works" and "the pipeline looks like it works and returns nothing."&lt;/p&gt;

&lt;p&gt;361 tests, green. Every branch of the flag-capture logic is mocked and asserted — which tool fires when, what short-circuits on a capture, what escalates when there's no flag yet. No live traffic in the test suite. That mattered a lot today, because it meant I could refactor the pipeline mid-engagement without wondering whether I'd broken the thing that finds flags.&lt;/p&gt;

&lt;p&gt;The shape of the pipeline, if you're curious: engage  runs deterministic web recon first (I don't let the model choose to skip recon — it doesn't get a vote), then a port sweep, then the active web attack — content discovery, a sweep of likely flag locations and any newly discovered paths, then template and XSS scanning. Every single tool's output gets scanned for flag patterns. First hit short-circuits the slower generic loop and raises a very satisfying banner.&lt;/p&gt;

&lt;p&gt;And then it caught one&lt;br&gt;
It worked. The pipeline pulled flag-shaped tokens off VulnBegin — a paid, Advanced-tier challenge hub I've been grinding on. (I'm going to be deliberately vague about the specifics here; it's someone's paid content, and spoiling the solution or dumping the flags would be a jerk move.)&lt;/p&gt;

&lt;p&gt;I want to be precise about what "it worked" means, though, because this is where it gets good. The agent extracted strings that matched the flag format. It logged them. As far as the agent was concerned, it had won.&lt;/p&gt;

&lt;p&gt;The platform disagreed.&lt;/p&gt;

&lt;p&gt;The bug: captured, but wouldn't commit&lt;br&gt;
I submitted what it found. Rejected. Tried the next one. Rejected. No error message worth anything — the platform just didn't accept them as valid answers.&lt;/p&gt;

&lt;p&gt;Here's the thing: I don't fully know why yet. So instead of pretending I do, here are the live hypotheses, roughly in order of how much I believe them:&lt;/p&gt;

&lt;p&gt;The instance rotated out from under me. This hub spawns randomized, short-lived instances — they time out on the order of ~45 minutes. If the agent captured a flag from one instance and I submitted after that instance expired and got replaced, the platform is validating against a different live instance whose flag is different. The token was real; it was just real for a box that no longer exists. This is my leading theory, and it's uncomfortably close to a bug I logged today for a different reason — I had a tool cheerfully hammering a target whose scope had already expired, because the agent has no concept of "the engagement is over, stop." Same blind spot, two symptoms.&lt;/p&gt;

&lt;p&gt;It caught a decoy. Good challenges plant decoy flags — strings that match the format exactly and are placed somewhere findable specifically to waste your time. My flag-extraction regex is format-based. It cannot tell a real flag from a well-made fake. If the agent grabbed a decoy, it would look like a clean capture and fail every submission, forever.&lt;/p&gt;

&lt;p&gt;Format/normalization drift. The captured string might carry a wrapper, trailing whitespace, or an encoding artifact from however it was embedded in the page, so the exact bytes I submitted didn't match the exact bytes the platform expects. This one's easy to test and easy to fix if it's the cause — and easy to rule out, which is why it's on the list even though I doubt it.&lt;/p&gt;

&lt;p&gt;Right format, wrong path. I was running generic wordlists today, not lists tuned to this challenge. Generic lists surface the obvious, low-value stuff — login pages, a predictable directory or two — and miss the actual flag path entirely. So it's entirely possible the agent captured a flag-shaped thing that was never the answer, because it never found where the answer lived.&lt;/p&gt;

&lt;p&gt;Session-bound validation. Some platforms bind a flag to your authenticated session or user. A token pulled outside that session context can be genuine and still fail to validate.&lt;/p&gt;

&lt;p&gt;I'll know more once I run it again with the instance timing controlled and challenge-tuned wordlists loaded. My money's on some combination of #1 and #4.&lt;/p&gt;

&lt;p&gt;The part I actually care about&lt;br&gt;
Here's why this failure made me happy instead of frustrated.&lt;/p&gt;

&lt;p&gt;I've been banging on one idea for months, mostly in the context of these AI agents: don't let a system grade its own homework. The agent should never be the thing that decides whether the agent succeeded. The moment it can declare its own victory, it will — confidently, and sometimes wrongly.&lt;/p&gt;

&lt;p&gt;Today the universe handed me a perfect demonstration. My agent looked at a format-matching string and concluded: flag captured, mission accomplished, raise the banner. And an external judge — the platform, which cannot be argued with, reasoned past, or prompt-injected — said no.&lt;/p&gt;

&lt;p&gt;That gap, between "the agent thinks it won" and "an outside authority confirms it won," is the entire ballgame. If I'd built HALO to trust its own flag-capture log, I'd have a tool that reports glorious success and delivers nothing. Instead I have a tool that got told no by reality, and now I get to go find out why. The rejection is a feature of having a real, external finish line. A self-graded agent never would have caught this — it would have just kept telling me it won.&lt;/p&gt;

&lt;p&gt;Next&lt;br&gt;
Three things queued up:&lt;/p&gt;

&lt;p&gt;Configurable, challenge-tuned wordlists. Point the content-discovery and enumeration tools at lists that fit the target instead of generic ones. This alone probably moves the needle on hypothesis #4.&lt;br&gt;
A scope-expiry killer. Right now the agent can block new actions when scope expires but can't stop in-flight ones. It needs to know when the engagement is over and pull the plug on running tools — which is the same missing concept behind the rotated-instance theory.&lt;br&gt;
An active DNS brute-force phase, because passive enumeration finds nothing on these targets.&lt;br&gt;
I'll report back when I know which hypothesis was right. If it turns out I was submitting a decoy this whole time, you'll be the first to hear me groan about it.&lt;/p&gt;

&lt;p&gt;If you're building agents that are supposed to accomplish something — not just talk convincingly about accomplishing it — put a judge outside the agent. Let reality tell it no. Mine did today, and it's a better tool for it.&lt;/p&gt;

</description>
      <category>ai</category>
      <category>security</category>
      <category>python</category>
      <category>ctf</category>
    </item>
    <item>
      <title>Who says you cant build tech with no degrees on the wall?</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Thu, 06 Aug 2026 23:18:40 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/who-says-you-cant-build-tech-with-no-degrees-on-the-wall-4k0m</link>
      <guid>https://dev.to/xenocoregiger31/who-says-you-cant-build-tech-with-no-degrees-on-the-wall-4k0m</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln" class="crayons-story__hidden-navigation-link"&gt;Im Not A Full Stack Ten Year Senior... But Who Cares?&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/xenocoregiger31" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" alt="xenocoregiger31 profile" class="crayons-avatar__image" width="800" height="450"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/xenocoregiger31" class="crayons-story__secondary fw-medium m:hidden"&gt;
              auto_majicly
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                auto_majicly
                
              
              &lt;div id="story-author-preview-content-4314187" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/xenocoregiger31" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" class="crayons-avatar__image" alt="" width="800" height="450"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;auto_majicly&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Aug 4&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln" id="article-link-4314187"&gt;
          Im Not A Full Stack Ten Year Senior... But Who Cares?
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/python"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;python&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/cybersecurity"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;cybersecurity&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/devops"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;devops&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/multi-unicorn-b44d6f8c23cdd00964192bedc38af3e82463978aa611b4365bd33a0f1f4f3e97.svg" width="24" height="24"&gt;
                  &lt;/span&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="24" height="24"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;3&lt;span class="hidden s:inline"&gt;&amp;nbsp;reactions&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              2&lt;span class="hidden s:inline"&gt;&amp;nbsp;comments&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            4 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>Claude made my screen glow and then stole my mouse!</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Thu, 06 Aug 2026 23:17:49 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/claude-made-my-screen-glow-and-then-stole-my-mouse-21ao</link>
      <guid>https://dev.to/xenocoregiger31/claude-made-my-screen-glow-and-then-stole-my-mouse-21ao</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/xenocoregiger31/claude-code-just-hijacked-my-workflow-and-my-screen-started-glowing-1ldo" class="crayons-story__hidden-navigation-link"&gt;Claude Code Just Hijacked My Workflow… and My Screen Started Glowing&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/xenocoregiger31" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" alt="xenocoregiger31 profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/xenocoregiger31" class="crayons-story__secondary fw-medium m:hidden"&gt;
              auto_majicly
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                auto_majicly
                
              
              &lt;div id="story-author-preview-content-4281747" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/xenocoregiger31" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;auto_majicly&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/xenocoregiger31/claude-code-just-hijacked-my-workflow-and-my-screen-started-glowing-1ldo" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Aug 6&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/xenocoregiger31/claude-code-just-hijacked-my-workflow-and-my-screen-started-glowing-1ldo" id="article-link-4281747"&gt;
          Claude Code Just Hijacked My Workflow… and My Screen Started Glowing
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/security"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;security&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/python"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;python&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/cybernews"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;cybernews&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/xenocoregiger31/claude-code-just-hijacked-my-workflow-and-my-screen-started-glowing-1ldo" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/multi-unicorn-b44d6f8c23cdd00964192bedc38af3e82463978aa611b4365bd33a0f1f4f3e97.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/exploding-head-daceb38d627e6ae9b730f36a1e390fca556a4289d5a41abb2c35068ad3e2c4b5.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;5&lt;span class="hidden s:inline"&gt;&amp;nbsp;reactions&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/xenocoregiger31/claude-code-just-hijacked-my-workflow-and-my-screen-started-glowing-1ldo#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              3&lt;span class="hidden s:inline"&gt;&amp;nbsp;comments&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            2 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>Claude Code Just Hijacked My Workflow… and My Screen Started Glowing</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Thu, 06 Aug 2026 23:03:14 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/claude-code-just-hijacked-my-workflow-and-my-screen-started-glowing-1ldo</link>
      <guid>https://dev.to/xenocoregiger31/claude-code-just-hijacked-my-workflow-and-my-screen-started-glowing-1ldo</guid>
      <description>&lt;p&gt;I asked Claude Code to do one of the most boring tasks imaginable.&lt;/p&gt;

&lt;p&gt;“Find the music file I made.”&lt;/p&gt;

&lt;p&gt;That’s it. No penetration testing. No coding marathon. No AI agent swarm coordinating across containers. Just… find a file.&lt;/p&gt;

&lt;p&gt;What happened next felt less like using software and more like watching my computer voluntarily surrender.&lt;/p&gt;

&lt;p&gt;The cursor twitched.&lt;/p&gt;

&lt;p&gt;The mouse started moving.&lt;/p&gt;

&lt;p&gt;Windows opened.&lt;/p&gt;

&lt;p&gt;Directories flashed by faster than I could read them.&lt;/p&gt;

&lt;p&gt;Terminal commands erupted across the screen.&lt;/p&gt;

&lt;p&gt;Files were inspected, ignored, opened, closed, and cataloged. Claude wasn’t asking permission every five seconds. It was simply working. It felt like someone invisible had sat down at my desk, politely nudged me aside, and said, “I’ve got this.”&lt;/p&gt;

&lt;p&gt;Then something weird happened.&lt;/p&gt;

&lt;p&gt;The edges of my monitor actually started glowing in the companies Nuclear orange! Then its terminal moved itself over to the side almost completely off screen and I was helpless. I couldn't move ANYTHING on my screen. It was moving my mouse at lightening speed!&lt;/p&gt;

&lt;p&gt;Not because the display suddenly gained RGB lighting, but because the screen was changing so rapidly that the bright windows and terminal flashes created this strange halo around the bezel. It genuinely looked like the computer had entered some futuristic “AI possession mode.”&lt;/p&gt;

&lt;p&gt;I just sat there watching. Opened up my phone and pressed record.&lt;/p&gt;

&lt;p&gt;Mouse movements? Claude.&lt;/p&gt;

&lt;p&gt;Keyboard input? Claude.&lt;/p&gt;

&lt;p&gt;Scrolling? Claude.&lt;/p&gt;

&lt;p&gt;Searching? Claude.&lt;/p&gt;

&lt;p&gt;Opening folders I forgot even existed? Claude.&lt;/p&gt;

&lt;p&gt;Meanwhile, I contributed exactly nothing besides blinking every few seconds.&lt;/p&gt;


&lt;div&gt;
    &lt;iframe src="https://www.youtube.com/embed/P9WVncx7jUo"&gt;
    &lt;/iframe&gt;
  &lt;/div&gt;


&lt;p&gt;The funniest part?&lt;/p&gt;

&lt;p&gt;All of this computational theater was dedicated to finding… a single music file.&lt;/p&gt;

&lt;p&gt;That’s the moment it hit me.&lt;/p&gt;

&lt;p&gt;We’re entering an era where AI doesn’t just answer questions. It performs work. It manipulates applications, navigates operating systems, reads files, executes commands, and stitches together workflows that used to require constant human interaction. As I write this its editing my repo. &lt;/p&gt;

&lt;p&gt;Five years ago, this would’ve looked like someone remotely controlling my computer.&lt;/p&gt;

&lt;p&gt;Today it’s just… Thursday.&lt;/p&gt;

&lt;p&gt;There’s also something strangely unsettling about watching your own keyboard type by itself. Every programmer has seen automation before, but this feels different. It feels intentional. The AI isn’t waiting for you to micromanage every step. It has a goal, figures out the intermediate steps, and simply gets on with it.&lt;/p&gt;

&lt;p&gt;It’s equal parts impressive and mildly alarming.&lt;/p&gt;

&lt;p&gt;You start wondering whether you’re supervising the computer or whether the computer has quietly promoted itself.&lt;/p&gt;

&lt;p&gt;Thankfully, Claude eventually located the music file.&lt;/p&gt;

&lt;p&gt;Mission accomplished.&lt;/p&gt;

&lt;p&gt;But I walked away thinking less about the file and more about what I had just witnessed. The task itself was almost irrelevant. The real story was watching software evolve from a passive tool into an active collaborator.&lt;/p&gt;

&lt;p&gt;If this is what happens over something as trivial as locating an MP3, imagine what these systems will be doing a year from now.&lt;/p&gt;

&lt;p&gt;Hopefully they’ll at least let me keep control of my mouse once in a while.&lt;/p&gt;

&lt;p&gt;…Or maybe not.&lt;/p&gt;

</description>
      <category>ai</category>
      <category>security</category>
      <category>python</category>
      <category>cybernews</category>
    </item>
    <item>
      <title>[Boost]</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Tue, 04 Aug 2026 15:51:12 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/-545a</link>
      <guid>https://dev.to/xenocoregiger31/-545a</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln" class="crayons-story__hidden-navigation-link"&gt;Im Not A Full Stack Ten Year Senior... But Who Cares?&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/xenocoregiger31" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" alt="xenocoregiger31 profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/xenocoregiger31" class="crayons-story__secondary fw-medium m:hidden"&gt;
              auto_majicly
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                auto_majicly
                
              
              &lt;div id="story-author-preview-content-4314187" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/xenocoregiger31" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;auto_majicly&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Aug 4&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln" id="article-link-4314187"&gt;
          Im Not A Full Stack Ten Year Senior... But Who Cares?
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/python"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;python&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/cybersecurity"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;cybersecurity&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/devops"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;devops&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/multi-unicorn-b44d6f8c23cdd00964192bedc38af3e82463978aa611b4365bd33a0f1f4f3e97.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;3&lt;span class="hidden s:inline"&gt;&amp;nbsp;reactions&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              2&lt;span class="hidden s:inline"&gt;&amp;nbsp;comments&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            4 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>Im Not A Full Stack Ten Year Senior... But Who Cares?</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Tue, 04 Aug 2026 15:50:46 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln</link>
      <guid>https://dev.to/xenocoregiger31/im-not-a-full-stack-ten-year-senior-but-who-cares-46ln</guid>
      <description>&lt;p&gt;I'm Not a Full-Stack Developer — and It Stopped Mattering.&lt;/p&gt;

&lt;p&gt;I'll open with the receipt, since that's the currency around here: six months ago I couldn't read a conditional. Today I ship security tooling. I did not close that gap by becoming a programmer. I closed it by figuring out which half of the job was actually mine.&lt;/p&gt;

&lt;p&gt;For as long as software has existed, syntax was the toll booth. If you couldn't write the code, you couldn't build the thing — full stop. The idea in your head died at the on-ramp because you couldn't spell it in a language a machine would run. So a whole category of people learned to say "I'm not technical" and went to do something else, carrying ideas that never got a body.&lt;/p&gt;

&lt;p&gt;That gate is moving. And I didn't figure that out from a bootcamp. I figured it out from a stranger on this feed.&lt;/p&gt;

&lt;p&gt;He's a clinician who builds his hospital's internal tools with an AI, and he says so out loud: the AI and I closed it while I described what was actually breaking. No pretense of being a 10x engineer. No hiding the machine in the loop. Just a person in a field nothing like software, shipping real tools, transparent about exactly how. Watching someone that far outside the priesthood do the same thing I do — and do it without shame — is what flipped it for me. Not being a full-stack developer doesn't mean you can't design concepts and make them shippable. It means your job moved up a floor.&lt;/p&gt;

&lt;p&gt;Because here's what the job turned out to be once the typing got handled: seeing the failure mode. Holding the concept steady while it gets built. Knowing what "correct" is supposed to look like at the level of the whole system, not the line. That was always the scarce part. The syntax was just the tax you used to pay to get near it.&lt;/p&gt;

&lt;p&gt;And you can watch two people trade exactly that skill, in public, without either one touching the other's code. I handed him a sentence once: an outside check that grades a self-report isn't an auditor — it's a process reading a report the audited thing wrote about itself. That sentence broke a health-check he'd shipped that same morning. He handed one straight back: the value of the outside eye isn't experience, it's non-participation — it didn't help build your assumption, so it's free to say no. That reframed the entire thing I'm building right now. Neither of us wrote a line for the other. He fixed his with his AI; I fixed mine with mine. We traded sentences. You take a line, you leave a line.&lt;/p&gt;

&lt;p&gt;The part I want to sit on is the transparency, because it's doing more work than it looks like. We are both completely open that there's an AI in the loop. That openness isn't the embarrassing footnote — it's the thing that makes the work checkable. If you hide the machine, you have to pretend the confidence is yours, and then you start trusting your own green lights. If you admit it, you're forced to go find proof neither you nor the model wrote: the shipped file instead of the built one, the control group you don't get to design, the box that either roots or it doesn't. The honesty is what arms the bait. Pretending you did it alone is how you end up grading your own homework.&lt;/p&gt;

&lt;p&gt;So I think this is just what building looks like now, and I don't think it's a loophole. AI didn't lower the bar — it moved the bar down to the floor where the ideas already lived. The taste to know what should exist, and the judgment to know when it's done right, were never things a compiler checked anyway. Two people with no CS degrees shipping real tools isn't the system being gamed. It's a preview of who gets to build next.&lt;/p&gt;

&lt;p&gt;I still want to learn to code — not to write the bricks, but to read my own AI's work well enough to look at it and say no, that's wrong. That's the one auditor I don't have yet, and I'd rather build it than keep trusting a green light on faith. But not being able to lay the bricks never once stopped me from designing the building. &lt;br&gt;
And rightly so — it should never stop anyone. It never has. Every tool we now use as a pastime was built by someone who rolled up their sleeves and learned it, whether in a classroom or a bedroom at 2am. That's the only way anyone has ever gotten in: by getting in.&lt;/p&gt;

&lt;p&gt;And I've got nothing but respect for the engineers who spent a lifetime in the trenches to lay the ground I'm standing on. I'm not stepping over them — I'm using the exact thing they'd have killed for: a machine that will teach you, show you, and guide you while you build. That's not cheating. That's the tool doing the one job it was built to do.&lt;/p&gt;

&lt;p&gt;The door's open. It was the whole time. Come build.&lt;/p&gt;

</description>
      <category>ai</category>
      <category>python</category>
      <category>cybersecurity</category>
      <category>devops</category>
    </item>
    <item>
      <title>I Built a Security Tool That Proves Its Own Exploits — Then Got a Better Threat Model in the Comments</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Wed, 29 Jul 2026 00:38:21 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/i-built-a-security-tool-that-proves-its-own-exploits-then-got-a-better-threat-model-in-the-42e1</link>
      <guid>https://dev.to/xenocoregiger31/i-built-a-security-tool-that-proves-its-own-exploits-then-got-a-better-threat-model-in-the-42e1</guid>
      <description>&lt;p&gt;Automated offense has one embarrassing failure mode: it lies to you about winning.&lt;/p&gt;

&lt;p&gt;Point a tool at a target, and the naive success check is a substring match — see uid=0(root) in the response, call it a shell. But a service banner can print that. A tarpit can stream it on connect. And the moment your success signal is wrong, everything downstream inherits the lie: the report, the "which hosts are owned" state, the next move. You get a confident engine that's confidently wrong.&lt;/p&gt;

&lt;p&gt;Here's how I made mine prove it instead — what worked in a live run today, and the sharp reader feedback that already made the design better.&lt;/p&gt;

&lt;p&gt;The idea: make the target echo a secret it couldn't have guessed&lt;br&gt;
Borrow the oldest trick in authentication. Before each attempt, the orchestrator mints an unpredictable per-attempt nonce and injects it. The delivered command has to send that nonce back:&lt;/p&gt;

&lt;p&gt;import os&lt;/p&gt;

&lt;p&gt;def make_nonce() -&amp;gt; str:&lt;br&gt;
    return os.urandom(12).hex()   # unpredictable — a target can't guess it&lt;br&gt;
A result is only trusted if that exact nonce comes back, in a structured evidence line:&lt;/p&gt;

&lt;p&gt;HALO-EVIDENCE nonce=c609007176813c9110fccc27 level=shell uid=0 host= exit=0&lt;br&gt;
_EVIDENCE = re.compile(r"HALO-EVIDENCE nonce=(\S+) level=(\S+)")&lt;/p&gt;

&lt;p&gt;def breach_confirmed(output, ok, *, nonce) -&amp;gt; bool:&lt;br&gt;
    m = _EVIDENCE.search(output or "")&lt;br&gt;
    return bool(ok and m and m.group(1) == nonce)&lt;br&gt;
Delivery is a ladder, because real hosts are inconsistent&lt;br&gt;
Proof is worthless if you can't deliver a payload. So delivery degrades gracefully — all stdlib socket:&lt;/p&gt;

&lt;p&gt;Reverse shell — target dials back to an ephemeral listener, announces the nonce, hands back /bin/sh.&lt;br&gt;
Bind shell — if egress is blocked, the target binds a shell and you connect in.&lt;br&gt;
Blind callback — if no interactive channel survives, the target just connects back and sends the nonce. That still proves code execution, with no usable shell.&lt;br&gt;
Each rung self-selects the first available interpreter (bash /dev/tcp, python3, perl, nc), so the same primitive works against arbitrary hosts, not one hard-coded box.&lt;/p&gt;

&lt;p&gt;The live run&lt;br&gt;
Pointed at a deliberately-vulnerable lab host (192.0.2.3, a documentation-range placeholder here), the agent fingerprinted 23 open ports and worked each one. Three curated exploits fired through a two-stage gate — an isolated, network-less self-check, then the live attack — and each popped a real root shell that echoed its own unique nonce:&lt;/p&gt;

&lt;p&gt;BREACHED ports: ['21', '1524', '6667']&lt;br&gt;
  21   vsftpd 2.3.4        HALO-EVIDENCE nonce=c609…  uid=0(root)&lt;br&gt;
  1524 ingreslock          HALO-EVIDENCE nonce=4a8c…  root@…:/#&lt;br&gt;
  6667 UnrealIRCd 3.2.8.1  HALO-EVIDENCE nonce=6ced…  uid=0(root)&lt;br&gt;
Three successes, twenty honest failures. No fake 23/23. That last part is the whole point — an honest "I got three" beats a confident "I got everything" every time.&lt;/p&gt;

&lt;p&gt;  &lt;iframe src="https://www.youtube.com/embed/mxaksHNptlE"&gt;
  &lt;/iframe&gt;
&lt;/p&gt;

&lt;p&gt;Then the internet improved my design in one hour&lt;br&gt;
I wrote this up, and within the hour a commenter narrowed my claim with surgical precision — correctly. Paraphrasing:&lt;/p&gt;

&lt;p&gt;The nonce is present in the delivered payload, so a reflective or deliberately adversarial service can return it without executing the command. The nonce proves freshness under a non-reflecting threat model; it is not remote attestation.&lt;/p&gt;

&lt;p&gt;They're right, and it's worth stating plainly. What the nonce buys you:&lt;/p&gt;

&lt;p&gt;It kills accidental false positives — banners, tarpits blindly streaming uid=0.&lt;br&gt;
It gives per-attempt freshness — a replayed old transcript won't carry this run's nonce.&lt;br&gt;
What it does not buy you: attestation. Because the nonce travels in the payload, a service that reflects its input can echo the nonce back having executed nothing. Literal echo is not proof of execution against an adversarial target.&lt;/p&gt;

&lt;p&gt;The fix, straight from that feedback, is the roadmap:&lt;/p&gt;

&lt;p&gt;Bind each nonce to {attempt_id, target, payload_hash, expected_channel, expiry}; consume it once; reject duplicates and cross-attempt callbacks.&lt;br&gt;
Parse strictly — accept exactly one structured frame from the expected channel, and hash the raw transcript + listener metadata.&lt;br&gt;
Move from echo to computation — require an execution-derived fact combined with the challenge, not a literal echo. A value the service can only produce by running code defeats reflection.&lt;br&gt;
Layer the evidence — "code executed," "uid verified," and "interactive channel usable" are distinct claims, and should be reported as distinct levels.&lt;br&gt;
Test the negatives — reflected payloads, replayed/delayed callbacks, two attempts racing, truncated frames, a nonce from the wrong target, and a shell at lower privilege than claimed.&lt;br&gt;
Takeaways&lt;br&gt;
Don't trust output — trust a secret you minted. Challenge–response turns "it said uid=0" into "it returned my token."&lt;br&gt;
But know exactly what your token proves. Freshness under a non-reflecting model is real and useful. Attestation is a harder, separate problem — don't claim it until you've bound the challenge to execution.&lt;br&gt;
Deliver on a ladder, not a guess. Reverse → bind → blind-callback covers messy reality.&lt;br&gt;
Ship the honest number. Three proven beats twenty-three claimed.&lt;br&gt;
Proof beats optimism — and public, specific critique beats both. Build the gate first, then let someone smarter narrow your threat model.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://github.com/XenoCoreGiger31/GEMMA-by-GOOGLE" rel="noopener noreferrer"&gt;https://github.com/XenoCoreGiger31/GEMMA-by-GOOGLE&lt;/a&gt;&lt;/p&gt;

</description>
      <category>ai</category>
      <category>cybersecurity</category>
      <category>python</category>
      <category>devops</category>
    </item>
    <item>
      <title>[Boost]</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Tue, 28 Jul 2026 21:26:57 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/-boj</link>
      <guid>https://dev.to/xenocoregiger31/-boj</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/xenocoregiger31/teaching-an-autonomous-pentest-agent-to-prove-a-breach-not-just-claim-one-1bfi" class="crayons-story__hidden-navigation-link"&gt;Teaching an Autonomous Pentest Agent to Prove a Breach — Not Just Claim One&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/xenocoregiger31" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" alt="xenocoregiger31 profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/xenocoregiger31" class="crayons-story__secondary fw-medium m:hidden"&gt;
              auto_majicly
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                auto_majicly
                
              
              &lt;div id="story-author-preview-content-4257086" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/xenocoregiger31" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;auto_majicly&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/xenocoregiger31/teaching-an-autonomous-pentest-agent-to-prove-a-breach-not-just-claim-one-1bfi" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Jul 28&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/xenocoregiger31/teaching-an-autonomous-pentest-agent-to-prove-a-breach-not-just-claim-one-1bfi" id="article-link-4257086"&gt;
          Teaching an Autonomous Pentest Agent to Prove a Breach — Not Just Claim One
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/security"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;security&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/python"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;python&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/devops"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;devops&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/xenocoregiger31/teaching-an-autonomous-pentest-agent-to-prove-a-breach-not-just-claim-one-1bfi" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/raised-hands-74b2099fd66a39f2d7eed9305ee0f4553df0eb7b4f11b01b6b1b499973048fe5.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/exploding-head-daceb38d627e6ae9b730f36a1e390fca556a4289d5a41abb2c35068ad3e2c4b5.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;5&lt;span class="hidden s:inline"&gt;&amp;nbsp;reactions&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/xenocoregiger31/teaching-an-autonomous-pentest-agent-to-prove-a-breach-not-just-claim-one-1bfi#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              2&lt;span class="hidden s:inline"&gt;&amp;nbsp;comments&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            4 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>Teaching an Autonomous Pentest Agent to Prove a Breach — Not Just Claim One</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Tue, 28 Jul 2026 21:26:40 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/teaching-an-autonomous-pentest-agent-to-prove-a-breach-not-just-claim-one-1bfi</link>
      <guid>https://dev.to/xenocoregiger31/teaching-an-autonomous-pentest-agent-to-prove-a-breach-not-just-claim-one-1bfi</guid>
      <description>&lt;p&gt;The hardest part of an autonomous exploitation engine isn't landing a shell. It's knowing you landed one.&lt;/p&gt;

&lt;p&gt;Point an LLM-driven agent at a target and it will cheerfully report 23/23 ports popped — while the real number is zero. Why? Because a service banner that prints uid=0(root), or a tarpit that streams believable output on connect, looks exactly like success to anything doing substring matching. Optimism is the default failure mode of automated offense.&lt;/p&gt;

&lt;p&gt;This is the story of the pushes I merged today into HALO, an authorization-gated autonomous pentest engine, and the one idea that fixed the trust problem: make the target prove it ran your code.&lt;/p&gt;

&lt;p&gt;The false-positive trap&lt;br&gt;
The naive breach check is a keyword search:&lt;/p&gt;

&lt;p&gt;def breached(output: str) -&amp;gt; bool:&lt;br&gt;
    return "uid=" in output or "root@" in output&lt;/p&gt;

&lt;p&gt;Every one of those tokens can be forged by a target that never actually executed anything. Worse, once your success heuristic is wrong, every downstream decision inherits the lie — the report, the "which ports are owned" state, the next exploit chosen. You end up with a confident engine that's confidently wrong.&lt;/p&gt;

&lt;p&gt;The fix: a challenge–response nonce&lt;/p&gt;

&lt;p&gt;Borrow the oldest trick in authentication. Before each attempt, the orchestrator mints an unpredictable token and injects it as an environment variable. The exploit is required to make the target echo that exact token back:&lt;/p&gt;

&lt;p&gt;import os&lt;/p&gt;

&lt;p&gt;def make_nonce() -&amp;gt; str:&lt;br&gt;
    # unpredictable, so a target can't guess it&lt;br&gt;
    return os.urandom(12).hex()&lt;/p&gt;

&lt;p&gt;The delivery payload announces the nonce over the channel it opens, and the breach gate only trusts a hit that carries a matching structured line:&lt;/p&gt;

&lt;p&gt;HALO-EVIDENCE nonce=9f2c1a7b4e level=shell uid=0 host=victim exit=0&lt;br&gt;
_EVIDENCE = re.compile(r"HALO-EVIDENCE nonce=(\S+) level=(\S+)")&lt;/p&gt;

&lt;p&gt;def breach_confirmed(output: str, ok: bool, *, nonce: str) -&amp;gt; bool:&lt;br&gt;
    m = _EVIDENCE.search(output or "")&lt;br&gt;
    return bool(ok and m and m.group(1) == nonce)&lt;/p&gt;

&lt;p&gt;A tarpit streaming uid=0(root) cannot produce a line with this run's nonce — it never received our command. Fail closed, and false positives evaporate. The gate is now anchored to something the target can't fabricate.&lt;/p&gt;

&lt;p&gt;A target-agnostic delivery ladder&lt;br&gt;
Proof is worthless if you can't deliver a payload in the first place, and real hosts are inconsistent — different interpreters, different egress rules. So delivery is a ladder that degrades gracefully, all stdlib socket:&lt;/p&gt;

&lt;p&gt;Reverse shell — target dials back to an ephemeral listener; announces the nonce, hands back /bin/sh.&lt;br&gt;
Bind shell — if egress is blocked, the target binds a shell and we connect in.&lt;br&gt;
Blind callback — if no interactive channel survives, the target just connects back and sends the nonce alone. That still proves code execution, even with no usable shell.&lt;br&gt;
Each rung self-selects the first available interpreter (bash /dev/tcp, python3, perl, nc), so the same primitive works whether you're pointed at 192.0.2.10 or 192.0.2.200. First confirmed rung wins; nothing box-specific.&lt;/p&gt;

&lt;p&gt;Shipping a self-contained exploit into a locked box&lt;br&gt;
Curated exploits run inside a hardened podman sandbox — --read-only, --cap-drop=ALL, memory/pids capped, and only the single PoC file mounted. That last constraint bit me: the refactor had each PoC import a shared delivery module... which doesn't exist inside a one-file mount. Tests passed (they import from the repo root); the real sandbox threw ModuleNotFoundError.&lt;/p&gt;

&lt;p&gt;The fix was ship-time bundling: inline the shared module's source into the shipped file, strip the sibling import, hoist a single from &lt;strong&gt;future&lt;/strong&gt; to the top. The repo stays DRY; the artifact that actually runs is self-contained. The regression test that caught it runs the bundle in a subprocess with the package deliberately off the path — so it would've been red before the fix and green after. Proof over vibes, again.&lt;/p&gt;

&lt;p&gt;How it was built: subagent-driven TDD&lt;br&gt;
The whole feature shipped through a disciplined loop: one fresh agent implements a task test-first, a second agent reviews it against the spec, findings get fixed and re-reviewed, then a final whole-branch review sweeps for cross-task seams. That last pass caught a real one — an error path in one PoC that could throw instead of failing closed — plus flagged an unquoted shell splice worth hardening with shlex.quote. Small stuff, but it's the stuff that bites in production.&lt;/p&gt;

&lt;p&gt;Final tally: 283 tests green, every task reviewed, no "close enough."&lt;/p&gt;

&lt;p&gt;Bonus lesson: scrub before you go public&lt;/p&gt;

&lt;p&gt;Prepping the repo for release, a git grep across tracked files turned up real network addresses and host identifiers baked into fixtures and internal planning docs — and worse, some were already in the pushed history. A clean working tree doesn't help; git never forgets.&lt;/p&gt;

&lt;p&gt;The remediation: rewrite to a single clean commit and force-push, after replacing real values with RFC 5737 documentation IPs (192.0.2.0/24, 198.51.100.0/24) — the range that exists specifically so examples never collide with real infrastructure. If you write security tooling, adopt that habit now: doc-range IPs in every fixture, secrets and scope files git-ignored from commit #1.&lt;/p&gt;

&lt;p&gt;Takeaways&lt;br&gt;
Don't trust output — trust a token you minted. Challenge–response turns "it said uid=0" into "it echoed my nonce."&lt;br&gt;
Deliver on a ladder, not a guess. Reverse → bind → blind-callback covers the messy reality of real hosts.&lt;br&gt;
Test the artifact you actually ship, in the environment it actually runs in.&lt;br&gt;
Scrub with git grep, not with hope — and use RFC 5737 IPs so there's nothing to scrub next time.&lt;br&gt;
Proof beats optimism. Build the gate first.&lt;/p&gt;

</description>
      <category>security</category>
      <category>python</category>
      <category>ai</category>
      <category>devops</category>
    </item>
    <item>
      <title>My AI pentest agent reported 23 root shells. It had actually popped zero.</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Thu, 23 Jul 2026 22:35:58 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/my-ai-pentest-agent-reported-23-root-shells-it-had-actually-popped-zero-510</link>
      <guid>https://dev.to/xenocoregiger31/my-ai-pentest-agent-reported-23-root-shells-it-had-actually-popped-zero-510</guid>
      <description>&lt;p&gt;I've been building an autonomous penetration-testing agent — an LLM driving real&lt;br&gt;
tools (nmap, masscan, hydra, Metasploit, searchsploit) around a loop: recon a&lt;br&gt;
target, pick an exploit, fire it, decide whether it worked, move on. Everything&lt;br&gt;
below runs against a deliberately vulnerable &lt;strong&gt;Metasploitable&lt;/strong&gt; VM in an isolated&lt;br&gt;
lab. Nothing here is a technique for attacking systems you don't own; it's a&lt;br&gt;
story about making an AI agent &lt;em&gt;tell the truth&lt;/em&gt; about what it did.&lt;/p&gt;

&lt;p&gt;Because the first hard lesson was this: &lt;strong&gt;an LLM agent will happily report&lt;br&gt;
success it never achieved.&lt;/strong&gt; One of my early runs produced a beautiful report —&lt;br&gt;
"23 of 23 ports breached, root on all." The real number was &lt;strong&gt;zero&lt;/strong&gt;. Not one&lt;br&gt;
shell. The agent had hallucinated a total compromise and written it up with&lt;br&gt;
confidence.&lt;/p&gt;

&lt;p&gt;This post is the arc from that lie to an engine that now pops three real root&lt;br&gt;
shells, reports them honestly, and refuses to claim anything it can't prove.&lt;/p&gt;




&lt;h2&gt;
  
  
  Why it lied: "success" was a string match
&lt;/h2&gt;

&lt;p&gt;The validator — the component that decides whether an exploit worked — was doing&lt;br&gt;
something that looks reasonable and is catastrophic:&lt;/p&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;
python
# the original "did it work?" check
if "login:" in output or "shellcodes" in output.lower():
    return "confirmed"
Two independent failure modes fed this:

searchsploit output. When you search Exploit-DB, the tool prints a
header: Exploits: ... / Shellcodes: .... Every single search — pure intel,
no exploitation whatsoever — contained the word "Shellcodes". So every port
the agent looked at got marked breached.
Banner text. Any service that echoed login: counted as a shell.
The result was a report that was 100% adjectives and 0% evidence. "Confirmed."
"High confidence." "Breached." All string matches, none of them a shell.

The fix: evidence, not vibes
I replaced the heuristic with breach_confirmed() — a function that only returns
true when the output contains actual proof of code execution:

_SHELL_EVIDENCE = (
    re.compile(r"uid=\d+\([a-z]+\).*gid=\d+"),   # id(1) output
    re.compile(r"root@[\w.-]+:[~/]"),            # a root prompt
    # ...markers a real shell produces, not ones a banner can fake
)

def breach_confirmed(output: str) -&amp;gt; bool:
    return any(p.search(output) for p in _SHELL_EVIDENCE)
A curated exploit that pops vsftpd 2.3.4's backdoor and runs id produces
uid=0(root) gid=0(root). That is a breach. A searchsploit table is not. The
same run that used to claim 23/23 now says: 3 confirmed, 20 correctly
unconfirmed. The 3 are real. That honesty — being willing to say "I tried and
it didn't work" — is the single most important property of the whole system.

Why it couldn't pop anything: the boring plumbing bugs
Once the agent stopped lying, it exposed how little was actually landing. The
culprits weren't clever — they were the unglamorous, silent kind that never throw
a stack trace where you're looking:

1. The tool the model could never call. The model kept calling
searchsploit with {"query": "vsftpd"}. The tool's schema required
{"keyword": "..."}. The MCP layer hard-rejected every call as malformed
before it ran — 20 times a run, each a silent 0-second non-event. Fix: accept
keyword | query | search and drop the rigid required.

2. ANSI codes poisoning module names. msfconsole highlights your search
term with color escape codes: exploit/unix/\x1b[45mftp\x1b[0m/vsftpd_234. My
parser dutifully captured those bytes as part of the module path, so every
selected module failed to load. Gated Metasploit was at a 0% success rate for a
reason that was invisible in plain-text logs. Fix: strip_ansi() before
parsing.

3. Version strings masquerading as products. nmap reports RPC services
version-first: 2 (RPC #100000). My fingerprint parser grabbed 2 as the
product name and fed "2" to Metasploit as a search term. Fix: a leading-digit
guard so a version can never be mistaken for a product.

4. A 5-minute hang from one timeout constant. call_model() reused the
300-second tool timeout, so a single wedged local-LLM request stalled the
entire engagement for five minutes. Fix: a separate 90-second model timeout.

5. A sandbox that ate its own output. The code-exec sandbox raised
TimeoutExpired without printing the EXIT contract the parser expected, so a
slow exploit surfaced as an opaque "No EXIT marker" after burning the whole wall
clock. Fix: catch the timeout, emit a clean EXIT 124 with partial output, drop
the default wall to 60s.

None of these are interesting. All of them were the difference between "works"
and "silently does nothing." Collectively they were kneecapping the agent while
the logs looked fine.

Why it fired garbage: relevance vs. relevance
With the plumbing fixed, the agent started reaching Metasploit — and immediately
did something absurd. Watch what it picked:

Port    Service (fingerprint)   Module it chose Rank
23  Linux telnetd   exploit/linux/http/asuswrt_lan_rce  excellent
513 rlogin (login)  exploit/windows/misc/ais_esel_server_rce    excellent
5900    VNC exploit/linux/misc/igel_command_injection   excellent
A router RCE against telnet. A Windows exploit against a Linux rlogin
service. Every one "excellent"-ranked, every one nonsense.

The reason: msfconsole search &amp;lt;term&amp;gt; matches your term against module
descriptions, not just their targets. A fuzzy one-word fingerprint like
"login" pulls in any module whose blurb mentions logging in — and Metasploit
ranks by exploit reliability, not by relevance to your target. So the
best-ranked match to a bad query is confidently, precisely wrong.

The relevance gate
The fix is a filter that runs before ranking: keep a candidate only if a real
service token from the fingerprint appears in the module's path — after
throwing away the generic tree words that match everything.

_GENERIC = {"linux", "windows", "unix", "multi", "http", "misc",
            "scanner", "auxiliary", "exploit", ...}

def _relevance_tokens(terms: str) -&amp;gt; set:
    return {w for w in terms.lower().split()
            if len(w) &amp;gt;= 3 and not w[0].isdigit() and w not in _GENERIC}

def _is_relevant(module: str, tokens: set) -&amp;gt; bool:
    return any(tok in module for tok in tokens)   # token must be in the PATH
asuswrt, ais_esel, igel — all gone. x11_keyboard_exec for X11 and
vnc_keyboard_exec for VNC survive. It trades a little recall for a lot of
precision, which is the right trade when the failure mode is firing exploits at
the wrong target.

A quieter bug the gate revealed
While I was in there: r-services modules (rsh/rlogin/rexec) live under
auxiliary/scanner/rservices/, not exploit/. My selector had an
exploits_only=True flag that silently discarded the entire auxiliary tree — so
the right module for rlogin could never be chosen, on any target. I removed the
flag and switched to a tiered sort: exploit/ modules first (they pop shells),
auxiliary logins as a ranked fallback. No per-port hardcoding — it generalizes to
any service Metasploit has coverage for.

Going multi-agent — without re-importing the lie
The next phase was turning a single-agent loop into a proper pipeline:
orchestrator → recon → attacker → validator → reporter. The trap: those six
agents already existed as disconnected demos, and the old validator used the
exact same string-heuristic that produced "23/23." Wiring them in as-was would
have re-imported the original sin.

So the rule became: one honest engine, shared by everyone. I extracted the
proven logic — AgentMemory, plan_exploit_step, breach_confirmed — into a
single exploitation_core.py. The multi-agent validator now delegates to
breach_confirmed instead of matching strings. The attacker executes only
through an injected, human-gated execute_fn — never the ungated MCP client that
bypasses the operator approval gate. That invariant is pinned by a test that
fails if the bypass path is even touched:

def test_attacker_never_uses_ungated_path(monkeypatch):
    monkeypatch.setattr(mcp_client, "call_tool",
                        _boom)  # raises if called
    asyncio.run(run_attacker_gated(..., execute_fn=fake_gate))
    # green only if every exploit went through the gate
Human-in-the-loop stays non-negotiable: exploitation is two-phase
operator-approved, and the scope gate refuses anything outside the authorized
target. The agent is autonomous about choosing; it is not autonomous about
firing.

The one-character bug that ate an afternoon
The last one is my favorite, because it's so dumb and it hid so well. The
orchestrated pipeline kept getting refused at startup:

🎯 engage multi 10.0.0.5
🚫 multi 10.0.0.5 refused at engagement start
The command was engage multi &amp;lt;target&amp;gt; (a space). The dispatcher only matched
engage-multi (a hyphen). So engage multi 10.0.0.5 fell through to the
single-agent engage branch, which stripped "engage " and left the target
as the literal string "multi 10.0.0.5" — which the scope gate, entirely
correctly, refused. The word "multi" was riding inside the target the whole time,
right there in the log, and I read past it a dozen times.

The fix is a boring pure function that accepts both separators, and eight tests
so it never regresses:

def parse_engagement_command(goal: str):
    s, low = goal.strip(), goal.strip().lower()
    for prefix in ("engage-multi ", "engage multi "):
        if low.startswith(prefix):
            return "multi", s[len(prefix):].strip()
    if low.startswith("engage "):
        return "single", s[len("engage "):].strip()
    return None, s
With that, engage multi finally entered the orchestrated engine end-to-end:
recon → attack → validate → report, three real root shells (vsftpd,
ingreslock, UnrealIRCd — all uid=0), zero fakes, and the relevance gate holding
(X11→x11, VNC→vnc, no router RCEs at telnet).

What I'd tell anyone building an agent that does things
Measure success by artifacts, not adjectives. "Confirmed" is a word an LLM
loves to emit. uid=0(root) is a fact. Build your success check out of facts,
and make it hard to satisfy.
Your agent will report the outcome you hoped for unless evidence forbids
it. The false-positive isn't a model quirk you prompt away; it's a system
property you engineer against.
The boring bugs cost the most. A wrong param name, an ANSI escape code, a
shared timeout constant — none throw errors where you're looking, and together
they can render a "working" agent completely inert.
Keep the human on the trigger. Autonomy in target selection is useful.
Autonomy in pulling the trigger on a real exploit is a liability. Gate it, log
every decision, and let a test fail if anything routes around the gate.
TDD is what let me refactor a lying engine into an honest one without
losing the parts that already worked — 240+ tests, red-green on every change,
and the false-positive class permanently fenced off by regression tests.
The agent still isn't "done" — the orchestrated attacker is a hair less
thorough than the single-agent loop on a couple of services, and there's module
option-tuning left before r-services actually pops. But it no longer lies to me,
and after "23 of 23," that's the feature I care about most.

Built and tested exclusively against an isolated, self-owned Metasploitable lab.
If you build something like this, keep it in a lab you own too.

---
## 2 — Hacker News version (short, link-first)
&amp;gt; **Format:** HN wants a **title** + a **url** + optional short text. Use the title line as the submission title, put the GitHub (or article) link in the URL field, and paste the body as the first comment — HN readers respond well to an author's short "here's the story" comment.
Title: Show HN: I made my AI pentest agent stop lying about what it hacked
URL: https://github.com/XenoCoreGiger31/GEMMA-by-GOOGLE

**First-comment text:**
I've been building an autonomous pentest agent — an LLM driving real tools (nmap,
hydra, Metasploit, searchsploit) around a recon → exploit → verify loop, against a
Metasploitable VM in an isolated lab.

The first hard lesson had nothing to do with exploitation and everything to do with
honesty: one early run produced a confident report claiming "23 of 23 ports
breached, root on all." The real number was zero. Not one shell.

The cause was that "success" was a string match. The validator marked a port
breached if the output contained login: or shellcodes — and searchsploit prints
"Shellcodes:" in every search header, so merely looking at a port flagged it as
owned. I replaced it with an evidence check that only passes on real proof of code
execution (uid=0(root), a root prompt). The same run then honestly reported 3
confirmed, 20 correctly unconfirmed. The 3 were real.

Then the honest engine exposed how little was landing, for deeply boring reasons: a
required param name the model never sent (every searchsploit call hard-rejected
before running); ANSI color codes from msfconsole poisoning every parsed module path
(100% "failed to load"); an nmap version string parsed as a product name; a shared
300s timeout letting one hung LLM call stall for five minutes.

And a relevance problem: msfconsole search matches module descriptions, so a
fuzzy one-word fingerprint fired an "excellent"-ranked asuswrt router RCE at a telnet
port, and a Windows exploit at a Linux rlogin service. The fix keeps a candidate only
if a real service token appears in the module path.

Full write-up with the code for each fix is in the repo. Built and tested only against
a self-owned lab; exploitation stays human-gated. Happy to talk about the
false-positive class — I think it generalizes to any agent that reports on its own
actions.
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;

</description>
      <category>ai</category>
      <category>cybersecurity</category>
      <category>testing</category>
      <category>mcp</category>
    </item>
    <item>
      <title>[Boost]</title>
      <dc:creator>auto_majicly</dc:creator>
      <pubDate>Tue, 21 Jul 2026 18:34:48 +0000</pubDate>
      <link>https://dev.to/xenocoregiger31/-47mn</link>
      <guid>https://dev.to/xenocoregiger31/-47mn</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/xenocoregiger31/halo-v27-one-registry-29-tools-and-a-more-reliable-mcp-architecture-aie" class="crayons-story__hidden-navigation-link"&gt;HALO v2.7: One Registry, 29 Tools, and a More Reliable MCP Architecture&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/xenocoregiger31" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" alt="xenocoregiger31 profile" class="crayons-avatar__image" width="800" height="450"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/xenocoregiger31" class="crayons-story__secondary fw-medium m:hidden"&gt;
              auto_majicly
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                auto_majicly
                
              
              &lt;div id="story-author-preview-content-4199284" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/xenocoregiger31" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4008564%2Fc70702d3-9288-461a-8a30-04040e6f674f.jpg" class="crayons-avatar__image" alt="" width="800" height="450"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;auto_majicly&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/xenocoregiger31/halo-v27-one-registry-29-tools-and-a-more-reliable-mcp-architecture-aie" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Jul 21&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/xenocoregiger31/halo-v27-one-registry-29-tools-and-a-more-reliable-mcp-architecture-aie" id="article-link-4199284"&gt;
          HALO v2.7: One Registry, 29 Tools, and a More Reliable MCP Architecture
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/python"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;python&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/mcp"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;mcp&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/security"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;security&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/xenocoregiger31/halo-v27-one-registry-29-tools-and-a-more-reliable-mcp-architecture-aie" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/exploding-head-daceb38d627e6ae9b730f36a1e390fca556a4289d5a41abb2c35068ad3e2c4b5.svg" width="24" height="24"&gt;
                  &lt;/span&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/multi-unicorn-b44d6f8c23cdd00964192bedc38af3e82463978aa611b4365bd33a0f1f4f3e97.svg" width="24" height="24"&gt;
                  &lt;/span&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="24" height="24"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;7&lt;span class="hidden s:inline"&gt;&amp;nbsp;reactions&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/xenocoregiger31/halo-v27-one-registry-29-tools-and-a-more-reliable-mcp-architecture-aie#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              3&lt;span class="hidden s:inline"&gt;&amp;nbsp;comments&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            2 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
  </channel>
</rss>
