<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:dc="http://purl.org/dc/elements/1.1/">
  <channel>
    <title>DEV Community: Gagandeep Singh</title>
    <description>The latest articles on DEV Community by Gagandeep Singh (@gagan1985).</description>
    <link>https://dev.to/gagan1985</link>
    <image>
      <url>https://media2.dev.to/dynamic/image/width=90,height=90,fit=cover,gravity=auto,format=auto/https:%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4011900%2F9f962f5d-c2a2-49df-9433-d1acdb6d6bdf.jpeg</url>
      <title>DEV Community: Gagandeep Singh</title>
      <link>https://dev.to/gagan1985</link>
    </image>
    <atom:link rel="self" type="application/rss+xml" href="https://dev.to/feed/gagan1985"/>
    <language>en</language>
    <item>
      <title>[Boost]</title>
      <dc:creator>Gagandeep Singh</dc:creator>
      <pubDate>Fri, 03 Jul 2026 07:24:02 +0000</pubDate>
      <link>https://dev.to/gagan1985/-32h5</link>
      <guid>https://dev.to/gagan1985/-32h5</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/innerkore/ai-can-build-your-ui-in-seconds-whos-handling-the-forms-1ab9" class="crayons-story__hidden-navigation-link"&gt;AI Can Build Your UI in Seconds. Who's Handling the Forms?&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;
          &lt;a class="crayons-logo crayons-logo--l" href="/innerkore"&gt;
            &lt;img alt="Innerkore Technologies logo" src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Forganization%2Fprofile_image%2F13877%2F5e5edc66-c801-42c4-b44f-e89782068e23.jpeg" class="crayons-logo__image" width="800" height="800"&gt;
          &lt;/a&gt;

          &lt;a href="/gagan1985" class="crayons-avatar  crayons-avatar--s absolute -right-2 -bottom-2 border-solid border-2 border-base-inverted  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4011900%2F9f962f5d-c2a2-49df-9433-d1acdb6d6bdf.jpeg" alt="gagan1985 profile" class="crayons-avatar__image" width="450" height="450"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/gagan1985" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Gagandeep Singh
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Gagandeep Singh
                
              
              &lt;div id="story-author-preview-content-4057556" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/gagan1985" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F4011900%2F9f962f5d-c2a2-49df-9433-d1acdb6d6bdf.jpeg" class="crayons-avatar__image" alt="" width="450" height="450"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Gagandeep Singh&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

            &lt;span&gt;
              &lt;span class="crayons-story__tertiary fw-normal"&gt; for &lt;/span&gt;&lt;a href="/innerkore" class="crayons-story__secondary fw-medium"&gt;Innerkore Technologies&lt;/a&gt;
            &lt;/span&gt;
          &lt;/div&gt;
          &lt;a href="https://dev.to/innerkore/ai-can-build-your-ui-in-seconds-whos-handling-the-forms-1ab9" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Jul 3&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/innerkore/ai-can-build-your-ui-in-seconds-whos-handling-the-forms-1ab9" id="article-link-4057556"&gt;
          AI Can Build Your UI in Seconds. Who's Handling the Forms?
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/forms"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;forms&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/webdev"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;webdev&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/productivity"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;productivity&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/innerkore/ai-can-build-your-ui-in-seconds-whos-handling-the-forms-1ab9" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="24" height="24"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;1&lt;span class="hidden s:inline"&gt;&amp;nbsp;reaction&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/innerkore/ai-can-build-your-ui-in-seconds-whos-handling-the-forms-1ab9#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            3 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>AI Can Build Your UI in Seconds. Who's Handling the Forms?</title>
      <dc:creator>Gagandeep Singh</dc:creator>
      <pubDate>Fri, 03 Jul 2026 07:23:37 +0000</pubDate>
      <link>https://dev.to/innerkore/ai-can-build-your-ui-in-seconds-whos-handling-the-forms-1ab9</link>
      <guid>https://dev.to/innerkore/ai-can-build-your-ui-in-seconds-whos-handling-the-forms-1ab9</guid>
      <description>&lt;h2&gt;
  
  
  We Keep Reinventing the Same Wheel
&lt;/h2&gt;

&lt;p&gt;Every project I've started in the last five years has had the same moment. You build the UI, you wire up state, you get the design looking sharp — and then someone says "we need a contact form." Or a feedback form. Or a waitlist sign-up.&lt;/p&gt;

&lt;p&gt;And you spend the next afternoon doing the same thing you've done a dozen times before: writing a route that accepts a POST, validating the body, pushing rows into a Google Sheet, sending a confirmation email, hoping nothing breaks at 2am.&lt;/p&gt;

&lt;p&gt;FormProxy exists because I got tired of that afternoon.&lt;/p&gt;




&lt;h2&gt;
  
  
  What FormProxy Actually Does
&lt;/h2&gt;

&lt;p&gt;FormProxy is a form backend. That sounds small until you think about what "form backend" actually means:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Accepting submissions&lt;/strong&gt; from any HTML form or API call, without you writing a handler&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Storing everything&lt;/strong&gt; with configurable retention (7 days on free, 90 days on paid plans)&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Routing submissions&lt;/strong&gt; to wherever your team actually lives — Slack, email, Google Sheets, webhooks, Zapier&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Managing multiple forms&lt;/strong&gt; across workspaces, with per-form signing secrets and activation toggles&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;You drop a form endpoint URL into your &lt;code&gt;&amp;lt;form action="..."&amp;gt;&lt;/code&gt; and you're done. No backend code. No database schema. No queue to maintain.&lt;/p&gt;




&lt;h2&gt;
  
  
  Why This Matters Even More Now That Everyone Has AI
&lt;/h2&gt;

&lt;p&gt;Here's the thing nobody says out loud: AI has made it dramatically easier to build &lt;em&gt;frontends&lt;/em&gt;, but the boring infrastructure behind forms hasn't gotten any smarter.&lt;/p&gt;

&lt;p&gt;ChatGPT or Claude can generate a beautiful, accessible contact form in seconds. Cursor can wire up the validation. But the moment that form needs to &lt;em&gt;do something&lt;/em&gt; — store a submission, notify Slack, sync to a spreadsheet — you're back to writing SMTP handlers and webhook endpoints by hand.&lt;/p&gt;

&lt;p&gt;The gap between "AI can build my UI" and "AI can run my backend" is real. Tools like v0 and Lovable and bolt.new are accelerating frontend development faster than infrastructure can keep up.&lt;/p&gt;

&lt;p&gt;FormProxy fills exactly that gap. When your AI-generated landing page needs a "Join Waitlist" form that:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Stores signups with timestamps&lt;/li&gt;
&lt;li&gt;Pings your team in Slack instantly
&lt;/li&gt;
&lt;li&gt;Syncs every row to a Google Sheet your non-technical co-founder can read&lt;/li&gt;
&lt;li&gt;Sends a confirmation email to the user&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;...you don't need to write any of that. You configure it in a UI and point your form at the endpoint.&lt;/p&gt;




&lt;h2&gt;
  
  
  The Integrations Are the Point
&lt;/h2&gt;

&lt;p&gt;The real value isn't storage — it's routing. Here's what FormProxy can do with a submission the moment it lands:&lt;/p&gt;

&lt;h3&gt;
  
  
  Google Sheets
&lt;/h3&gt;

&lt;p&gt;Every submission becomes a row. Columns are inferred automatically from your form fields — add a new field to your form and it appears as a new column on the next submission, no schema migration required.&lt;/p&gt;

&lt;p&gt;We use Google's OAuth flow with &lt;code&gt;drive.file&lt;/code&gt; scope (not the overly broad &lt;code&gt;spreadsheets&lt;/code&gt; scope), so you retain control of exactly which files FormProxy can touch. You pick the spreadsheet via a Picker UI — no copy-pasting spreadsheet IDs.&lt;/p&gt;

&lt;h3&gt;
  
  
  Slack
&lt;/h3&gt;

&lt;p&gt;Instant notification to any channel. Good for lead forms, support requests, anything your team wants to see in real-time.&lt;/p&gt;

&lt;h3&gt;
  
  
  Webhooks
&lt;/h3&gt;

&lt;p&gt;Full POST to any URL with HMAC signing (using the form's signing secret), so your own services can verify the submission is genuine before processing it.&lt;/p&gt;

&lt;h3&gt;
  
  
  Email
&lt;/h3&gt;

&lt;p&gt;Confirmation emails to submitters, notification emails to your team. Configurable templates, no mail server setup on your end.&lt;/p&gt;

&lt;h3&gt;
  
  
  Zapier
&lt;/h3&gt;

&lt;p&gt;One integration that unlocks 5,000+ apps. If FormProxy doesn't have a native integration for your tool, Zapier fills the gap.&lt;/p&gt;




&lt;h2&gt;
  
  
  A Real Example: AI-Generated SaaS Landing Page
&lt;/h2&gt;

&lt;p&gt;Here's a workflow I've used in production:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;Use v0 to generate a landing page with a "Get Early Access" form&lt;/li&gt;
&lt;li&gt;Point &lt;code&gt;&amp;lt;form action="https://app.formproxy.com/f/{uid}"&amp;gt;&lt;/code&gt; at FormProxy&lt;/li&gt;
&lt;li&gt;Configure: Google Sheets sync (so I have a CRM-lite spreadsheet), Slack notification (so I see signups immediately), email reply (so users get a confirmation)&lt;/li&gt;
&lt;li&gt;Done in under 10 minutes&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;No Netlify Functions. No serverless cold starts. No "oh we missed 3 signups because the function timed out."&lt;/p&gt;

&lt;p&gt;The AI built the frontend. FormProxy handled everything else.&lt;/p&gt;




&lt;h2&gt;
  
  
  What's Next
&lt;/h2&gt;

&lt;p&gt;We're actively building:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;AI-powered submission summaries&lt;/strong&gt; — ask natural language questions about your form data&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Conditional routing&lt;/strong&gt; — send to Slack only if a field contains a certain value&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Team comments&lt;/strong&gt; on submissions&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;File uploads&lt;/strong&gt; — already supported on paid plans, expanding to more types&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;If any of this resonates, try it: &lt;a href="https://app.formproxy.com" rel="noopener noreferrer"&gt;app.formproxy.com&lt;/a&gt;. Free tier is genuinely useful — 3 forms, 7-day retention, all integrations.&lt;/p&gt;

&lt;p&gt;Would love to hear what integrations you'd want to see next. Drop a comment.&lt;/p&gt;




&lt;p&gt;&lt;em&gt;FormProxy is built with FastAPI + Next.js&lt;/em&gt;&lt;/p&gt;

</description>
      <category>forms</category>
      <category>ai</category>
      <category>webdev</category>
      <category>productivity</category>
    </item>
    <item>
      <title>Part 2: Why We Built Our Own TinyBERT (and How It Beat Shiprocket's) - Indian Address Parser</title>
      <dc:creator>Gagandeep Singh</dc:creator>
      <pubDate>Fri, 03 Jul 2026 06:41:45 +0000</pubDate>
      <link>https://dev.to/gagan1985/part-2-why-we-built-our-own-tinybert-and-how-it-beat-shiprockets-indian-address-parser-1jne</link>
      <guid>https://dev.to/gagan1985/part-2-why-we-built-our-own-tinybert-and-how-it-beat-shiprockets-indian-address-parser-1jne</guid>
      <description>&lt;p&gt;&lt;em&gt;This is Part 2 of a series on building an open-source Indian address parser. &lt;a href="https://dev.to/gagan1985/building-an-open-source-indian-address-parser-from-raw-mcabank-data-to-a-fine-tuned-llm-2n9c"&gt;Part 1&lt;/a&gt; covered fine-tuning Qwen3-0.6B with LoRA and our first benchmark against Shiprocket's &lt;code&gt;open-tinybert-indian-address-ner&lt;/code&gt;. This post covers the third and final model in the series, and what happened when we pointed it at the exact model that inspired it.&lt;/em&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  The itch Shiprocket's benchmark left behind
&lt;/h2&gt;

&lt;p&gt;When we first benchmarked our Qwen3-0.6B model against Shiprocket's &lt;code&gt;open-tinybert-indian-address-ner&lt;/code&gt;, the headline was good news wrapped in a caveat: we won on every one of the nine conceptually-shared fields, sometimes by a wide margin — but Shiprocket's model was doing it with a 6-layer, 768-hidden BERT variant that ran in &lt;strong&gt;19 milliseconds&lt;/strong&gt; per address on CPU. Ours took over four seconds. That's not a rounding error; that's a 240x gap.&lt;/p&gt;

&lt;p&gt;It's easy to wave that away — "different tradeoff, different use case" — and mostly that's true. But it kept nagging at us. Shiprocket had clearly made a deliberate choice: trade some accuracy for a model small enough to run in a hot path, cheaply, at scale. We'd built the opposite thing. Neither choice is wrong, but we only had one point on that curve.&lt;/p&gt;

&lt;p&gt;So the natural next step wasn't "make Qwen faster." It was: &lt;strong&gt;what if we made the same architectural bet Shiprocket did, but trained it on our own gold-labeled data?&lt;/strong&gt; Same idea — small BERT encoder, BIO tagging instead of JSON generation — applied to our 13-field schema instead of theirs.&lt;/p&gt;

&lt;p&gt;That's &lt;code&gt;huawei-noah/TinyBERT_General_4L_312D&lt;/code&gt;: 4 layers, 312 hidden dimensions, about 14 million parameters. For comparison, that's roughly 5x smaller than our flan-t5-small model and 40x smaller than the Qwen3-0.6B LoRA setup. It fine-tunes in about &lt;strong&gt;two minutes&lt;/strong&gt; on a laptop.&lt;/p&gt;

&lt;h2&gt;
  
  
  The catch: our data wasn't built for this
&lt;/h2&gt;

&lt;p&gt;Here's the thing nobody tells you when you decide "let's just add a BERT-style token classifier": your training data has to actually support it, and if you've been building a generative-model pipeline for months, it probably doesn't — not in the shape you need.&lt;/p&gt;

&lt;p&gt;Every model in this project so far had been trained the same way: given a raw address string, generate a JSON object mapping 13 field names to substrings of that address. The gold labels were always &lt;strong&gt;verbatim extractions&lt;/strong&gt; — never paraphrased, never normalized. If the source text said &lt;code&gt;"Kamrup Unclassified AS 781029"&lt;/code&gt;, the district field was exactly &lt;code&gt;"Kamrup"&lt;/code&gt;, copied character-for-character, not corrected to &lt;code&gt;"Kamrup Metropolitan"&lt;/code&gt; or expanded to &lt;code&gt;"Assam"&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;Token classification wants something different: a BIO tag (&lt;code&gt;B-district&lt;/code&gt;, &lt;code&gt;I-district&lt;/code&gt;, &lt;code&gt;O&lt;/code&gt;, ...) on every single token. We didn't have that. What we had was JSON.&lt;/p&gt;

&lt;p&gt;The good news is that "verbatim extraction" and "convertible to BIO tags" are almost the same property. If a gold value is a real substring of the raw address, you can find its character span, then map that span onto whatever tokens your tokenizer produces. We measured how often that actually holds:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;exact substring found: 25,910 / 25,915 gold field values (99.98%)
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Nearly perfect. The handful of misses were genuine data artifacts — cases where two fields got glued together during earlier normalization (&lt;code&gt;"PALASHBARI Kamrup"&lt;/code&gt; spanning a comma that got collapsed somewhere upstream). Not something a training script should paper over; we just skip labeling those and move on.&lt;/p&gt;

&lt;h2&gt;
  
  
  The bug that almost shipped: duplicate values
&lt;/h2&gt;

&lt;p&gt;The harder problem showed up once we started converting real examples. Consider an address where both &lt;code&gt;city&lt;/code&gt; and &lt;code&gt;district&lt;/code&gt; are gold-labeled as &lt;code&gt;"Chandigarh"&lt;/code&gt; — genuinely happens, since Chandigarh is its own city and district. A naive &lt;code&gt;raw_text.find("Chandigarh")&lt;/code&gt; always returns the &lt;em&gt;first&lt;/em&gt; occurrence. Both fields collapse onto the same span. One of them silently loses its label.&lt;/p&gt;

&lt;p&gt;We caught this by measuring, not by inspection — we ran a full round-trip test (gold → character spans → BIO tags → reconstructed fields) across the training set and it came back at 93.75% instead of the &amp;gt;99% we expected. Digging into the mismatches surfaced the collision pattern immediately: any two fields sharing a value, wherever the text happened to repeat that value, were fighting over the same characters.&lt;/p&gt;

&lt;p&gt;The fix: track &lt;em&gt;every&lt;/em&gt; occurrence of a value in the text, not just the first, and let the overlap-resolution logic (which already claims longer spans before shorter ones, so &lt;code&gt;"village"&lt;/code&gt; doesn't get clobbered by a &lt;code&gt;"locality"&lt;/code&gt; substring it happens to contain) pick a distinct occurrence for each field. That brought the round-trip ceiling up to &lt;strong&gt;96.68%&lt;/strong&gt; — which we now treat as this data's honest upper bound, not a bug to keep chasing. The residual gap is two well-understood cases: fields sharing a value with too few occurrences to give each one its own span, and tokens that straddle a span boundary on already-documented data artifacts (the same glued-substring issue that shows up elsewhere in this project's known limitations).&lt;/p&gt;

&lt;h2&gt;
  
  
  Training and the surprising result
&lt;/h2&gt;

&lt;p&gt;With the BIO conversion pipeline verified, training itself was almost anticlimactic. Ten epochs, batch size 32, cosine learning rate schedule — done in 133 seconds on Apple Silicon. Eval loss dropped from 1.91 to 0.77 and plateaued cleanly around epoch 8.&lt;/p&gt;

&lt;p&gt;Then evaluation came back and it was, frankly, better than expected:&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Model&lt;/th&gt;
&lt;th&gt;Params&lt;/th&gt;
&lt;th&gt;Mean field accuracy&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;Qwen3-0.6B + LoRA&lt;/td&gt;
&lt;td&gt;~596M&lt;/td&gt;
&lt;td&gt;82.4%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;flan-t5-small&lt;/td&gt;
&lt;td&gt;~77M&lt;/td&gt;
&lt;td&gt;80.6%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;TinyBERT 4L/312D&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;~14M&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;78.8%&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;A model 40x smaller than our best one landed within four points of it. It's not free — &lt;code&gt;subLocality&lt;/code&gt; and &lt;code&gt;village&lt;/code&gt; recall are both effectively 0%, meaning the model just defaults to null on those far more than gold does, a real weakness worth being upfront about. But on the fields that matter most for downstream use (district, state, city, pincode), it's solidly in the same range as its much bigger siblings.&lt;/p&gt;

&lt;h2&gt;
  
  
  The comparison that actually mattered
&lt;/h2&gt;

&lt;p&gt;All of this was interesting on its own, but it wasn't the real test. The real test was going back to the model that started this whole detour: Shiprocket's &lt;code&gt;open-tinybert-indian-address-ner&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;Same name. Same task family — BIO tagging on Indian addresses. This should be the closest thing to an apples-to-apples comparison in the whole project.&lt;/p&gt;

&lt;p&gt;Except it wasn't, and we found that out the moment we checked the config instead of assuming:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight yaml"&gt;&lt;code&gt;&lt;span class="s"&gt;shiprocket-ai/open-tinybert-indian-address-ner&lt;/span&gt;
  &lt;span class="s"&gt;hidden_size&lt;/span&gt;&lt;span class="err"&gt;:&lt;/span&gt; &lt;span class="s"&gt;768, num_hidden_layers&lt;/span&gt;&lt;span class="err"&gt;:&lt;/span&gt; &lt;span class="m"&gt;6&lt;/span&gt;
  &lt;span class="na"&gt;params&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;66,382,103&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Despite the "tinybert" name, Shiprocket's model is a 6-layer, 768-hidden BERT — closer in scale to BERT-base than to the original TinyBERT paper's smallest configuration. It has &lt;strong&gt;4.7x more parameters&lt;/strong&gt; than ours. We reported that up front rather than letting a same-name comparison imply a same-size one.&lt;/p&gt;

&lt;p&gt;With that caveat stated plainly, we ran both models on the same 237-example held-out gold test set:&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Field&lt;/th&gt;
&lt;th&gt;Ours (4L/312D, ~14M)&lt;/th&gt;
&lt;th&gt;Shiprocket (6L/768D, ~66.4M)&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;houseNumber&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;79.8%&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;27.1%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;houseName&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;81.7%&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;72.1%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;street&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;50.0%&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;27.0%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;locality&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;36.5%&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;6.7%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;city&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;82.6%&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;17.4%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;state&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;84.2%&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;41.5%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;pincode&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;99.2%&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;69.2%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;poi&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;20.5%&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;10.3%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;subLocality&lt;/td&gt;
&lt;td&gt;0.0%&lt;/td&gt;
&lt;td&gt;0.0%&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;We won on all nine shared fields. Not narrowly — on &lt;code&gt;city&lt;/code&gt; it's 82.6% vs 17.4%; on &lt;code&gt;houseNumber&lt;/code&gt; it's 79.8% vs 27.1%. And it wasn't slower for the privilege: 11ms/address vs 16ms/address, despite being the smaller model on paper.&lt;/p&gt;

&lt;p&gt;We didn't take the win at face value either. When we inspected Shiprocket's raw, unaggregated per-token predictions, the pattern was clear: on longer administrative-suffix text, the model's tag predictions genuinely flip-flop mid-word, with confidence scores dropping to 0.3–0.5 exactly where that happens. For &lt;code&gt;"Kamrup Unclassified"&lt;/code&gt;, the token &lt;code&gt;"Kam"&lt;/code&gt; gets tagged &lt;code&gt;B-sub_locality&lt;/code&gt; at 0.45 confidence, and the very next token, &lt;code&gt;"rup"&lt;/code&gt;, gets tagged &lt;code&gt;I-locality&lt;/code&gt; at 0.42 — genuinely uncertain, internally inconsistent output, not an artifact of how we ran the comparison.&lt;/p&gt;

&lt;p&gt;Our read on &lt;em&gt;why&lt;/em&gt; the gap is this large: fine-tuning on task-specific gold data seems to matter more here than raw parameter count. Shiprocket's model is bigger, but it wasn't fine-tuned on this exact 13-field taxonomy and this exact address distribution. Ours was — on the same 4,110 verbatim-extraction examples that trained the Qwen3 and flan-t5-small models before it.&lt;/p&gt;

&lt;h2&gt;
  
  
  Where this leaves the project
&lt;/h2&gt;

&lt;p&gt;Three models now sit behind one interface:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight python"&gt;&lt;code&gt;&lt;span class="kn"&gt;from&lt;/span&gt; &lt;span class="n"&gt;indian_address_parser&lt;/span&gt; &lt;span class="kn"&gt;import&lt;/span&gt; &lt;span class="n"&gt;AddressParser&lt;/span&gt;

&lt;span class="n"&gt;parser&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="nc"&gt;AddressParser&lt;/span&gt;&lt;span class="p"&gt;()&lt;/span&gt;                  &lt;span class="c1"&gt;# tinybert — the default now
&lt;/span&gt;&lt;span class="n"&gt;parser&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="nc"&gt;AddressParser&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;backend&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;t5&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;       &lt;span class="c1"&gt;# a couple points more accurate, slower
&lt;/span&gt;&lt;span class="n"&gt;parser&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="nc"&gt;AddressParser&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;backend&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;qwen&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;     &lt;span class="c1"&gt;# the most accurate, and the heaviest
&lt;/span&gt;&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;TinyBERT became the default in &lt;code&gt;v0.3.0&lt;/code&gt; deliberately, not by default-by-omission. It's the cheapest model in the series to download and run — a single forward pass instead of autoregressive generation, no adapter/base-model split to manage — and it gives up only a few points of accuracy to do it. For most people integrating this into a pipeline, that's the right trade to start from; the other two backends are one keyword argument away when the accuracy matters more than the footprint.&lt;/p&gt;

&lt;p&gt;All three models, the benchmark scripts, and the full per-field breakdowns are public:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Code &amp;amp; benchmarks&lt;/strong&gt;: &lt;a href="https://github.com/innerkorehq/indian-address-parser" rel="noopener noreferrer"&gt;github.com/innerkorehq/indian-address-parser&lt;/a&gt;
&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;TinyBERT model&lt;/strong&gt;: &lt;a href="https://huggingface.co/gagan1985/tinybert-4l-312d-indian-address-parser" rel="noopener noreferrer"&gt;huggingface.co/gagan1985/tinybert-4l-312d-indian-address-parser&lt;/a&gt;
&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;PyPI&lt;/strong&gt;: &lt;code&gt;pip install indian-address-parser&lt;/code&gt;
&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;If you want to reproduce the Shiprocket comparison yourself, it's a five-minute run:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;pip &lt;span class="nb"&gt;install &lt;/span&gt;indian-address-parser transformers torch
git clone https://github.com/innerkorehq/indian-address-parser
&lt;span class="nb"&gt;cd &lt;/span&gt;indian-address-parser/benchmarks
python compare_tinybert.py &lt;span class="nt"&gt;--out&lt;/span&gt; results.json
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;






&lt;p&gt;&lt;em&gt;Have you run into the same "same architecture name, different actual size" trap comparing models? Curious what other small-model fine-tunes people have benchmarked against their pretrained namesakes — drop it in the comments.&lt;/em&gt;&lt;/p&gt;

</description>
      <category>machinelearning</category>
      <category>nlp</category>
      <category>opensource</category>
      <category>showdev</category>
    </item>
    <item>
      <title>Building an Open-Source Indian Address Parser: From Raw MCA/Bank Data to a Fine-Tuned LLM</title>
      <dc:creator>Gagandeep Singh</dc:creator>
      <pubDate>Thu, 02 Jul 2026 08:32:43 +0000</pubDate>
      <link>https://dev.to/gagan1985/building-an-open-source-indian-address-parser-from-raw-mcabank-data-to-a-fine-tuned-llm-2n9c</link>
      <guid>https://dev.to/gagan1985/building-an-open-source-indian-address-parser-from-raw-mcabank-data-to-a-fine-tuned-llm-2n9c</guid>
      <description>&lt;p&gt;&lt;em&gt;Cross-posting the full pipeline — data labeling, LoRA fine-tuning, cross-framework conversion, and a benchmark against an existing NER model — because most of the interesting bugs weren't in the ML at all.&lt;/em&gt;&lt;/p&gt;




&lt;h2&gt;
  
  
  The problem
&lt;/h2&gt;

&lt;p&gt;Indian addresses are notoriously unstructured. A single line can look like this:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;FLAT NO.32, UTTARA TOWERS, MG ROAD GUWAHATI , Kamrup Unclassified AS 781029
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;House number, building name, street, locality, district, state, and pincode — all jammed into one free-text string with zero consistent formatting. If you've worked with Indian company registry data, bank KYC records, or delivery logistics, you already know this pain.&lt;/p&gt;

&lt;p&gt;I set out to build something that turns strings like the above into:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight json"&gt;&lt;code&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;"houseNumber"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"FLAT NO.32"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;"houseName"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"UTTARA TOWERS"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;"street"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"MG ROAD"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;"city"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"GUWAHATI"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;"district"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"Kamrup"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;"state"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"AS"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;"pincode"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"781029"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;"poi"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="kc"&gt;null&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="nl"&gt;"subsubLocality"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="kc"&gt;null&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="nl"&gt;"subLocality"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="kc"&gt;null&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;"locality"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="kc"&gt;null&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="nl"&gt;"village"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="kc"&gt;null&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="nl"&gt;"subDistrict"&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="kc"&gt;null&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;13 fields, always present, &lt;code&gt;null&lt;/code&gt; when absent. Here's the whole pipeline, warts included.&lt;/p&gt;

&lt;h2&gt;
  
  
  Getting labeled data without a labeling budget
&lt;/h2&gt;

&lt;p&gt;Starting point: 4.37M raw addresses from two very differently-shaped sources — Indian MCA (Ministry of Corporate Affairs) company registrations, and bank/business-correspondent branch records. No labels.&lt;/p&gt;

&lt;p&gt;Manual labeling doesn't scale to that volume, so the pipeline is layered:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;
&lt;strong&gt;Rule-based tagging&lt;/strong&gt; — regex + gazetteer cross-checks (pincode → district/state lookup from India Post's official pincode CSV) give every record a confidence score. High-confidence ones auto-accept as "silver" labels.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;LLM-assisted labeling for the rest&lt;/strong&gt; — batched calls to an LLM via OpenRouter, with a system prompt that requires every extracted value to be copied &lt;em&gt;verbatim&lt;/em&gt; from the source text. If the model's field value isn't a substring of the input, it gets dropped rather than trusted. This alone eliminates a whole class of hallucination.&lt;/li&gt;
&lt;li&gt;A small &lt;strong&gt;human-reviewed slice&lt;/strong&gt; as a sanity check against the LLM's own accuracy before scaling up.&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;One subtlety that actually mattered: MCA addresses have a machine-generated tail like &lt;code&gt;"...Kamrup Unclassified AS 781029"&lt;/code&gt;, where &lt;code&gt;"Unclassified"&lt;/code&gt; is a fixed placeholder meaning "no sub-district classification recorded" — not a place name. Early runs had the LLM tagging &lt;code&gt;"Unclassified"&lt;/code&gt; as a &lt;code&gt;subDistrict&lt;/code&gt; value. Fixed by explicitly teaching the model about this convention in the prompt. Small thing, but it's the kind of domain quirk no generic address parser would know to avoid.&lt;/p&gt;

&lt;p&gt;Also worth calling out: &lt;strong&gt;field taxonomy design is harder than model training&lt;/strong&gt;. The first schema (Google Maps' full geocoding component taxonomy, 35 types) was too granular for anyone — human or LLM — to label consistently. Collapsed it to 13 fields based on what a human reviewer could actually apply without agonizing over edge cases.&lt;/p&gt;

&lt;h2&gt;
  
  
  Fine-tuning
&lt;/h2&gt;

&lt;p&gt;LoRA on &lt;code&gt;Qwen/Qwen3-0.6B&lt;/code&gt;, trained via MLX on an M4 Mac (&lt;code&gt;mlx-lm&lt;/code&gt;'s &lt;code&gt;lora&lt;/code&gt; command — genuinely pleasant to work with on Apple Silicon, no CUDA/bitsandbytes wrangling).&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight properties"&gt;&lt;code&gt;&lt;span class="py"&gt;rank&lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="s"&gt;16, alpha=32, dropout=0.05&lt;/span&gt;
&lt;span class="py"&gt;target_modules&lt;/span&gt;&lt;span class="p"&gt;:&lt;/span&gt; &lt;span class="s"&gt;q_proj, k_proj, v_proj, o_proj, gate_proj, up_proj, down_proj&lt;/span&gt;
&lt;span class="err"&gt;16&lt;/span&gt; &lt;span class="err"&gt;of&lt;/span&gt; &lt;span class="err"&gt;28&lt;/span&gt; &lt;span class="err"&gt;layers&lt;/span&gt; &lt;span class="err"&gt;fine-tuned,&lt;/span&gt; &lt;span class="err"&gt;2000&lt;/span&gt; &lt;span class="err"&gt;iterations,&lt;/span&gt; &lt;span class="err"&gt;~1.8&lt;/span&gt; &lt;span class="err"&gt;hours&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;&lt;strong&gt;Results on a 237-example held-out gold test set:&lt;/strong&gt;&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Metric&lt;/th&gt;
&lt;th&gt;Value&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;JSON parse rate&lt;/td&gt;
&lt;td&gt;100%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Mean per-field accuracy&lt;/td&gt;
&lt;td&gt;82.4%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Overall exact match (all fields)&lt;/td&gt;
&lt;td&gt;30.8%&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;The gap between per-field accuracy and exact-match is the interesting bit. Digging into disagreements, most of it isn't the model being wrong — it's &lt;strong&gt;schema ambiguity&lt;/strong&gt;. &lt;code&gt;locality&lt;/code&gt;/&lt;code&gt;subLocality&lt;/code&gt;/&lt;code&gt;subsubLocality&lt;/code&gt;/&lt;code&gt;village&lt;/code&gt; represent the same "named area, different granularity" concept, and even the gold labels are sometimes inconsistent about which bucket a given place name belongs in (I found gold records where the &lt;em&gt;same string&lt;/em&gt; was labeled as both &lt;code&gt;locality&lt;/code&gt; and &lt;code&gt;village&lt;/code&gt; simultaneously). That's a taxonomy problem, not a model problem, and no amount of additional training fixes it without a firmer labeling convention.&lt;/p&gt;

&lt;h2&gt;
  
  
  Getting it to run outside MLX
&lt;/h2&gt;

&lt;p&gt;This is where most of the actual debugging time went, and none of it was ML.&lt;/p&gt;

&lt;p&gt;&lt;code&gt;mlx-lm&lt;/code&gt; produces its own adapter format — not PEFT-compatible. To make the model usable on CUDA/CPU (not just Apple Silicon), I had to hand-derive the weight conversion:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight python"&gt;&lt;code&gt;&lt;span class="c1"&gt;# mlx-lm: lora_a [in_features, r], lora_b [r, out_features], used as x @ A @ B
# PEFT:   lora_A.weight [r, in_features], lora_B.weight [out_features, r]
# So: peft_A = mlx_a.T, peft_B = mlx_b.T
&lt;/span&gt;&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;I verified this against &lt;code&gt;mlx-lm&lt;/code&gt;'s own &lt;code&gt;fuse()&lt;/code&gt; source (&lt;code&gt;delta = (scale * lora_b.T) @ lora_a.T&lt;/code&gt;) rather than trusting my own derivation, then confirmed numerically — ran the same 15 addresses through both the original MLX adapter and the converted PEFT version. 13/15 identical outputs; the 2 mismatches landed exactly on the already-known-ambiguous fields, consistent with floating-point differences between backends on a near-tied softmax decision rather than a conversion bug.&lt;/p&gt;

&lt;h2&gt;
  
  
  Publishing, and the dependency-floor whack-a-mole
&lt;/h2&gt;

&lt;p&gt;Published the model to Hugging Face (both formats — PEFT at root, MLX in a subfolder), then wrapped it as a &lt;code&gt;pip install&lt;/code&gt;-able package: &lt;a href="https://pypi.org/project/indian-address-parser/" rel="noopener noreferrer"&gt;&lt;code&gt;indian-address-parser&lt;/code&gt;&lt;/a&gt; on PyPI, source on &lt;a href="https://github.com/innerkorehq/indian-address-parser" rel="noopener noreferrer"&gt;GitHub&lt;/a&gt;.&lt;/p&gt;

&lt;p&gt;Then real users tried to install it into their existing environments (Anaconda base envs, specifically), and things broke in sequence:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;&lt;code&gt;peft&lt;/code&gt; imports &lt;code&gt;transformers.BloomPreTrainedModel&lt;/code&gt;&lt;/strong&gt;, whose lazy-loading chain unconditionally does &lt;code&gt;import tensorflow&lt;/code&gt;. In a conda env with a mismatched TF/numpy/h5py install, that crashed the whole thing before ever touching TensorFlow functionality. Fix: &lt;code&gt;os.environ["USE_TF"] = "0"&lt;/code&gt; before any transformers/peft import, so transformers' TF-detection short-circuits.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;&lt;code&gt;qwen3&lt;/code&gt; model type not recognized.&lt;/strong&gt; Turns out &lt;code&gt;transformers&lt;/code&gt; only added Qwen3 support at exactly version &lt;code&gt;4.51.0&lt;/code&gt; — verified by bisecting real PyPI releases (&lt;code&gt;4.50.0&lt;/code&gt;: no, &lt;code&gt;4.51.0&lt;/code&gt;: yes). My dependency floor (&lt;code&gt;&amp;gt;=4.45.0&lt;/code&gt;) was loose enough that pip left an old transformers in place instead of upgrading it.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;&lt;code&gt;hf_hub_download() got an unexpected keyword argument 'use_auth_token'&lt;/code&gt;.&lt;/strong&gt; &lt;code&gt;peft&amp;lt;0.18.0&lt;/code&gt; unconditionally passes &lt;code&gt;use_auth_token=None&lt;/code&gt; into &lt;code&gt;hf_hub_download&lt;/code&gt;, regardless of whether the caller asked for it. Recent &lt;code&gt;huggingface_hub&lt;/code&gt; (1.x) dropped that deprecated kwarg entirely. Bisected peft's source across ten versions to find the exact fix boundary (0.17.1: unconditional pass, 0.18.0: conditional via walrus operator).&lt;/p&gt;&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;Each fix was verified against the &lt;em&gt;actual reported failure&lt;/em&gt;, not just plausible-sounding — I built a venv pinned to the exact stale dependency trio from the bug report, installed the patched package, confirmed pip auto-upgraded everything, and ran real inference before calling it fixed.&lt;/p&gt;

&lt;p&gt;The lesson, if there is one: &lt;strong&gt;&lt;code&gt;&amp;gt;=X.Y.Z&lt;/code&gt; floors need to be the actual minimum that works, verified, not "whatever I happened to have installed while developing."&lt;/strong&gt; Loose floors don't fail for you — they fail for whoever has an older version already sitting in their environment.&lt;/p&gt;

&lt;h2&gt;
  
  
  Benchmarking against an existing model
&lt;/h2&gt;

&lt;p&gt;Once things were stable, I compared against &lt;a href="https://huggingface.co/shiprocket-ai/open-tinybert-indian-address-ner" rel="noopener noreferrer"&gt;Shiprocket's &lt;code&gt;open-tinybert-indian-address-ner&lt;/code&gt;&lt;/a&gt; — a 6-layer TinyBERT doing BIO-tagged token classification, a fundamentally different architecture (and a different field taxonomy) than a 0.6B causal LM generating JSON.&lt;/p&gt;

&lt;p&gt;Built an explicit field mapping covering the 9 conceptually-overlapping fields (their &lt;code&gt;house_details&lt;/code&gt; ↔ my &lt;code&gt;houseNumber&lt;/code&gt;, &lt;code&gt;road&lt;/code&gt; ↔ &lt;code&gt;street&lt;/code&gt;, etc.) and scored both against the same 237-example held-out set:&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Field&lt;/th&gt;
&lt;th&gt;Mine&lt;/th&gt;
&lt;th&gt;Shiprocket's&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;city&lt;/td&gt;
&lt;td&gt;91.3%&lt;/td&gt;
&lt;td&gt;17.4%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;state&lt;/td&gt;
&lt;td&gt;96.2%&lt;/td&gt;
&lt;td&gt;41.5%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;pincode&lt;/td&gt;
&lt;td&gt;100.0%&lt;/td&gt;
&lt;td&gt;69.2%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;houseNumber&lt;/td&gt;
&lt;td&gt;84.5%&lt;/td&gt;
&lt;td&gt;27.1%&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;Higher accuracy on every shared field — but Shiprocket's model is &lt;strong&gt;~240x faster per address&lt;/strong&gt; (19ms vs 4.6s). That's not a quality artifact, it's architecture: a 6-layer classifier doing a single forward pass vs. autoregressive generation. If your use case needs high-throughput/low-latency parsing over perfect accuracy, that's a legitimate reason to pick the other model. I'd rather publish that tradeoff honestly than pretend the comparison only cuts one way.&lt;/p&gt;

&lt;h2&gt;
  
  
  Publishing the data too
&lt;/h2&gt;

&lt;p&gt;Also shipped the underlying data as two HF datasets:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;a href="https://huggingface.co/datasets/gagan1985/indian-addresses-raw" rel="noopener noreferrer"&gt;&lt;code&gt;indian-addresses-raw&lt;/code&gt;&lt;/a&gt; — the full 4.37M-record unlabeled corpus&lt;/li&gt;
&lt;li&gt;
&lt;a href="https://huggingface.co/datasets/gagan1985/indian-addresses-gold" rel="noopener noreferrer"&gt;&lt;code&gt;indian-addresses-gold&lt;/code&gt;&lt;/a&gt; — 4,834 span-labeled training examples&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Before publishing the raw corpus, I found something worth mentioning: bank/BC address records are KYC-style data and some of them embed real customer phone numbers and relational-name markers (&lt;code&gt;S/O&lt;/code&gt;/&lt;code&gt;D/O&lt;/code&gt;/&lt;code&gt;W/O&lt;/code&gt;/&lt;code&gt;C/O&lt;/code&gt; — "son of"/"care of", standard on Indian address forms). That's different from MCA's superficially similar &lt;code&gt;C/O &amp;lt;company director&amp;gt;&lt;/code&gt; convention, which is already public disclosure. Wrote a targeted redaction pass for the bank source (verified against the corpus, not assumed — caught a "Door No." vs "D/O [name]" false-positive collision along the way), and for the gold dataset specifically, &lt;strong&gt;dropped&lt;/strong&gt; the small number of affected records instead of redacting in place, since redacting text shifts the character offsets that the span labels depend on.&lt;/p&gt;

&lt;h2&gt;
  
  
  Try it
&lt;/h2&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;pip &lt;span class="nb"&gt;install &lt;/span&gt;indian-address-parser
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;





&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight python"&gt;&lt;code&gt;&lt;span class="kn"&gt;from&lt;/span&gt; &lt;span class="n"&gt;indian_address_parser&lt;/span&gt; &lt;span class="kn"&gt;import&lt;/span&gt; &lt;span class="n"&gt;AddressParser&lt;/span&gt;

&lt;span class="n"&gt;parser&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="nc"&gt;AddressParser&lt;/span&gt;&lt;span class="p"&gt;()&lt;/span&gt;  &lt;span class="c1"&gt;# pulls weights from HF automatically
&lt;/span&gt;&lt;span class="n"&gt;parser&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="nf"&gt;parse&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="s"&gt;FLAT NO.32, UTTARA TOWERS, MG ROAD GUWAHATI , Kamrup Unclassified AS 781029&lt;/span&gt;&lt;span class="sh"&gt;"&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Everything's open source and Apache 2.0: &lt;a href="https://huggingface.co/gagan1985/qwen3-0.6b-indian-address-parser" rel="noopener noreferrer"&gt;model&lt;/a&gt; · &lt;a href="https://github.com/innerkorehq/indian-address-parser" rel="noopener noreferrer"&gt;GitHub&lt;/a&gt; · &lt;a href="https://pypi.org/project/indian-address-parser/" rel="noopener noreferrer"&gt;PyPI&lt;/a&gt; · &lt;a href="https://huggingface.co/datasets/gagan1985/indian-addresses-gold" rel="noopener noreferrer"&gt;datasets&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Feedback and PRs welcome, especially on the locality/subLocality boundary ambiguity — I have a hypothesis for a firmer labeling convention that would help, but haven't tested whether it actually resolves the disagreement rate or just moves it around.&lt;/p&gt;

</description>
      <category>python</category>
      <category>huggingface</category>
      <category>ai</category>
      <category>opensource</category>
    </item>
  </channel>
</rss>
