
  <rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
    <channel>
      <title>Nimbus Systems</title>
      <link>https://nimbusarchitect.com/blog</link>
      <description>Solutions Architect at AWS. Writing about AI/ML, cloud architecture, cybersecurity, and quantitative research.</description>
      <language>en-us</language>
      <managingEditor>lilphd99@gmail.com (Benjamin Lee)</managingEditor>
      <webMaster>lilphd99@gmail.com (Benjamin Lee)</webMaster>
      <lastBuildDate>Sun, 21 Jun 2026 00:00:00 GMT</lastBuildDate>
      <atom:link href="https://nimbusarchitect.com/tags/nlp/feed.xml" rel="self" type="application/rss+xml"/>
      
  <item>
    <guid>https://nimbusarchitect.com/blog/tokenization-the-encoding-layer-that-decides-how-models-learn</guid>
    <title>A Token Effort</title>
    <link>https://nimbusarchitect.com/blog/tokenization-the-encoding-layer-that-decides-how-models-learn</link>
    <description>The humble tokenizer decides what a model can ever learn. Why BPE, WordPiece and Unigram differ, and how to build, test and monitor one properly.</description>
    <pubDate>Sun, 21 Jun 2026 00:00:00 GMT</pubDate>
    <author>lilphd99@gmail.com (Benjamin Lee)</author>
    <category>tokenization</category><category>llm</category><category>ai</category><category>ml-systems</category><category>nlp</category><category>pretraining</category>
  </item>

    </channel>
  </rss>
