<?xml version="1.0" encoding="UTF-8" ?>
<?xml-stylesheet href="/static/feed.xsl" type="text/xsl" ?>
<feed xmlns="http://www.w3.org/2005/Atom" xmlns:quartz="https://quartz.jzhao.xyz/ns">
  <script xmlns="http://www.w3.org/1999/xhtml" src="/static/scripts/xslt-polyfill-214fd525.js"></script>
  <title>/wiki / imgen</title>
  <subtitle>Recent notes in /wiki / imgen on Animesh Mishra</subtitle>
  <link href="https://animishraa05.github.io/wiki/imgen" />
  <link rel="alternate" type="text/html" href="https://animishraa05.github.io/wiki/imgen" />
  <category term="wiki/imgen" />
  <id>https://animishraa05.github.io/wiki/imgen</id>
  <updated>2026-09-02T00:00:00.000Z</updated>
  <contributor>
    <name>Animesh Mishra</name>
    <email>animesh.mishra818@gmail.com</email>
  </contributor>
  <logo>https://animishraa05.github.io/icon.png</logo>
  <icon>https://animishraa05.github.io/icon.png</icon>
  <generator>Quartz v4.6.0 -- quartz.jzhao.xyz</generator>
  <rights type="html">&amp;amp;copy; 2026 Animesh Mishra</rights>
  
  <entry>
    <title>Imgen — Map of Content</title>
    <link href="https://animishraa05.github.io/wiki/imgen/imgen-moc" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/imgen-moc.md" />
    <summary> Map of Content for Imgen — 13 concepts. Start here to navigate imgen.</summary>
    <published>2026-09-02T00:00:00.000Z</published>
    <updated>2026-09-02T00:00:00.000Z</updated>
    <publishedTime>Sep 02, 2026</publishedTime>
    <updatedTime>Sep 02, 2026</updatedTime>
    <category term="meta" label="meta" />
<category term="imgen" label="imgen" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;blockquote&gt;
&lt;p dir=&quot;auto&quot;&gt;Map of Content for &lt;strong&gt;Imgen&lt;/strong&gt; — 13 concepts. Start here to navigate imgen.&lt;/p&gt;
&lt;/blockquote&gt;
&lt;h2 id=&quot;core-concepts&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Concepts&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-concepts&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;a href=&quot;../../clip&quot; class=&quot;internal&quot; data-slug=&quot;clip&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;clip&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../controlnet&quot; class=&quot;internal&quot; data-slug=&quot;controlnet&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;controlnet&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../diffusion-models&quot; class=&quot;internal&quot; data-slug=&quot;diffusion-models&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;diffusion-models&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;flux-architecture&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../glyph-injection&quot; class=&quot;internal&quot; data-slug=&quot;glyph-injection&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;glyph-injection&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../hybrid-pipeline&quot; class=&quot;internal&quot; data-slug=&quot;hybrid-pipeline&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;hybrid-pipeline&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../lora-finetuning&quot; class=&quot;internal&quot; data-slug=&quot;lora-finetuning&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;lora-finetuning&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../neural-networks&quot; class=&quot;internal alias&quot; data-slug=&quot;neural-networks&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;Neural Networks&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../t5-encoder&quot; class=&quot;internal&quot; data-slug=&quot;t5-encoder&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;t5-encoder&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../text-rendering-problem&quot; class=&quot;internal&quot; data-slug=&quot;text-rendering-problem&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;text-rendering-problem&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../text-rendering-solutions&quot; class=&quot;internal&quot; data-slug=&quot;text-rendering-solutions&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;text-rendering-solutions&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../transformers&quot; class=&quot;internal alias&quot; data-slug=&quot;transformers&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;Transformers&lt;/a&gt; — concept&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../vae&quot; class=&quot;internal&quot; data-slug=&quot;vae&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;vae&lt;/a&gt; — concept&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;mechanisms--how-things-work&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Mechanisms &amp;#x26; How Things Work&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#mechanisms--how-things-work&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;&lt;em&gt;Auto-generated — edit to curate. Pages that explain processes/flows.&lt;/em&gt;&lt;/p&gt;
&lt;h2 id=&quot;comparisons--tradeoffs&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Comparisons &amp;#x26; Tradeoffs&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#comparisons--tradeoffs&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;em&gt;(No synthesis yet — create via ingest or query —save)&lt;/em&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;suggested-reading-order&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Suggested Reading Order&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#suggested-reading-order&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ol dir=&quot;auto&quot;&gt;
&lt;li&gt;Browse Core Concepts above — start with most-linked page&lt;/li&gt;
&lt;li&gt;Follow Connections sections for dense graph traversal&lt;/li&gt;
&lt;/ol&gt;</content>
  </entry><entry>
    <title>neural-networks</title>
    <link href="https://animishraa05.github.io/wiki/imgen/neural-networks" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/neural-networks.md" />
    <summary>The Problem Traditional AI used symbolic, rule-based approaches that couldn’t handle perception tasks (image recognition, speech) or learn from examples.</summary>
    <published>2026-04-29T00:00:00.000Z</published>
    <updated>2026-04-29T00:00:00.000Z</updated>
    <publishedTime>Apr 29, 2026</publishedTime>
    <updatedTime>Apr 29, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="ml" label="ml" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Traditional AI used symbolic, rule-based approaches that couldn’t handle perception tasks (image recognition, speech) or learn from examples. There was no system that could learn patterns from data the way biological brains do.&lt;/p&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Neural networks are computing systems inspired by biological neurons. They learn by adjusting connection weights between layers of artificial neurons, enabling pattern recognition, classification, and generation tasks.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;A neural network consists of:&lt;/p&gt;
&lt;ol dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;Input layer&lt;/strong&gt; — receives raw data (pixels, words, features)&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Hidden layers&lt;/strong&gt; — transform data through weighted connections and activation functions&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Output layer&lt;/strong&gt; — produces predictions or generated content&lt;/li&gt;
&lt;/ol&gt;
&lt;p dir=&quot;auto&quot;&gt;Training uses &lt;strong&gt;backpropagation&lt;/strong&gt;: forward pass computes output, loss function measures error, backward pass adjusts weights via gradient descent.&lt;/p&gt;
&lt;p dir=&quot;auto&quot;&gt;Modern networks like Transformers, CNNs, and Diffusion models are all neural networks with specialized architectures.&lt;/p&gt;
&lt;h2 id=&quot;visual-explanation&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Visual Explanation&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#visual-explanation&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;div class=&quot;graphviz-container&quot; style=&quot;display: flex; justify-content: center; margin: 1.5rem 0; overflow-x: auto;&quot; dir=&quot;auto&quot;&gt;&lt;!--?xml version=&quot;1.0&quot; encoding=&quot;UTF-8&quot; standalone=&quot;no&quot;?--&gt;

&lt;!-- Generated by graphviz version 15.1.0 (0)
 --&gt;
&lt;!-- Title: G Pages: 1 --&gt;
&lt;svg width=&quot;204pt&quot; height=&quot;226pt&quot; viewBox=&quot;0.00 0.00 204.00 226.00&quot; xmlns=&quot;http://www.w3.org/2000/svg&quot; xmlns:xlink=&quot;http://www.w3.org/1999/xlink&quot;&gt;
&lt;g id=&quot;graph0&quot; class=&quot;graph&quot; transform=&quot;scale(1 1) rotate(0) translate(4 222.39)&quot;&gt;
&lt;title&gt;G&lt;/title&gt;
&lt;polygon fill=&quot;white&quot; stroke=&quot;none&quot; points=&quot;-4,4 -4,-222.39 200.16,-222.39 200.16,4 -4,4&quot;&gt;&lt;/polygon&gt;
&lt;!-- x1 --&gt;
&lt;g id=&quot;node1&quot; class=&quot;node&quot;&gt;
&lt;title&gt;x1&lt;/title&gt;
&lt;ellipse fill=&quot;lightblue&quot; stroke=&quot;black&quot; cx=&quot;20.69&quot; cy=&quot;-197.69&quot; rx=&quot;20.69&quot; ry=&quot;20.69&quot;&gt;&lt;/ellipse&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;20.69&quot; y=&quot;-193.49&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;x1&lt;/text&gt;
&lt;/g&gt;
&lt;!-- h1 --&gt;
&lt;g id=&quot;node4&quot; class=&quot;node&quot;&gt;
&lt;title&gt;h1&lt;/title&gt;
&lt;ellipse fill=&quot;lightblue&quot; stroke=&quot;black&quot; cx=&quot;98.08&quot; cy=&quot;-197.69&quot; rx=&quot;20.69&quot; ry=&quot;20.69&quot;&gt;&lt;/ellipse&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;98.08&quot; y=&quot;-193.49&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;h1&lt;/text&gt;
&lt;/g&gt;
&lt;!-- x1&amp;#45;&amp;gt;h1 --&gt;
&lt;g id=&quot;edge1&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;x1-&gt;h1&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M41.5,-197.69C48.89,-197.69 57.47,-197.69 65.64,-197.69&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;65.63,-201.19 75.63,-197.69 65.63,-194.19 65.63,-201.19&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- h2 --&gt;
&lt;g id=&quot;node5&quot; class=&quot;node&quot;&gt;
&lt;title&gt;h2&lt;/title&gt;
&lt;ellipse fill=&quot;lightblue&quot; stroke=&quot;black&quot; cx=&quot;98.08&quot; cy=&quot;-138.69&quot; rx=&quot;20.69&quot; ry=&quot;20.69&quot;&gt;&lt;/ellipse&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;98.08&quot; y=&quot;-134.49&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;h2&lt;/text&gt;
&lt;/g&gt;
&lt;!-- x1&amp;#45;&amp;gt;h2 --&gt;
&lt;g id=&quot;edge2&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;x1-&gt;h2&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M37.68,-185.18C47.68,-177.36 60.76,-167.12 72.1,-158.24&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;74.21,-161.04 79.92,-152.12 69.89,-155.53 74.21,-161.04&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- h3 --&gt;
&lt;g id=&quot;node6&quot; class=&quot;node&quot;&gt;
&lt;title&gt;h3&lt;/title&gt;
&lt;ellipse fill=&quot;lightblue&quot; stroke=&quot;black&quot; cx=&quot;98.08&quot; cy=&quot;-79.69&quot; rx=&quot;20.69&quot; ry=&quot;20.69&quot;&gt;&lt;/ellipse&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;98.08&quot; y=&quot;-75.49&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;h3&lt;/text&gt;
&lt;/g&gt;
&lt;!-- x1&amp;#45;&amp;gt;h3 --&gt;
&lt;g id=&quot;edge3&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;x1-&gt;h3&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M33.31,-180.93C36,-176.95 38.83,-172.69 41.39,-168.69 58.13,-142.49 60.64,-134.9 77.39,-108.69 77.95,-107.82 78.52,-106.93 79.1,-106.04&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;82.01,-107.98 84.63,-97.72 76.18,-104.11 82.01,-107.98&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- x2 --&gt;
&lt;g id=&quot;node2&quot; class=&quot;node&quot;&gt;
&lt;title&gt;x2&lt;/title&gt;
&lt;ellipse fill=&quot;lightblue&quot; stroke=&quot;black&quot; cx=&quot;20.69&quot; cy=&quot;-138.69&quot; rx=&quot;20.69&quot; ry=&quot;20.69&quot;&gt;&lt;/ellipse&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;20.69&quot; y=&quot;-134.49&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;x2&lt;/text&gt;
&lt;/g&gt;
&lt;!-- x2&amp;#45;&amp;gt;h1 --&gt;
&lt;g id=&quot;edge4&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;x2-&gt;h1&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M37.68,-151.21C47.68,-159.03 60.76,-169.27 72.1,-178.15&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;69.89,-180.86 79.92,-184.27 74.21,-175.35 69.89,-180.86&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- x2&amp;#45;&amp;gt;h2 --&gt;
&lt;g id=&quot;edge5&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;x2-&gt;h2&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M41.5,-138.69C48.89,-138.69 57.47,-138.69 65.64,-138.69&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;65.63,-142.19 75.63,-138.69 65.63,-135.19 65.63,-142.19&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- h4 --&gt;
&lt;g id=&quot;node7&quot; class=&quot;node&quot;&gt;
&lt;title&gt;h4&lt;/title&gt;
&lt;ellipse fill=&quot;lightblue&quot; stroke=&quot;black&quot; cx=&quot;98.08&quot; cy=&quot;-20.69&quot; rx=&quot;20.69&quot; ry=&quot;20.69&quot;&gt;&lt;/ellipse&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;98.08&quot; y=&quot;-16.49&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;h4&lt;/text&gt;
&lt;/g&gt;
&lt;!-- x2&amp;#45;&amp;gt;h4 --&gt;
&lt;g id=&quot;edge6&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;x2-&gt;h4&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M33.31,-121.93C36,-117.95 38.83,-113.69 41.39,-109.69 58.13,-83.49 60.64,-75.9 77.39,-49.69 77.95,-48.82 78.52,-47.93 79.1,-47.04&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;82.01,-48.98 84.63,-38.72 76.18,-45.11 82.01,-48.98&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- x3 --&gt;
&lt;g id=&quot;node3&quot; class=&quot;node&quot;&gt;
&lt;title&gt;x3&lt;/title&gt;
&lt;ellipse fill=&quot;lightblue&quot; stroke=&quot;black&quot; cx=&quot;20.69&quot; cy=&quot;-79.69&quot; rx=&quot;20.69&quot; ry=&quot;20.69&quot;&gt;&lt;/ellipse&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;20.69&quot; y=&quot;-75.49&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;x3&lt;/text&gt;
&lt;/g&gt;
&lt;!-- x3&amp;#45;&amp;gt;h2 --&gt;
&lt;g id=&quot;edge7&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;x3-&gt;h2&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M37.68,-92.21C47.68,-100.03 60.76,-110.27 72.1,-119.15&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;69.89,-121.86 79.92,-125.27 74.21,-116.35 69.89,-121.86&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- x3&amp;#45;&amp;gt;h3 --&gt;
&lt;g id=&quot;edge8&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;x3-&gt;h3&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M41.5,-79.69C48.89,-79.69 57.47,-79.69 65.64,-79.69&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;65.63,-83.19 75.63,-79.69 65.63,-76.19 65.63,-83.19&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- x3&amp;#45;&amp;gt;h4 --&gt;
&lt;g id=&quot;edge9&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;x3-&gt;h4&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M37.68,-67.18C47.68,-59.36 60.76,-49.12 72.1,-40.24&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;74.21,-43.04 79.92,-34.12 69.89,-37.53 74.21,-43.04&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- y1 --&gt;
&lt;g id=&quot;node8&quot; class=&quot;node&quot;&gt;
&lt;title&gt;y1&lt;/title&gt;
&lt;ellipse fill=&quot;lightblue&quot; stroke=&quot;black&quot; cx=&quot;175.47&quot; cy=&quot;-167.69&quot; rx=&quot;20.69&quot; ry=&quot;20.69&quot;&gt;&lt;/ellipse&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;175.47&quot; y=&quot;-163.49&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;y1&lt;/text&gt;
&lt;/g&gt;
&lt;!-- h1&amp;#45;&amp;gt;y1 --&gt;
&lt;g id=&quot;edge10&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;h1-&gt;y1&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M117.72,-190.28C126.05,-186.96 136.08,-182.97 145.36,-179.27&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;146.38,-182.64 154.38,-175.69 143.79,-176.13 146.38,-182.64&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- h2&amp;#45;&amp;gt;y1 --&gt;
&lt;g id=&quot;edge11&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;h2-&gt;y1&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M117.72,-145.86C125.96,-149.03 135.86,-152.84 145.06,-156.38&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;143.78,-159.64 154.37,-159.96 146.3,-153.11 143.78,-159.64&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- y2 --&gt;
&lt;g id=&quot;node9&quot; class=&quot;node&quot;&gt;
&lt;title&gt;y2&lt;/title&gt;
&lt;ellipse fill=&quot;lightblue&quot; stroke=&quot;black&quot; cx=&quot;175.47&quot; cy=&quot;-50.69&quot; rx=&quot;20.69&quot; ry=&quot;20.69&quot;&gt;&lt;/ellipse&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;175.47&quot; y=&quot;-46.49&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;y2&lt;/text&gt;
&lt;/g&gt;
&lt;!-- h3&amp;#45;&amp;gt;y2 --&gt;
&lt;g id=&quot;edge12&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;h3-&gt;y2&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M117.72,-72.52C125.96,-69.35 135.86,-65.55 145.06,-62.01&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;146.3,-65.28 154.37,-58.42 143.78,-58.75 146.3,-65.28&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- h4&amp;#45;&amp;gt;y2 --&gt;
&lt;g id=&quot;edge13&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;h4-&gt;y2&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M117.72,-28.11C126.05,-31.43 136.08,-35.42 145.36,-39.11&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;143.79,-42.25 154.38,-42.7 146.38,-35.75 143.79,-42.25&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;/g&gt;
&lt;/svg&gt;
&lt;/div&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;Universal approximation&lt;/strong&gt; — can approximate any continuous function given enough neurons&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Learning from examples&lt;/strong&gt; — no need for hand-coded rules&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Parallel processing&lt;/strong&gt; — many computations happen simultaneously&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Generalization&lt;/strong&gt; — can make predictions on unseen data&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../transformers&quot; class=&quot;internal alias&quot; data-slug=&quot;transformers&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;Transformers&lt;/a&gt; — specific neural network architecture for sequence modeling&lt;/li&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../diffusion-models&quot; class=&quot;internal alias&quot; data-slug=&quot;diffusion-models&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;Diffusion Models&lt;/a&gt; — neural networks trained to reverse noise processes&lt;/li&gt;
&lt;li&gt;Contrasts with: &lt;a href=&quot;../../symbolic-ai&quot; class=&quot;internal alias&quot; data-slug=&quot;symbolic-ai&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;Symbolic AI&lt;/a&gt; — rule-based vs learning-based approaches&lt;/li&gt;
&lt;li&gt;Related: &lt;a href=&quot;../../clip&quot; class=&quot;internal alias&quot; data-slug=&quot;clip&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;CLIP Encoder&lt;/a&gt; — neural network for text-image alignment&lt;/li&gt;
&lt;li&gt;Related: &lt;a href=&quot;../../t5-encoder&quot; class=&quot;internal alias&quot; data-slug=&quot;t5-encoder&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;T5 Encoder&lt;/a&gt; — transformer-based neural network for text encoding&lt;/li&gt;
&lt;li&gt;Related: &lt;a href=&quot;../../lora-finetuning&quot; class=&quot;internal alias&quot; data-slug=&quot;lora-finetuning&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;LoRA Fine-tuning&lt;/a&gt; — efficient neural network adaptation technique&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;Overfitting&lt;/strong&gt; — memorizing training data instead of learning general patterns&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Black box&lt;/strong&gt; — hard to interpret why a neural network makes a specific decision&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Data hungry&lt;/strong&gt; — need large datasets and significant compute to train effectively&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Adversarial examples&lt;/strong&gt; — small perturbations can fool networks into wrong predictions&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>transformers</title>
    <link href="https://animishraa05.github.io/wiki/imgen/transformers" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/transformers.md" />
    <summary>The Problem Recurrent Neural Networks (RNNs) and LSTMs processed sequences sequentially, making them slow to train and unable to capture long-range dependencies effectively.</summary>
    <published>2026-04-29T00:00:00.000Z</published>
    <updated>2026-04-29T00:00:00.000Z</updated>
    <publishedTime>Apr 29, 2026</publishedTime>
    <updatedTime>Apr 29, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="ml" label="ml" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Recurrent Neural Networks (RNNs) and LSTMs processed sequences sequentially, making them slow to train and unable to capture long-range dependencies effectively. There was no parallelizable architecture for sequence modeling.&lt;/p&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Transformers use self-attention mechanisms to process all tokens in a sequence simultaneously, enabling parallel training and capturing relationships between any two tokens regardless of distance. They became the foundation of modern LLMs.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;The Transformer architecture consists of:&lt;/p&gt;
&lt;ol dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;Self-attention&lt;/strong&gt; — each token attends to all other tokens, computing weighted representations&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Multi-head attention&lt;/strong&gt; — multiple attention mechanisms run in parallel for different relationship types&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Feed-forward networks&lt;/strong&gt; — process each token’s representation independently&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Positional encoding&lt;/strong&gt; — adds position information since there’s no recurrence&lt;/li&gt;
&lt;/ol&gt;
&lt;p dir=&quot;auto&quot;&gt;The architecture has an encoder (for understanding) and decoder (for generation). Decoder-only models (like GPT) use only the decoder stack.&lt;/p&gt;
&lt;h2 id=&quot;visual-explanation&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Visual Explanation&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#visual-explanation&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;div class=&quot;graphviz-container&quot; style=&quot;display: flex; justify-content: center; margin: 1.5rem 0; overflow-x: auto;&quot; dir=&quot;auto&quot;&gt;&lt;!--?xml version=&quot;1.0&quot; encoding=&quot;UTF-8&quot; standalone=&quot;no&quot;?--&gt;

&lt;!-- Generated by graphviz version 15.1.0 (0)
 --&gt;
&lt;!-- Title: G Pages: 1 --&gt;
&lt;svg width=&quot;197pt&quot; height=&quot;404pt&quot; viewBox=&quot;0.00 0.00 197.00 404.00&quot; xmlns=&quot;http://www.w3.org/2000/svg&quot; xmlns:xlink=&quot;http://www.w3.org/1999/xlink&quot;&gt;
&lt;g id=&quot;graph0&quot; class=&quot;graph&quot; transform=&quot;scale(1 1) rotate(0) translate(4 400)&quot;&gt;
&lt;title&gt;G&lt;/title&gt;
&lt;polygon fill=&quot;white&quot; stroke=&quot;none&quot; points=&quot;-4,4 -4,-400 193.29,-400 193.29,4 -4,4&quot;&gt;&lt;/polygon&gt;
&lt;!-- Input Tokens --&gt;
&lt;g id=&quot;node1&quot; class=&quot;node&quot;&gt;
&lt;title&gt;Input Tokens&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M101.98,-396C101.98,-396 35.72,-396 35.72,-396 29.72,-396 23.72,-390 23.72,-384 23.72,-384 23.72,-372 23.72,-372 23.72,-366 29.72,-360 35.72,-360 35.72,-360 101.98,-360 101.98,-360 107.98,-360 113.98,-366 113.98,-372 113.98,-372 113.98,-384 113.98,-384 113.98,-390 107.98,-396 101.98,-396&quot;&gt;&lt;/path&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;68.85&quot; y=&quot;-373.8&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;Input Tokens&lt;/text&gt;
&lt;/g&gt;
&lt;!-- Token Embeddings --&gt;
&lt;g id=&quot;node2&quot; class=&quot;node&quot;&gt;
&lt;title&gt;Token Embeddings&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M119.48,-324C119.48,-324 18.22,-324 18.22,-324 12.22,-324 6.22,-318 6.22,-312 6.22,-312 6.22,-300 6.22,-300 6.22,-294 12.22,-288 18.22,-288 18.22,-288 119.48,-288 119.48,-288 125.48,-288 131.48,-294 131.48,-300 131.48,-300 131.48,-312 131.48,-312 131.48,-318 125.48,-324 119.48,-324&quot;&gt;&lt;/path&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;68.85&quot; y=&quot;-301.8&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;Token Embeddings&lt;/text&gt;
&lt;/g&gt;
&lt;!-- Input Tokens&amp;#45;&amp;gt;Token Embeddings --&gt;
&lt;g id=&quot;edge1&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;Input Tokens-&gt;Token Embeddings&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M68.85,-359.7C68.85,-352.41 68.85,-343.73 68.85,-335.54&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;72.35,-335.62 68.85,-325.62 65.35,-335.62 72.35,-335.62&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- Positional Encoding --&gt;
&lt;g id=&quot;node3&quot; class=&quot;node&quot;&gt;
&lt;title&gt;Positional Encoding&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M121.43,-252C121.43,-252 16.27,-252 16.27,-252 10.27,-252 4.27,-246 4.27,-240 4.27,-240 4.27,-228 4.27,-228 4.27,-222 10.27,-216 16.27,-216 16.27,-216 121.43,-216 121.43,-216 127.43,-216 133.43,-222 133.43,-228 133.43,-228 133.43,-240 133.43,-240 133.43,-246 127.43,-252 121.43,-252&quot;&gt;&lt;/path&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;68.85&quot; y=&quot;-229.8&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;Positional Encoding&lt;/text&gt;
&lt;/g&gt;
&lt;!-- Token Embeddings&amp;#45;&amp;gt;Positional Encoding --&gt;
&lt;g id=&quot;edge2&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;Token Embeddings-&gt;Positional Encoding&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M68.85,-287.7C68.85,-280.41 68.85,-271.73 68.85,-263.54&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;72.35,-263.62 68.85,-253.62 65.35,-263.62 72.35,-263.62&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- Multi&amp;#45;Head Attention --&gt;
&lt;g id=&quot;node4&quot; class=&quot;node&quot;&gt;
&lt;title&gt;Multi-Head Attention&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M125.7,-180C125.7,-180 12,-180 12,-180 6,-180 0,-174 0,-168 0,-168 0,-156 0,-156 0,-150 6,-144 12,-144 12,-144 125.7,-144 125.7,-144 131.7,-144 137.7,-150 137.7,-156 137.7,-156 137.7,-168 137.7,-168 137.7,-174 131.7,-180 125.7,-180&quot;&gt;&lt;/path&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;68.85&quot; y=&quot;-157.8&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;Multi-Head Attention&lt;/text&gt;
&lt;/g&gt;
&lt;!-- Positional Encoding&amp;#45;&amp;gt;Multi&amp;#45;Head Attention --&gt;
&lt;g id=&quot;edge3&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;Positional Encoding-&gt;Multi-Head Attention&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M68.85,-215.7C68.85,-208.41 68.85,-199.73 68.85,-191.54&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;72.35,-191.62 68.85,-181.62 65.35,-191.62 72.35,-191.62&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- Add &amp;amp; Norm --&gt;
&lt;g id=&quot;node5&quot; class=&quot;node&quot;&gt;
&lt;title&gt;Add &amp;#x26; Norm&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M102.18,-108C102.18,-108 35.52,-108 35.52,-108 29.52,-108 23.52,-102 23.52,-96 23.52,-96 23.52,-84 23.52,-84 23.52,-78 29.52,-72 35.52,-72 35.52,-72 102.18,-72 102.18,-72 108.18,-72 114.18,-78 114.18,-84 114.18,-84 114.18,-96 114.18,-96 114.18,-102 108.18,-108 102.18,-108&quot;&gt;&lt;/path&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;68.85&quot; y=&quot;-85.8&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;Add &amp;#x26; Norm&lt;/text&gt;
&lt;/g&gt;
&lt;!-- Multi&amp;#45;Head Attention&amp;#45;&amp;gt;Add &amp;amp; Norm --&gt;
&lt;g id=&quot;edge4&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;Multi-Head Attention-&gt;Add &amp;#x26; Norm&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M68.85,-143.7C68.85,-136.41 68.85,-127.73 68.85,-119.54&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;72.35,-119.62 68.85,-109.62 65.35,-119.62 72.35,-119.62&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- Feed Forward --&gt;
&lt;g id=&quot;node6&quot; class=&quot;node&quot;&gt;
&lt;title&gt;Feed Forward&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M103.92,-36C103.92,-36 33.78,-36 33.78,-36 27.78,-36 21.78,-30 21.78,-24 21.78,-24 21.78,-12 21.78,-12 21.78,-6 27.78,0 33.78,0 33.78,0 103.92,0 103.92,0 109.92,0 115.92,-6 115.92,-12 115.92,-12 115.92,-24 115.92,-24 115.92,-30 109.92,-36 103.92,-36&quot;&gt;&lt;/path&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;68.85&quot; y=&quot;-13.8&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;Feed Forward&lt;/text&gt;
&lt;/g&gt;
&lt;!-- Add &amp;amp; Norm&amp;#45;&amp;gt;Feed Forward --&gt;
&lt;g id=&quot;edge5&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;Add &amp;#x26; Norm-&gt;Feed Forward&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M62.93,-71.7C62.18,-64.41 61.94,-55.73 62.2,-47.54&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;65.69,-47.82 62.86,-37.61 58.71,-47.36 65.69,-47.82&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- Output --&gt;
&lt;g id=&quot;node7&quot; class=&quot;node&quot;&gt;
&lt;title&gt;Output&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M177.29,-36C177.29,-36 146.41,-36 146.41,-36 140.41,-36 134.41,-30 134.41,-24 134.41,-24 134.41,-12 134.41,-12 134.41,-6 140.41,0 146.41,0 146.41,0 177.29,0 177.29,0 183.29,0 189.29,-6 189.29,-12 189.29,-12 189.29,-24 189.29,-24 189.29,-30 183.29,-36 177.29,-36&quot;&gt;&lt;/path&gt;
&lt;text xml:space=&quot;preserve&quot; text-anchor=&quot;middle&quot; x=&quot;161.85&quot; y=&quot;-13.8&quot; font-family=&quot;Times,serif&quot; font-size=&quot;14.00&quot;&gt;Output&lt;/text&gt;
&lt;/g&gt;
&lt;!-- Add &amp;amp; Norm&amp;#45;&amp;gt;Output --&gt;
&lt;g id=&quot;edge7&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;Add &amp;#x26; Norm-&gt;Output&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M91.84,-71.7C103.24,-63.11 117.2,-52.61 129.65,-43.24&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;131.71,-46.07 137.59,-37.26 127.5,-40.47 131.71,-46.07&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;!-- Feed Forward&amp;#45;&amp;gt;Add &amp;amp; Norm --&gt;
&lt;g id=&quot;edge6&quot; class=&quot;edge&quot;&gt;
&lt;title&gt;Feed Forward-&gt;Add &amp;#x26; Norm&lt;/title&gt;
&lt;path fill=&quot;none&quot; stroke=&quot;black&quot; d=&quot;M74.75,-36.1C75.51,-43.37 75.76,-52.04 75.5,-60.24&quot;&gt;&lt;/path&gt;
&lt;polygon fill=&quot;black&quot; stroke=&quot;black&quot; points=&quot;72.01,-59.98 74.86,-70.19 79,-60.43 72.01,-59.98&quot;&gt;&lt;/polygon&gt;
&lt;/g&gt;
&lt;/g&gt;
&lt;/svg&gt;
&lt;/div&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;Parallelizable&lt;/strong&gt; — all tokens processed simultaneously (unlike RNNs)&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Long-range dependencies&lt;/strong&gt; — attention can connect any two positions directly&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Scalable&lt;/strong&gt; — performance improves with more parameters and data (scaling laws)&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Transfer learning&lt;/strong&gt; — pre-trained models can be fine-tuned for specific tasks&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../neural-networks&quot; class=&quot;internal alias&quot; data-slug=&quot;neural-networks&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;Neural Networks&lt;/a&gt; — Transformers are a specific neural network architecture&lt;/li&gt;
&lt;li&gt;Builds into: &lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal alias&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;Flux Architecture&lt;/a&gt; — uses DiT (Diffusion Transformer)&lt;/li&gt;
&lt;li&gt;Builds into: &lt;a href=&quot;../../t5-encoder&quot; class=&quot;internal alias&quot; data-slug=&quot;t5-encoder&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;T5 Encoder&lt;/a&gt; — transformer-based text encoder&lt;/li&gt;
&lt;li&gt;Related: &lt;a href=&quot;../../clip&quot; class=&quot;internal alias&quot; data-slug=&quot;clip&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;CLIP Encoder&lt;/a&gt; — uses transformer for text encoding&lt;/li&gt;
&lt;li&gt;Related: &lt;a href=&quot;../../lora-finetuning&quot; class=&quot;internal alias&quot; data-slug=&quot;lora-finetuning&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;LoRA Fine-tuning&lt;/a&gt; — adapts transformer weights efficiently&lt;/li&gt;
&lt;li&gt;Contrasts with: &lt;a href=&quot;../../rnn&quot; class=&quot;internal alias&quot; data-slug=&quot;rnn&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;RNN&lt;/a&gt; — sequential vs parallel processing&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;Quadratic complexity&lt;/strong&gt; — attention is O(n²) in sequence length; limits context window&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;No recurrence&lt;/strong&gt; — needs positional encoding to understand token order&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Data and compute hungry&lt;/strong&gt; — large transformers require massive resources to train&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>CLIP Text Encoder</title>
    <link href="https://animishraa05.github.io/wiki/imgen/clip" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/clip.md" />
    <summary>The Problem How do you connect text (language) to images (vision) in a way that lets a diffusion model “understand” what you want when you type a prompt? You need a text encoder that maps language to the same representation space as images.</summary>
    <published>2026-04-12T00:00:00.000Z</published>
    <updated>2026-04-12T00:00:00.000Z</updated>
    <publishedTime>Apr 12, 2026</publishedTime>
    <updatedTime>Apr 12, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="ml" label="ml" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;How do you connect text (language) to images (vision) in a way that lets a diffusion model “understand” what you want when you type a prompt? You need a text encoder that maps language to the same representation space as images.&lt;/p&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;CLIP (Contrastive Language-Image Pre-training) uses contrastive learning — it looks at millions of image-text pairs and learns to align the vector representation of an image with its corresponding text description. Once trained, CLIP can encode any text into a vector that “makes sense” to image-based models.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;h3 id=&quot;training&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Training&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#training&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;Show the model many (image, text) pairs. For each batch, the model tries to match the correct image-text pairs while distinguishing incorrect ones. This creates aligned embedding spaces.&lt;/p&gt;
&lt;h3 id=&quot;text-encoding&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Text Encoding&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#text-encoding&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;When you type a prompt, the text encoder converts each word/token into a vector. These vectors live in the same space as image features — so “a dog” and an image of a dog have similar vectors.&lt;/p&gt;
&lt;h3 id=&quot;limitations-for-text-rendering&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Limitations for Text Rendering&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#limitations-for-text-rendering&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;CLIP was trained on natural image-text pairs (captions, descriptions). It learned semantic meaning — “dog” means the animal concept. It did NOT learn to distinguish between “Lakme” and “LAKME” or track exact character sequences. For CLIP, these are all the same concept.&lt;/p&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;77 token context limit&lt;/li&gt;
&lt;li&gt;Semantic understanding (concepts, not characters)&lt;/li&gt;
&lt;li&gt;Used in SDXL, some Flux components&lt;/li&gt;
&lt;li&gt;Fast inference&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../transformers&quot; class=&quot;internal&quot; data-slug=&quot;transformers&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;transformers&lt;/a&gt;, &lt;a href=&quot;../../neural-networks&quot; class=&quot;internal&quot; data-slug=&quot;neural-networks&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;neural-networks&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Built into: &lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;flux-architecture&lt;/a&gt;, &lt;a href=&quot;../../hybrid-pipeline&quot; class=&quot;internal&quot; data-slug=&quot;hybrid-pipeline&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;hybrid-pipeline&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Related to: &lt;a href=&quot;../../text-rendering-problem&quot; class=&quot;internal&quot; data-slug=&quot;text-rendering-problem&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;text-rendering-problem&lt;/a&gt;, &lt;a href=&quot;../../t5-encoder&quot; class=&quot;internal&quot; data-slug=&quot;t5-encoder&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;t5-encoder&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Character-blind: “Lakme”, “LAKME”, “lakme” = same vector&lt;/li&gt;
&lt;li&gt;77 token limit = struggles with very long prompts&lt;/li&gt;
&lt;li&gt;Semantic focus misses fine details like letter shapes&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>ControlNet</title>
    <link href="https://animishraa05.github.io/wiki/imgen/controlnet" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/controlnet.md" />
    <summary>The Problem How do you tell a diffusion model exactly WHERE something should be in the image, or WHAT specific spatial structure to follow — not just what’s in the prompt? Standard text-to-image only gives you coarse control through prompts.</summary>
    <published>2026-04-12T00:00:00.000Z</published>
    <updated>2026-04-12T00:00:00.000Z</updated>
    <publishedTime>Apr 12, 2026</publishedTime>
    <updatedTime>Apr 12, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="diffusion" label="diffusion" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;How do you tell a diffusion model exactly WHERE something should be in the image, or WHAT specific spatial structure to follow — not just what’s in the prompt? Standard text-to-image only gives you coarse control through prompts.&lt;/p&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;ControlNet is an additional neural network that runs in parallel with the diffusion model. It takes an extra input (edge map, depth map, pose skeleton, or glyph image) and adds spatial constraints to the generation process. The model must follow your control signal while still generating based on the text prompt.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;h3 id=&quot;architecture&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Architecture&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#architecture&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;ControlNet copies the UNet/DiT encoder layers and trains them separately on the control task. The trained ControlNet weights connect to the main model via “zero convolutions” that gradually merge the control signal.&lt;/p&gt;
&lt;h3 id=&quot;control-types&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Control Types&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#control-types&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;Canny Edge&lt;/strong&gt;: Follow edges from an image&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Depth&lt;/strong&gt;: Follow depth map structure&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Pose&lt;/strong&gt;: Follow human pose skeleton&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Segmentation&lt;/strong&gt;: Follow object layout&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Glyph&lt;/strong&gt;: Follow exact text stroke positions (for text rendering)&lt;/li&gt;
&lt;/ul&gt;
&lt;h3 id=&quot;for-glyph-injection&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;For Glyph Injection&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#for-glyph-injection&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;When we render “Made by Lakme” as a black-on-white image and pass it to ControlNet, the model receives pixel-level spatial constraints: “there must be dark strokes at these coordinates.” It doesn’t need to understand letters — it just follows the stroke pattern.&lt;/p&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Pre-trained weights available for all control types&lt;/li&gt;
&lt;li&gt;Zero training required for glyph control&lt;/li&gt;
&lt;li&gt;controlnet_scale (0-1) controls strength&lt;/li&gt;
&lt;li&gt;Works with Flux, SDXL, other diffusers&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../diffusion-models&quot; class=&quot;internal&quot; data-slug=&quot;diffusion-models&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;diffusion-models&lt;/a&gt;, &lt;a href=&quot;../../vae&quot; class=&quot;internal&quot; data-slug=&quot;vae&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;vae&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Used in: &lt;a href=&quot;../../glyph-injection&quot; class=&quot;internal&quot; data-slug=&quot;glyph-injection&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;glyph-injection&lt;/a&gt;, &lt;a href=&quot;../../hybrid-pipeline&quot; class=&quot;internal&quot; data-slug=&quot;hybrid-pipeline&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;hybrid-pipeline&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Related to: &lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;flux-architecture&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Thin strokes = weak signal = drift&lt;/li&gt;
&lt;li&gt;Scale too high = looks pasted on&lt;/li&gt;
&lt;li&gt;Scale too low = doesn’t follow constraints&lt;/li&gt;
&lt;li&gt;Additional VRAM needed&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>Diffusion Models</title>
    <link href="https://animishraa05.github.io/wiki/imgen/diffusion-models" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/diffusion-models.md" />
    <summary>The Problem How do you teach a neural network to generate entirely new images from scratch? Simple classification won’t work — you need the model to learn the structure of images well enough to create novel ones.</summary>
    <published>2026-04-12T00:00:00.000Z</published>
    <updated>2026-04-12T00:00:00.000Z</updated>
    <publishedTime>Apr 12, 2026</publishedTime>
    <updatedTime>Apr 12, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="ml" label="ml" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;How do you teach a neural network to generate entirely new images from scratch? Simple classification won’t work — you need the model to learn the structure of images well enough to create novel ones.&lt;/p&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Diffusion models work by learning to reverse a gradual noising process. Start with an image, add noise step by step until it’s pure random noise, then train the model to reverse this — going from noise back to a clean image. Once trained, you start with pure noise and run the reverse process to generate new images.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;h3 id=&quot;forward-process-noising&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Forward Process (noising)&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#forward-process-noising&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;pre class=&quot;notranslate&quot;&gt;&lt;span class=&quot;clipboard-button&quot; type=&quot;button&quot; aria-label=&quot;copy source&quot;&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 -8 24 24&quot; fill=&quot;currentColor&quot; stroke=&quot;none&quot; stroke-width=&quot;0&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot; class=&quot;copy-icon&quot;&gt;&lt;use href=&quot;#github-copy&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 -8 24 24&quot; fill=&quot;currentColor&quot; stroke=&quot;none&quot; stroke-width=&quot;0&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot; class=&quot;check-icon&quot;&gt;&lt;use href=&quot;#github-check&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/span&gt;&lt;code class=&quot;notranslate&quot;&gt;Image → Add small noise → Add small noise → ... → Pure noise
&lt;/code&gt;&lt;/pre&gt;
&lt;p dir=&quot;auto&quot;&gt;This is fixed — no model needed. At each step a small amount of Gaussian noise is added.&lt;/p&gt;
&lt;h3 id=&quot;reverse-process-denoising&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Reverse Process (denoising)&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#reverse-process-denoising&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;pre class=&quot;notranslate&quot;&gt;&lt;span class=&quot;clipboard-button&quot; type=&quot;button&quot; aria-label=&quot;copy source&quot;&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 -8 24 24&quot; fill=&quot;currentColor&quot; stroke=&quot;none&quot; stroke-width=&quot;0&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot; class=&quot;copy-icon&quot;&gt;&lt;use href=&quot;#github-copy&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 -8 24 24&quot; fill=&quot;currentColor&quot; stroke=&quot;none&quot; stroke-width=&quot;0&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot; class=&quot;check-icon&quot;&gt;&lt;use href=&quot;#github-check&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/span&gt;&lt;code class=&quot;notranslate&quot;&gt;Pure noise ← Remove noise ← Remove noise ← ... ← Original image
&lt;/code&gt;&lt;/pre&gt;
&lt;p dir=&quot;auto&quot;&gt;This is learned. The model predicts how much noise to remove at each step.&lt;/p&gt;
&lt;h3 id=&quot;sampling&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Sampling&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#sampling&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;In practice, you start with random noise and run 25-50 denoising steps. At each step, the model looks at the current noisy image and your text prompt, then predicts what the less-noisy version should look like.&lt;/p&gt;
&lt;h3 id=&quot;ddpm-vs-ddim&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;DDPM vs DDIM&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#ddpm-vs-ddim&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;DDPM: Original approach, high quality but slow (many steps)&lt;/li&gt;
&lt;li&gt;DDIM: Faster sampling, fewer steps with comparable quality&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;State-of-the-art for photorealistic image generation&lt;/li&gt;
&lt;li&gt;Text-conditioned via cross-attention or joint attention&lt;/li&gt;
&lt;li&gt;Latent diffusion (LDM) runs in compressed space for efficiency&lt;/li&gt;
&lt;li&gt;Most open-source models (SDXL, Flux) use this approach&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../vae&quot; class=&quot;internal&quot; data-slug=&quot;vae&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;vae&lt;/a&gt;, &lt;a href=&quot;../../neural-networks&quot; class=&quot;internal&quot; data-slug=&quot;neural-networks&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;neural-networks&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Built into: &lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;flux-architecture&lt;/a&gt;, &lt;a href=&quot;../../lora-finetuning&quot; class=&quot;internal&quot; data-slug=&quot;lora-finetuning&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;lora-finetuning&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Related: &lt;a href=&quot;../../text-rendering-problem&quot; class=&quot;internal&quot; data-slug=&quot;text-rendering-problem&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;text-rendering-problem&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Slow inference (25-50 steps per image)&lt;/li&gt;
&lt;li&gt;High-frequency details (like text) are hardest to generate&lt;/li&gt;
&lt;li&gt;Mode collapse possible if training data insufficient&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>Flux Architecture</title>
    <link href="https://animishraa05.github.io/wiki/imgen/flux-architecture" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/flux-architecture.md" />
    <summary>The Problem Standard diffusion models (SDXL, Stable Diffusion) have architecture limitations that make text rendering and long-prompt handling difficult.</summary>
    <published>2026-04-12T00:00:00.000Z</published>
    <updated>2026-04-12T00:00:00.000Z</updated>
    <publishedTime>Apr 12, 2026</publishedTime>
    <updatedTime>Apr 12, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="diffusion" label="diffusion" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Standard diffusion models (SDXL, Stable Diffusion) have architecture limitations that make text rendering and long-prompt handling difficult. SDXL uses a UNet-based architecture with CLIP-only encoding and 4-channel VAE — all contributing to the text rendering problem.&lt;/p&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Flux.1-dev is a modern diffusion model using DiT (Diffusion Transformer) architecture instead of UNet, dual text encoders (CLIP + T5), and a 16-channel VAE. These architectural choices directly address the three-layer text rendering problem.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;h3 id=&quot;dit-architecture-vs-unet&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;DiT Architecture (vs UNet)&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#dit-architecture-vs-unet&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;Traditional UNet models process image and text information separately, with cross-attention only at specific layers. DiT processes both in a unified attention mechanism — text tokens and image tokens interact at every single layer throughout generation.&lt;/p&gt;
&lt;h3 id=&quot;dual-text-encoders&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Dual Text Encoders&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#dual-text-encoders&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;Flux uses BOTH CLIP-L and T5-XXL:&lt;/p&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;CLIP-L&lt;/strong&gt;: Semantic understanding — “Lakme = luxury cosmetics brand”&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;T5-XXL&lt;/strong&gt;: Character-level encoding — “L-a-k-m-e = exact character sequence”&lt;/li&gt;
&lt;/ul&gt;
&lt;p dir=&quot;auto&quot;&gt;Both are projected into a joint 3072-dimensional space, giving the model both conceptual and character-level information.&lt;/p&gt;
&lt;h3 id=&quot;16-channel-vae&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;16-Channel VAE&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#16-channel-vae&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;Previous models used 4-channel VAE. Flux’s 16-channel VAE has 4x the information capacity in latent space, allowing fine strokes (letter details) to survive compression and decompression.&lt;/p&gt;
&lt;h3 id=&quot;flow-matching&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Flow Matching&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#flow-matching&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;Flux uses flow matching training instead of traditional DDPM. This produces sharper results with fewer sampling steps.&lt;/p&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;12B parameters&lt;/li&gt;
&lt;li&gt;Apache 2.0 license (schnell) / Proprietary (dev)&lt;/li&gt;
&lt;li&gt;T5-XXL context: 4096 tokens (vs CLIP’s 77)&lt;/li&gt;
&lt;li&gt;Native 1024x1024 output&lt;/li&gt;
&lt;li&gt;~24GB VRAM for full quality inference&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../diffusion-models&quot; class=&quot;internal&quot; data-slug=&quot;diffusion-models&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;diffusion-models&lt;/a&gt;, &lt;a href=&quot;../../transformers&quot; class=&quot;internal&quot; data-slug=&quot;transformers&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;transformers&lt;/a&gt;, &lt;a href=&quot;../../t5-encoder&quot; class=&quot;internal&quot; data-slug=&quot;t5-encoder&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;t5-encoder&lt;/a&gt;, &lt;a href=&quot;../../clip&quot; class=&quot;internal&quot; data-slug=&quot;clip&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;clip&lt;/a&gt;, &lt;a href=&quot;../../vae&quot; class=&quot;internal&quot; data-slug=&quot;vae&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;vae&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Part of: &lt;a href=&quot;../../hybrid-pipeline&quot; class=&quot;internal&quot; data-slug=&quot;hybrid-pipeline&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;hybrid-pipeline&lt;/a&gt;, &lt;a href=&quot;../../text-rendering-problem&quot; class=&quot;internal&quot; data-slug=&quot;text-rendering-problem&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;text-rendering-problem&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Related: &lt;a href=&quot;../../lora-finetuning&quot; class=&quot;internal&quot; data-slug=&quot;lora-finetuning&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;lora-finetuning&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;VRAM intensive — needs A100 or equivalent for comfortable inference&lt;/li&gt;
&lt;li&gt;T5-XXL loading adds latency (~3 seconds)&lt;/li&gt;
&lt;li&gt;Quality degrades significantly below 16GB VRAM without optimization&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>Glyph Injection via ControlNet</title>
    <link href="https://animishraa05.github.io/wiki/imgen/glyph-injection" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/glyph-injection.md" />
    <summary>The Problem Even with T5 encoding and 16-channel VAE, pure diffusion models still struggle with exact text rendering.</summary>
    <published>2026-04-12T00:00:00.000Z</published>
    <updated>2026-04-12T00:00:00.000Z</updated>
    <publishedTime>Apr 12, 2026</publishedTime>
    <updatedTime>Apr 12, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="diffusion" label="diffusion" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Even with T5 encoding and 16-channel VAE, pure diffusion models still struggle with exact text rendering. Semantic drift can override even character-level encoding when the model decides the “concept” of text matters more than its shape. For marketing use, where brand names must be pixel-perfect, this is unacceptable.&lt;/p&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Don’t ask the model to generate letters — give it the exact letter shapes as a spatial constraint. Render the text as a clean image using standard font libraries, then use ControlNet to force the diffusion model to follow those exact strokes.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;pre class=&quot;notranslate&quot;&gt;&lt;span class=&quot;clipboard-button&quot; type=&quot;button&quot; aria-label=&quot;copy source&quot;&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 -8 24 24&quot; fill=&quot;currentColor&quot; stroke=&quot;none&quot; stroke-width=&quot;0&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot; class=&quot;copy-icon&quot;&gt;&lt;use href=&quot;#github-copy&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 -8 24 24&quot; fill=&quot;currentColor&quot; stroke=&quot;none&quot; stroke-width=&quot;0&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot; class=&quot;check-icon&quot;&gt;&lt;use href=&quot;#github-check&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/span&gt;&lt;code class=&quot;notranslate&quot;&gt;1. Extract text_to_render from LLM schema
2. Use PIL/Pillow to render text as black-on-white image at 1024x1024
3. Pass this glyph image to ControlNet as conditioning
4. During denoising, ControlNet enforces stroke geometry at every step
5. Flux applies style/color on top of exact stroke positions
&lt;/code&gt;&lt;/pre&gt;
&lt;h3 id=&quot;why-this-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Why This Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#why-this-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;ControlNet operates at the pixel/spatial level — it doesn’t “understand” letters, it just sees dark pixels and enforces dark pixels at the same locations. The model isn’t generating “M”, it’s being told “there must be dark strokes at these specific coordinates.”&lt;/p&gt;
&lt;h3 id=&quot;controlnet-scale&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;ControlNet Scale&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#controlnet-scale&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;The &lt;code class=&quot;notranslate&quot;&gt;controlnet_scale&lt;/code&gt; parameter (0.0-1.0) controls how strongly ControlNet overrides Flux:&lt;/p&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;0.85: Exact brand names, logos&lt;/li&gt;
&lt;li&gt;0.75: Taglines, product names (sweet spot)&lt;/li&gt;
&lt;li&gt;0.60: Creative/artistic text&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Zero training required — uses pre-trained ControlNet weights&lt;/li&gt;
&lt;li&gt;Works with any font (select based on brand guidelines)&lt;/li&gt;
&lt;li&gt;Multi-line text supported in single glyph image&lt;/li&gt;
&lt;li&gt;Font weight affects reliability (bold = more reliable)&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Solves: &lt;a href=&quot;../../text-rendering-problem&quot; class=&quot;internal&quot; data-slug=&quot;text-rendering-problem&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;text-rendering-problem&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../controlnet&quot; class=&quot;internal&quot; data-slug=&quot;controlnet&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;controlnet&lt;/a&gt;, &lt;a href=&quot;../../vae&quot; class=&quot;internal&quot; data-slug=&quot;vae&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;vae&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Used in: &lt;a href=&quot;../../hybrid-pipeline&quot; class=&quot;internal&quot; data-slug=&quot;hybrid-pipeline&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;hybrid-pipeline&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Related: &lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;flux-architecture&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Thin decorative fonts cause weak ControlNet signal → drift&lt;/li&gt;
&lt;li&gt;Position must be specified (center, top-left, etc.)&lt;/li&gt;
&lt;li&gt;Scale too high = text looks “pasted on”, not integrated&lt;/li&gt;
&lt;li&gt;Scale too low = spell errors return&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>Hybrid LLM-Guided Diffusion Pipeline</title>
    <link href="https://animishraa05.github.io/wiki/imgen/hybrid-pipeline" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/hybrid-pipeline.md" />
    <summary>The Problem Marketing image generation requires both language reasoning (understanding complex briefs, brand requirements, structured output) and high-fidelity visual synthesis (photorealistic images, precise text rendering, brand-consistent styling).</summary>
    <published>2026-04-12T00:00:00.000Z</published>
    <updated>2026-04-12T00:00:00.000Z</updated>
    <publishedTime>Apr 12, 2026</publishedTime>
    <updatedTime>Apr 12, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="diffusion" label="diffusion" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Marketing image generation requires both &lt;strong&gt;language reasoning&lt;/strong&gt; (understanding complex briefs, brand requirements, structured output) and &lt;strong&gt;high-fidelity visual synthesis&lt;/strong&gt; (photorealistic images, precise text rendering, brand-consistent styling). No single model family handles both well enough for commercial production use.&lt;/p&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Pure diffusion pipelines can’t reason over complex briefs&lt;/li&gt;
&lt;li&gt;Pure LLM pipelines can’t produce photorealistic marketing imagery&lt;/li&gt;
&lt;li&gt;Unified multimodal models lack fine-tuning granularity for brand control&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;A hybrid architecture that combines an LLM for language understanding/reasoning with a diffusion model for pixel synthesis. The LLM handles cognitive tasks (brief decomposition, prompt structuring), while the diffusion model handles visual generation (image synthesis, text rendering). They communicate through structured data schemas, not raw text.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;pre class=&quot;notranslate&quot;&gt;&lt;span class=&quot;clipboard-button&quot; type=&quot;button&quot; aria-label=&quot;copy source&quot;&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 -8 24 24&quot; fill=&quot;currentColor&quot; stroke=&quot;none&quot; stroke-width=&quot;0&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot; class=&quot;copy-icon&quot;&gt;&lt;use href=&quot;#github-copy&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 -8 24 24&quot; fill=&quot;currentColor&quot; stroke=&quot;none&quot; stroke-width=&quot;0&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot; class=&quot;check-icon&quot;&gt;&lt;use href=&quot;#github-check&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/span&gt;&lt;code class=&quot;notranslate&quot;&gt;User Brief (natural language)
         ↓
   Brief Decomposer (LLM)
   → Structured Schema: {subject, style, lighting, camera, mood, text_to_render}
         ↓
   Prompt Composer (LLM)
   → Weighted prompt string + negative prompt + glyph instructions
         ↓
   Reference Asset Injection (IP-Adapter) → Optional brand image conditioning
         ↓
   ControlNet (optional) → Spatial structure / glyph enforcement
         ↓
   Flux.1-dev + Client LoRA → Image synthesis
         ↓
   Quality Gate (CLIP + Aesthetic) → Auto-reject/retries
         ↓
   Upscaler → Final high-res output
&lt;/code&gt;&lt;/pre&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;Separation of concerns&lt;/strong&gt;: LLM handles cognition, diffusion handles generation&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Structured intermediate outputs&lt;/strong&gt;: JSON schemas make system auditable and debuggable&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Modular components&lt;/strong&gt;: Each stage is swappable (LLM, diffusion model, control mechanisms)&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Production-ready&lt;/strong&gt;: Supports async job queues, per-client LoRA hot-swapping, quality gates&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../diffusion-models&quot; class=&quot;internal&quot; data-slug=&quot;diffusion-models&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;diffusion-models&lt;/a&gt;, &lt;a href=&quot;../../transformers&quot; class=&quot;internal&quot; data-slug=&quot;transformers&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;transformers&lt;/a&gt;, &lt;a href=&quot;../../clip&quot; class=&quot;internal&quot; data-slug=&quot;clip&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;clip&lt;/a&gt;, &lt;a href=&quot;../../t5-encoder&quot; class=&quot;internal&quot; data-slug=&quot;t5-encoder&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;t5-encoder&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Builds into: &lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;flux-architecture&lt;/a&gt;, &lt;a href=&quot;../../lora-finetuning&quot; class=&quot;internal&quot; data-slug=&quot;lora-finetuning&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;lora-finetuning&lt;/a&gt;, &lt;a href=&quot;../../controlnet&quot; class=&quot;internal&quot; data-slug=&quot;controlnet&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;controlnet&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Related: &lt;a href=&quot;../../text-rendering-problem&quot; class=&quot;internal&quot; data-slug=&quot;text-rendering-problem&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;text-rendering-problem&lt;/a&gt;, &lt;a href=&quot;../../latent-space&quot; class=&quot;internal&quot; data-slug=&quot;latent-space&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;latent-space&lt;/a&gt;, &lt;a href=&quot;../../vae&quot; class=&quot;internal&quot; data-slug=&quot;vae&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;vae&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;LLM quality determines everything&lt;/strong&gt;: A mediocre prompt decomposer ruins downstream output&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Latency compounding&lt;/strong&gt;: Each LLM call adds 1-3 seconds; pipeline needs async handling&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;CLIP vs T5 tradeoff&lt;/strong&gt;: CLIP faster but character-blind; T5 slower but spelling-aware&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>LoRA Fine-tuning for Brand Consistency</title>
    <link href="https://animishraa05.github.io/wiki/imgen/lora-finetuning" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/lora-finetuning.md" />
    <summary>The Problem How do you make a diffusion model produce outputs in a specific client’s brand style (colors, lighting, composition) without retraining the entire model from scratch? Full fine-tuning is expensive, slow, and requires massive GPU resources.</summary>
    <published>2026-04-12T00:00:00.000Z</published>
    <updated>2026-04-12T00:00:00.000Z</updated>
    <publishedTime>Apr 12, 2026</publishedTime>
    <updatedTime>Apr 12, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="ml" label="ml" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;How do you make a diffusion model produce outputs in a specific client’s brand style (colors, lighting, composition) without retraining the entire model from scratch? Full fine-tuning is expensive, slow, and requires massive GPU resources.&lt;/p&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;LoRA (Low-Rank Adaptation) adds small “adapter” weights to a pre-trained model. Instead of updating all 12B parameters, LoRA adds ~100MB of new weights that modify the model’s behavior. The base model stays frozen — you just load different adapters for different clients.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;h3 id=&quot;low-rank-decomposition&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Low-Rank Decomposition&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#low-rank-decomposition&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;LoRA adds two small matrices (rank r, typically 8-16) that approximate the weight change. Instead of updating a W matrix directly, you learn W + BA where B and A are small.&lt;/p&gt;
&lt;h3 id=&quot;training&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Training&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#training&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Freeze the base model (Flux.1-dev)&lt;/li&gt;
&lt;li&gt;Add LoRA layers in parallel to attention weights&lt;/li&gt;
&lt;li&gt;Train on 20-50 brand images for a few hours&lt;/li&gt;
&lt;li&gt;Result: ~100MB .safetensors file&lt;/li&gt;
&lt;/ul&gt;
&lt;h3 id=&quot;inference&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Inference&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#inference&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Load base Flux.1-dev (one time)&lt;/li&gt;
&lt;li&gt;Hot-swap LoRA files per client&lt;/li&gt;
&lt;li&gt;Merge at runtime — no model reload needed&lt;/li&gt;
&lt;/ul&gt;
&lt;h3 id=&quot;three-lora-types-in-our-pipeline&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Three LoRA Types in Our Pipeline&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#three-lora-types-in-our-pipeline&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;ol dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;Base Aesthetic LoRA&lt;/strong&gt; — Trained on LAION-Art, gives professional photography look&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Brand-Specific LoRA&lt;/strong&gt; — Trained on each client’s 20-50 images&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Client LoRA&lt;/strong&gt; — Hot-swapped per job&lt;/li&gt;
&lt;/ol&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;~100MB per LoRA (vs 12GB for full model)&lt;/li&gt;
&lt;li&gt;Training takes hours, not days&lt;/li&gt;
&lt;li&gt;Hot-swappable at inference time&lt;/li&gt;
&lt;li&gt;Can merge multiple LoRAs (base + brand)&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../diffusion-models&quot; class=&quot;internal&quot; data-slug=&quot;diffusion-models&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;diffusion-models&lt;/a&gt;, &lt;a href=&quot;../../neural-networks&quot; class=&quot;internal&quot; data-slug=&quot;neural-networks&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;neural-networks&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Used in: &lt;a href=&quot;../../hybrid-pipeline&quot; class=&quot;internal&quot; data-slug=&quot;hybrid-pipeline&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;hybrid-pipeline&lt;/a&gt;, &lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;flux-architecture&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Related: &lt;a href=&quot;../../ip-adapter&quot; class=&quot;internal&quot; data-slug=&quot;ip-adapter&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;ip-adapter&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;LoRA good for style, weak for exact product shapes&lt;/li&gt;
&lt;li&gt;Need quality training data (20-50 consistent images)&lt;/li&gt;
&lt;li&gt;Too many LoRAs = management overhead&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>T5 Text Encoder</title>
    <link href="https://animishraa05.github.io/wiki/imgen/t5-encoder" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/t5-encoder.md" />
    <summary>The Problem CLIP’s semantic encoding can’t distinguish between “Lakme” and “LAKME” or track exact spelling.</summary>
    <published>2026-04-12T00:00:00.000Z</published>
    <updated>2026-04-12T00:00:00.000Z</updated>
    <publishedTime>Apr 12, 2026</publishedTime>
    <updatedTime>Apr 12, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="ml" label="ml" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;CLIP’s semantic encoding can’t distinguish between “Lakme” and “LAKME” or track exact spelling. For text rendering in images, we need a text encoder that actually understands character-level details, not just concepts.&lt;/p&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;T5 (Text-to-Text Transfer Transformer) is a pure language model trained on massive text corpora — books, websites, code. Unlike CLIP which learned from image-text pairs, T5 learned from raw text and therefore maintains character-level information. “Lakme” and “LAKME” are different sequences that T5 encodes differently.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;h3 id=&quot;character-level-processing&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Character-Level Processing&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#character-level-processing&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;T5 tokenizes text at a sub-word level. “Made by Lakme” becomes a sequence of tokens that preserve the exact character information. The encoding captures letter sequences, not just overall meaning.&lt;/p&gt;
&lt;h3 id=&quot;context-length&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Context Length&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#context-length&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;T5-XXL supports 4096 tokens — far more than CLIP’s 77. This enables handling long, complex marketing briefs with many details.&lt;/p&gt;
&lt;h3 id=&quot;dual-encoding-in-flux&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Dual Encoding in Flux&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#dual-encoding-in-flux&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;Flux uses BOTH CLIP and T5 simultaneously:&lt;/p&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;CLIP provides semantic understanding (brand, style, mood)&lt;/li&gt;
&lt;li&gt;T5 provides character-level detail (spelling, exact words)&lt;/li&gt;
&lt;li&gt;Both projected to a joint 3072-dim space&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;4096 token context (vs CLIP’s 77)&lt;/li&gt;
&lt;li&gt;Character-aware: case, spelling, exact words matter&lt;/li&gt;
&lt;li&gt;Larger model (T5-XXL ~4B params) = slower than CLIP&lt;/li&gt;
&lt;li&gt;Used in Flux for text encoding&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../transformers&quot; class=&quot;internal&quot; data-slug=&quot;transformers&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;transformers&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Used in: &lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;flux-architecture&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Solves: &lt;a href=&quot;../../text-rendering-problem&quot; class=&quot;internal&quot; data-slug=&quot;text-rendering-problem&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;text-rendering-problem&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Related: &lt;a href=&quot;../../clip&quot; class=&quot;internal&quot; data-slug=&quot;clip&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;clip&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Slower than CLIP encoding&lt;/li&gt;
&lt;li&gt;Doesn’t “understand” images — only text&lt;/li&gt;
&lt;li&gt;Need to pair with CLIP for best results&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>Text Rendering Problem in Diffusion Models</title>
    <link href="https://animishraa05.github.io/wiki/imgen/text-rendering-problem" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/text-rendering-problem.md" />
    <summary>The Problem When users prompt a diffusion model to render specific text like “Made by Lakme” on an image, the output frequently contains spelling errors (“Maed by Lakm3”), wrong characters, or illegible text.</summary>
    <published>2026-04-12T00:00:00.000Z</published>
    <updated>2026-04-12T00:00:00.000Z</updated>
    <publishedTime>Apr 12, 2026</publishedTime>
    <updatedTime>Apr 12, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="diffusion" label="diffusion" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;When users prompt a diffusion model to render specific text like “Made by Lakme” on an image, the output frequently contains spelling errors (“Maed by Lakm3”), wrong characters, or illegible text. This is unacceptable for commercial marketing where brand names and taglines must be exact.&lt;/p&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;The text rendering problem isn’t one issue — it’s three separate problems occurring at three different layers of the generation pipeline. Each layer needs its own solution.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;h3 id=&quot;problem-1-character-blind-text-encoder-clip&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Problem 1: Character-Blind Text Encoder (CLIP)&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#problem-1-character-blind-text-encoder-clip&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;When you write “Made by Lakme”, the text encoder converts it to a vector. CLIP was trained on image-text pairs to understand &lt;strong&gt;semantic meaning&lt;/strong&gt; — “Lakme” means a cosmetics brand concept. CLIP has no concept of the specific letters L-a-k-m-e.&lt;/p&gt;
&lt;pre class=&quot;notranslate&quot;&gt;&lt;span class=&quot;clipboard-button&quot; type=&quot;button&quot; aria-label=&quot;copy source&quot;&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 -8 24 24&quot; fill=&quot;currentColor&quot; stroke=&quot;none&quot; stroke-width=&quot;0&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot; class=&quot;copy-icon&quot;&gt;&lt;use href=&quot;#github-copy&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 -8 24 24&quot; fill=&quot;currentColor&quot; stroke=&quot;none&quot; stroke-width=&quot;0&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot; class=&quot;check-icon&quot;&gt;&lt;use href=&quot;#github-check&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/span&gt;&lt;code class=&quot;notranslate&quot;&gt;&quot;made by lakme&quot; → [concept: luxury brand, cosmetics, ...]
&quot;MAde By LAKME&quot; → [same concept, no difference]
&lt;/code&gt;&lt;/pre&gt;
&lt;h3 id=&quot;problem-2-semantic-drift-in-attention&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Problem 2: Semantic Drift in Attention&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#problem-2-semantic-drift-in-attention&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;During image generation, the model’s attention mechanism prioritizes the &lt;strong&gt;concept&lt;/strong&gt; of “Lakme” (brand, luxury, glossy) over the &lt;strong&gt;letter shapes&lt;/strong&gt; (L, a, k, m, e). The word “GOLD” literally gets rendered as golden shimmer instead of the letters G-O-L-D because the semantic concept overrides the glyph information.&lt;/p&gt;
&lt;h3 id=&quot;problem-3-vae-stroke-destruction&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Problem 3: VAE Stroke Destruction&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#problem-3-vae-stroke-destruction&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;Diffusion models work in a compressed “latent space” created by a VAE (Variational Autoencoder). The VAE compresses 1024x1024 pixels into 128x128 latent representations. During this compression, fine details like thin letter strokes get blurred or lost entirely. Standard 4-channel VAEs (used in SDXL) don’t have enough capacity to preserve high-frequency text details.&lt;/p&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;strong&gt;Multi-layer problem&lt;/strong&gt;: Character-blind encoder + semantic drift + VAE compression = three failure points&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Training data scarcity&lt;/strong&gt;: Diffusion models trained on LAION where most images have no text&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Industry-wide&lt;/strong&gt;: All major models (SDXL, Midjourney, DALL-E) struggle with this&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Solution requires multiple fixes&lt;/strong&gt;: No single approach solves all three layers&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Related to: &lt;a href=&quot;../../semantic-drift&quot; class=&quot;internal&quot; data-slug=&quot;semantic-drift&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;semantic-drift&lt;/a&gt;, &lt;a href=&quot;../../vae&quot; class=&quot;internal&quot; data-slug=&quot;vae&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;vae&lt;/a&gt;, &lt;a href=&quot;../../clip&quot; class=&quot;internal&quot; data-slug=&quot;clip&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;clip&lt;/a&gt;, &lt;a href=&quot;../../t5-encoder&quot; class=&quot;internal&quot; data-slug=&quot;t5-encoder&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;t5-encoder&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Solved by: &lt;a href=&quot;../../glyph-injection&quot; class=&quot;internal&quot; data-slug=&quot;glyph-injection&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;glyph-injection&lt;/a&gt;, &lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;flux-architecture&lt;/a&gt;, &lt;a href=&quot;../../controlnet&quot; class=&quot;internal&quot; data-slug=&quot;controlnet&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;controlnet&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Part of: &lt;a href=&quot;../../hybrid-pipeline&quot; class=&quot;internal&quot; data-slug=&quot;hybrid-pipeline&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;hybrid-pipeline&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Short words (2-3 letters) often render correctly; longer brand names fail more&lt;/li&gt;
&lt;li&gt;Numbers fail more than letters (0 vs O confusion)&lt;/li&gt;
&lt;li&gt;Multi-language text (Hindi + English) is even harder&lt;/li&gt;
&lt;li&gt;Sans-serif fonts render better than serif (fewer fine details to lose)&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>Text Rendering Solutions — Layer by Layer</title>
    <link href="https://animishraa05.github.io/wiki/imgen/text-rendering-solutions" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/text-rendering-solutions.md" />
    <summary>What This Synthesizes This page combines three core concepts to show how the text rendering problem is actually solved across multiple architectural layers: text-rendering-problem flux-architecture glyph-injection The Three-Layer Solution Layer 1: Character-Aware Encoding (T5) Problem: CLIP is chara...</summary>
    <published>2026-04-12T00:00:00.000Z</published>
    <updated>2026-04-12T00:00:00.000Z</updated>
    <publishedTime>Apr 12, 2026</publishedTime>
    <updatedTime>Apr 12, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="diffusion" label="diffusion" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;what-this-synthesizes&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;What This Synthesizes&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#what-this-synthesizes&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;This page combines three core concepts to show how the text rendering problem is actually solved across multiple architectural layers:&lt;/p&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;a href=&quot;../../text-rendering-problem&quot; class=&quot;internal&quot; data-slug=&quot;text-rendering-problem&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;text-rendering-problem&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;flux-architecture&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../glyph-injection&quot; class=&quot;internal&quot; data-slug=&quot;glyph-injection&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;glyph-injection&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;the-three-layer-solution&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Three-Layer Solution&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-three-layer-solution&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;h3 id=&quot;layer-1-character-aware-encoding-t5&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Layer 1: Character-Aware Encoding (T5)&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#layer-1-character-aware-encoding-t5&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;&lt;strong&gt;Problem:&lt;/strong&gt; CLIP is character-blind — “Lakme” and “LAKME” are the same vector.&lt;/p&gt;
&lt;p dir=&quot;auto&quot;&gt;&lt;strong&gt;Solution:&lt;/strong&gt; Flux uses T5-XXL alongside CLIP. T5 was trained on pure text and maintains character-level information. “L-a-k-m-e” is preserved as distinct tokens.&lt;/p&gt;
&lt;p dir=&quot;auto&quot;&gt;&lt;strong&gt;Result:&lt;/strong&gt; The model receives both semantic meaning AND character sequence.&lt;/p&gt;
&lt;h3 id=&quot;layer-2-higher-capacity-latent-space-16-channel-vae&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Layer 2: Higher Capacity Latent Space (16-channel VAE)&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#layer-2-higher-capacity-latent-space-16-channel-vae&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;&lt;strong&gt;Problem:&lt;/strong&gt; Standard 4-channel VAE loses thin strokes during compression.&lt;/p&gt;
&lt;p dir=&quot;auto&quot;&gt;&lt;strong&gt;Solution:&lt;/strong&gt; Flux uses 16-channel VAE with 4x information capacity. Fine letter details survive the compression/decompression cycle.&lt;/p&gt;
&lt;p dir=&quot;auto&quot;&gt;&lt;strong&gt;Result:&lt;/strong&gt; Strokes retain sharpness through the generation process.&lt;/p&gt;
&lt;h3 id=&quot;layer-3-spatial-constraint-enforcement-controlnet&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Layer 3: Spatial Constraint Enforcement (ControlNet)&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#layer-3-spatial-constraint-enforcement-controlnet&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;&lt;strong&gt;Problem:&lt;/strong&gt; Even with T5 and better VAE, semantic drift can override character info during generation.&lt;/p&gt;
&lt;p dir=&quot;auto&quot;&gt;&lt;strong&gt;Solution:&lt;/strong&gt; Don’t ask the model to generate letters — give it the exact stroke positions via ControlNet. Render text as a clean image, pass to ControlNet, which enforces pixel-level geometry at every denoising step.&lt;/p&gt;
&lt;p dir=&quot;auto&quot;&gt;&lt;strong&gt;Result:&lt;/strong&gt; Text is exactly correct, with style applied on top.&lt;/p&gt;
&lt;h2 id=&quot;why-all-three-layers-matter&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Why All Three Layers Matter&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#why-all-three-layers-matter&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Trying to solve with just one or two layers doesn’t work:&lt;/p&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;T5 alone → VAE still blurs strokes&lt;/li&gt;
&lt;li&gt;T5 + 16-channel VAE → semantic drift still happens&lt;/li&gt;
&lt;li&gt;ControlNet alone → no semantic understanding of the rest of the image&lt;/li&gt;
&lt;/ul&gt;
&lt;p dir=&quot;auto&quot;&gt;The three-layer approach works because each addresses a different failure mode at a different architectural level.&lt;/p&gt;
&lt;h2 id=&quot;connection-summary&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connection Summary&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connection-summary&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Layer&lt;/th&gt;
&lt;th&gt;Problem Addressed&lt;/th&gt;
&lt;th&gt;Solution Component&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Character-blind encoding&lt;/td&gt;
&lt;td&gt;T5-XXL encoder&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;2&lt;/td&gt;
&lt;td&gt;Stroke destruction&lt;/td&gt;
&lt;td&gt;16-channel VAE&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;3&lt;/td&gt;
&lt;td&gt;Semantic drift&lt;/td&gt;
&lt;td&gt;ControlNet glyph injection&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;
&lt;h2 id=&quot;related-concepts&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Related Concepts&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#related-concepts&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;&lt;a href=&quot;../../hybrid-pipeline&quot; class=&quot;internal&quot; data-slug=&quot;hybrid-pipeline&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;hybrid-pipeline&lt;/a&gt; — uses this solution in Stage 4-5&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../diffusion-models&quot; class=&quot;internal&quot; data-slug=&quot;diffusion-models&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;diffusion-models&lt;/a&gt; — where the generation happens&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../vae&quot; class=&quot;internal&quot; data-slug=&quot;vae&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;vae&lt;/a&gt; — where compression happens&lt;/li&gt;
&lt;li&gt;&lt;a href=&quot;../../controlnet&quot; class=&quot;internal&quot; data-slug=&quot;controlnet&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;controlnet&lt;/a&gt; — the mechanism for Layer 3&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry><entry>
    <title>Variational Autoencoder (VAE)</title>
    <link href="https://animishraa05.github.io/wiki/imgen/vae" />
    <link rel="alternate" type="text/markdown" href="https://animishraa05.github.io/wiki/imgen/vae.md" />
    <summary>The Problem Generating images at full pixel resolution (1024x1024 = 1M pixels) directly would be computationally infeasible — the model would need to reason about millions of values per image.</summary>
    <published>2026-04-12T00:00:00.000Z</published>
    <updated>2026-04-12T00:00:00.000Z</updated>
    <publishedTime>Apr 12, 2026</publishedTime>
    <updatedTime>Apr 12, 2026</updatedTime>
    <category term="ai" label="ai" />
<category term="ml" label="ml" />
    <author>
      <name>Animesh Mishra</name>
      <email>animesh.mishra818@gmail.com</email>
    </author>
    <content type="html">&lt;h2 id=&quot;the-problem&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;The Problem&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#the-problem&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;Generating images at full pixel resolution (1024x1024 = 1M pixels) directly would be computationally infeasible — the model would need to reason about millions of values per image. Diffusion models need a more efficient representation to work with.&lt;/p&gt;
&lt;h2 id=&quot;core-idea&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Core Idea&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#core-idea&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;p dir=&quot;auto&quot;&gt;A VAE compresses an image into a smaller “latent” representation (e.g., 128x128 = 16K values), works with that compressed version during generation, then decompresses back to full resolution. This compression is where the efficiency comes from, but it’s also where fine details like text strokes can get lost.&lt;/p&gt;
&lt;h2 id=&quot;how-it-works&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;How It Works&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#how-it-works&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;h3 id=&quot;encoder&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Encoder&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#encoder&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;Takes an input image and produces a compressed representation in latent space. Instead of outputting a single vector, it outputs a probability distribution (mean and variance) — this is what “variational” means.&lt;/p&gt;
&lt;h3 id=&quot;latent-space&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Latent Space&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#latent-space&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;The compressed representation lives in a lower-dimensional space. Similar images end up close together in this space. The dimensionality is a tradeoff — smaller = faster but more information loss.&lt;/p&gt;
&lt;h3 id=&quot;decoder&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Decoder&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#decoder&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;Takes a latent representation and reconstructs the original image. During generation, the diffusion model works in latent space, then the decoder converts the result back to pixels.&lt;/p&gt;
&lt;h3 id=&quot;4-channel-vs-16-channel&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;4-channel vs 16-channel&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#4-channel-vs-16-channel&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h3&gt;
&lt;p dir=&quot;auto&quot;&gt;Standard SDXL uses 4 channels in latent space (R, G, B, plus one extra). Flux uses 16 channels. More channels = more capacity to preserve fine details like thin text strokes.&lt;/p&gt;
&lt;h2 id=&quot;key-properties&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Key Properties&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#key-properties&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Lossy compression — some information is always lost&lt;/li&gt;
&lt;li&gt;Latent diffusion (LDM) runs the diffusion process in this compressed space&lt;/li&gt;
&lt;li&gt;16-channel VAE preserves more high-frequency detail than 4-channel&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;connections&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Connections&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#connections&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Built from: &lt;a href=&quot;../../neural-networks&quot; class=&quot;internal&quot; data-slug=&quot;neural-networks&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;neural-networks&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Built into: &lt;a href=&quot;../../diffusion-models&quot; class=&quot;internal&quot; data-slug=&quot;diffusion-models&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;diffusion-models&lt;/a&gt;, &lt;a href=&quot;../../flux-architecture&quot; class=&quot;internal&quot; data-slug=&quot;flux-architecture&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;flux-architecture&lt;/a&gt;&lt;/li&gt;
&lt;li&gt;Related to: &lt;a href=&quot;../../text-rendering-problem&quot; class=&quot;internal&quot; data-slug=&quot;text-rendering-problem&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;text-rendering-problem&lt;/a&gt;&lt;/li&gt;
&lt;/ul&gt;
&lt;h2 id=&quot;edge-cases--gotchas&quot; dir=&quot;auto&quot;&gt;&lt;span class=&quot;highlight-span&quot;&gt;Edge Cases &amp;#x26; Gotchas&lt;/span&gt;&lt;a data-role=&quot;anchor&quot; data-no-popover href=&quot;#edge-cases--gotchas&quot; class=&quot;internal&quot;&gt;&lt;span class=&quot;indicator-hook&quot;&gt;&lt;/span&gt;&lt;svg xmlns=&quot;http://www.w3.org/2000/svg&quot; width=&quot;16&quot; height=&quot;16&quot; viewBox=&quot;0 0 24 24&quot; fill=&quot;none&quot; stroke=&quot;currentColor&quot; stroke-width=&quot;2&quot; stroke-linecap=&quot;round&quot; stroke-linejoin=&quot;round&quot;&gt;&lt;use href=&quot;#github-anchor&quot;&gt;&lt;/use&gt;&lt;/svg&gt;&lt;/a&gt;&lt;/h2&gt;
&lt;ul dir=&quot;auto&quot;&gt;
&lt;li&gt;Fine details (text strokes, fine textures) blur during compression&lt;/li&gt;
&lt;li&gt;4-channel VAEs lose more than 16-channel ones&lt;/li&gt;
&lt;li&gt;Decoder quality matters as much as encoder&lt;/li&gt;
&lt;/ul&gt;</content>
  </entry>
</feed>