Transformer Architecture
Tokens are embedded, passed through a stack of transformer blocks, then projected and softmaxed into the next-token distribution.
Rendering…
Make it your own.
{
"nodes": [
{ "id":"tok", "type":"rounded", "label":"Tokens", "x":40, "y":160, "fill":"#dbeafe" },
{ "id":"emb", "type":"neural-net", "label":"Embedding", "x":240, "y":140, "fill":"#ede9fe" },
{ "id":"tb1", "type":"transformer-block", "label":"Block ×N", "x":450, "y":130, "fill":"#c7d2fe" },
{ "id":"tb2", "type":"transformer-block", "label":"Block", "x":630, "y":130, "fill":"#c7d2fe" },
{ "id":"lin", "type":"rounded", "label":"Linear", "x":830, "y":150, "fill":"#dbeafe" },
{ "id":"sm", "type":"donut-3d", "label":"Softmax", "x":1020, "y":150, "fill":"#dbeafe" },
{ "id":"out", "type":"rounded", "label":"Next token", "x":1220, "y":160, "fill":"#dcfce7" }
],
"edges": [
{ "id":"e1","source":"tok","target":"emb" },
{ "id":"e2","source":"emb","target":"tb1","label":"+ pos" },
{ "id":"e3","source":"tb1","target":"tb2" },
{ "id":"e4","source":"tb2","target":"lin" },
{ "id":"e5","source":"lin","target":"sm","label":"logits" },
{ "id":"e6","source":"sm","target":"out" }
]
}