<?xml version="1.0" encoding="UTF-8"?><rss version="2.0"
	xmlns:content="http://purl.org/rss/1.0/modules/content/"
	xmlns:wfw="http://wellformedweb.org/CommentAPI/"
	xmlns:dc="http://purl.org/dc/elements/1.1/"
	xmlns:atom="http://www.w3.org/2005/Atom"
	xmlns:sy="http://purl.org/rss/1.0/modules/syndication/"
	xmlns:slash="http://purl.org/rss/1.0/modules/slash/"
	>

<channel>
	<title>theory Archives - Urban Geo Analytics</title>
	<atom:link href="https://urbangeoanalytics.com/category/theory/feed/" rel="self" type="application/rss+xml" />
	<link>https://urbangeoanalytics.com/category/theory/</link>
	<description>Spatial Analysis, GeoAI &#38; Machine Learning</description>
	<lastBuildDate>Mon, 17 Aug 2026 18:46:41 +0000</lastBuildDate>
	<language>en-US</language>
	<sy:updatePeriod>
	hourly	</sy:updatePeriod>
	<sy:updateFrequency>
	1	</sy:updateFrequency>
	<generator>https://wordpress.org/?v=7.1</generator>

<image>
	<url>https://urbangeoanalytics.com/wp-content/uploads/2025/11/cropped-logo-urban-geo_512-32x32.png</url>
	<title>theory Archives - Urban Geo Analytics</title>
	<link>https://urbangeoanalytics.com/category/theory/</link>
	<width>32</width>
	<height>32</height>
</image> 
	<item>
		<title>The AI Reading Series · From Perceptrons to Agents · Lecture 2: From Neurons to Machines That Talk: How Connectionism Conquered Language</title>
		<link>https://urbangeoanalytics.com/from-neurons-to-llms-transformer-history/</link>
					<comments>https://urbangeoanalytics.com/from-neurons-to-llms-transformer-history/#respond</comments>
		
		<dc:creator><![CDATA[Joan Perez]]></dc:creator>
		<pubDate>Tue, 11 Aug 2026 11:28:52 +0000</pubDate>
				<category><![CDATA[Getting Started]]></category>
		<category><![CDATA[theory]]></category>
		<category><![CDATA[AI]]></category>
		<category><![CDATA[lecture]]></category>
		<guid isPermaLink="false">https://urbangeoanalytics.com/?p=2928</guid>

					<description><![CDATA[<p>Part 2 of the AI reading series. The previous post ended with AlexNet's 2012 earthquake in image recognition. This one tells the road to language: how researchers turned words into vectors, taught networks to read sequences, discovered attention — and why, in 2017, eight Google researchers decided attention was all you need.</p>
<p>The post <a href="https://urbangeoanalytics.com/from-neurons-to-llms-transformer-history/">The AI Reading Series · From Perceptrons to Agents · Lecture 2: From Neurons to Machines That Talk: How Connectionism Conquered Language</a> appeared first on <a href="https://urbangeoanalytics.com">Urban Geo Analytics</a>.</p>
]]></description>
										<content:encoded><![CDATA[<p><div class="fusion-fullwidth fullwidth-box fusion-builder-row-1 fusion-flex-container has-pattern-background has-mask-background nonhundred-percent-fullwidth non-hundred-percent-height-scrolling" style="--awb-border-radius-top-left:0px;--awb-border-radius-top-right:0px;--awb-border-radius-bottom-right:0px;--awb-border-radius-bottom-left:0px;--awb-flex-wrap:wrap;" id="contenu" ><div class="fusion-builder-row fusion-row fusion-flex-align-items-flex-start fusion-flex-content-wrap" style="max-width:1248px;margin-left: calc(-4% / 2 );margin-right: calc(-4% / 2 );"><div class="fusion-layout-column fusion_builder_column fusion-builder-column-0 fusion_builder_column_1_1 1_1 fusion-flex-column" style="--awb-bg-size:cover;--awb-width-large:100%;--awb-margin-top-large:0px;--awb-spacing-right-large:1.92%;--awb-margin-bottom-large:20px;--awb-spacing-left-large:1.92%;--awb-width-medium:100%;--awb-order-medium:0;--awb-spacing-right-medium:1.92%;--awb-spacing-left-medium:1.92%;--awb-width-small:100%;--awb-order-small:0;--awb-spacing-right-small:1.92%;--awb-spacing-left-small:1.92%;"><div class="fusion-column-wrapper fusion-column-has-shadow fusion-flex-justify-content-flex-start fusion-content-layout-column"><div class="fusion-image-element " style="--awb-caption-title-font-family:var(--h2_typography-font-family);--awb-caption-title-font-weight:var(--h2_typography-font-weight);--awb-caption-title-font-style:var(--h2_typography-font-style);--awb-caption-title-size:var(--h2_typography-font-size);--awb-caption-title-transform:var(--h2_typography-text-transform);--awb-caption-title-line-height:var(--h2_typography-line-height);--awb-caption-title-letter-spacing:var(--h2_typography-letter-spacing);"><span class=" fusion-imageframe imageframe-none imageframe-1 hover-type-none"><img fetchpriority="high" decoding="async" width="1804" height="673" title="ai reading lecture 2" src="https://urbangeoanalytics.com/wp-content/uploads/2026/08/ai-reading-lecture-2.png" alt class="img-responsive wp-image-2975" srcset="https://urbangeoanalytics.com/wp-content/uploads/2026/08/ai-reading-lecture-2-200x75.png 200w, https://urbangeoanalytics.com/wp-content/uploads/2026/08/ai-reading-lecture-2-400x149.png 400w, https://urbangeoanalytics.com/wp-content/uploads/2026/08/ai-reading-lecture-2-600x224.png 600w, https://urbangeoanalytics.com/wp-content/uploads/2026/08/ai-reading-lecture-2-800x298.png 800w, https://urbangeoanalytics.com/wp-content/uploads/2026/08/ai-reading-lecture-2-1200x448.png 1200w, https://urbangeoanalytics.com/wp-content/uploads/2026/08/ai-reading-lecture-2.png 1804w" sizes="(max-width: 640px) 100vw, 1200px" /></span></div></div></div><div class="fusion-layout-column fusion_builder_column fusion-builder-column-1 fusion_builder_column_3_4 3_4 fusion-flex-column" style="--awb-bg-size:cover;--awb-width-large:75%;--awb-margin-top-large:0px;--awb-spacing-right-large:2.56%;--awb-margin-bottom-large:20px;--awb-spacing-left-large:2.56%;--awb-width-medium:75%;--awb-order-medium:0;--awb-spacing-right-medium:2.56%;--awb-spacing-left-medium:2.56%;--awb-width-small:100%;--awb-order-small:0;--awb-spacing-right-small:1.92%;--awb-spacing-left-small:1.92%;" id="contenu" data-scroll-devices="small-visibility,medium-visibility,large-visibility"><div class="fusion-column-wrapper fusion-column-has-shadow fusion-flex-justify-content-flex-start fusion-content-layout-column"><div class="fusion-text fusion-text-1"><p><em>Part 2 of the AI reading series. Read this after <a class="keychainify-checked" href="https://urbangeoanalytics.com/understanding-modern-ai-lecture-1-cardon-neurons-spike-back/">Part 1: Where Machine Learning Came From.</a></em></p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="9:1-9:412;283-694">The paper covered in the previous post of this series ends at the moment neural networks triumph: 2012, image recognition, ImageNet, the AlexNet &#8220;earthquake.&#8221; It gave us a precious reading grid — the <strong>world</strong>, the <strong>calculator</strong>, the <strong>horizon</strong> — and one central idea: connectionist machines won by <em>emptying the calculator</em> of all explicit rules, letting enormous masses of data shape the program themselves.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="11:1-11:707;696-1402">But Cardon and his colleagues talk mostly about <strong>images</strong>. The paper you are about to read next, &#8220;<a class="keychainify-checked" href="https://arxiv.org/abs/1706.03762">Attention Is All You Need</a>&#8220;, is about <strong>language</strong> — machine translation, to be precise. And there is a world of difference between recognizing a rhinoceros in a photo and translating a sentence from French to English. This post tells the story of that road: how, between the mid-1980s and 2017, researchers adapted neural networks to the most &#8220;symbolic&#8221; problem there is — words, sentences, meaning — until they produced the architecture that now dominates all of artificial intelligence: the <strong>Transformer</strong>, the building block of <strong>Large Language Models</strong> (LLMs) such as ChatGPT, Claude, Gemini, and Qwen.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="13:1-13:164;1404-1567">We will start simple and ramp up the technicality progressively. At the end you will find a timeline and a glossary: keep them at hand while reading Vaswani et al.</p>
</div><div class="fusion-title title fusion-title-1 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">1. The problem images never posed: sequence</span></h2></div><div class="fusion-text fusion-text-2 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="19:1-19:436;1622-2057">Let&#8217;s pick up where the previous paper left off. In 2012, a convolutional neural network (CNN) crushes the ImageNet competition. Why does <strong>convolution</strong> work so well on images? Because an image is a <em>spatial</em> object: a rhinoceros is still a rhinoceros whether it stands on the left or the right of the photo. The CNN exploits this property by sliding small filters across the image, like a magnifying glass sweeping over a photograph.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="21:1-21:115;2059-2173">Language poses a different problem. A sentence is a <strong>sequence</strong>: a <em>temporal</em>, ordered object of variable length.</p>
<ul class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-disc flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr" data-sourcepos="23:1-25:85;2175-2611">
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="23:1-23:112;2175-2286">&#8220;The dog bites the man&#8221; and &#8220;The man bites the dog&#8221; contain the same words, but <strong>order changes everything</strong>.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="24:1-24:240;2287-2526">The meaning of a word can depend on words that are far away: in &#8220;The key that I left on the kitchen table last night <strong>has disappeared</strong>,&#8221; the verb agrees with &#8220;the key,&#8221; a dozen words earlier. This is called a <strong>long-range dependency</strong>.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="25:1-25:85;2527-2611">A sentence can be 3 words or 300: the network must accept inputs of variable size.</li>
</ul>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="27:1-27:209;2613-2821">Remember the formula from the previous post: for connectionists, the goal is to &#8220;put the world into a vector.&#8221; Two questions immediately arise, and the whole story that follows is a series of answers to them:</p>
<ol class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-decimal flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr" data-sourcepos="29:1-30:92;2823-2988">
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="29:1-29:74;2823-2896"><strong>How do you turn a word into a vector?</strong> (the representation problem)</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="30:1-30:92;2897-2988"><strong>How do you process a variable-length sequence of vectors?</strong> (the architecture problem)</li>
</ol>
</div><div class="fusion-title title fusion-title-2 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">2. First answer: turning words into vectors</span></h2></div><div class="fusion-text fusion-text-3 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="mt-2 -mb-1 text-base font-bold" dir="ltr" data-sourcepos="36:1-36:52;3043-3094"><strong>The word as a number (and why it is not enough)</strong></p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="38:1-38:496;3096-3591">The naive approach gives each dictionary word a number: &#8220;cat&#8221; = 4,812, &#8220;dog&#8221; = 1,953. Technically, one uses a so-called <em>one-hot</em> vector: a huge vector of zeros with a single 1 at the word&#8217;s position. The problem? In this representation, &#8220;cat&#8221; is no closer to &#8220;dog&#8221; than to &#8220;umbrella.&#8221; All resemblance between words is lost. It is a <strong>symbolic</strong> representation in the sense of the previous post: each word is a discrete symbol, identical to or different from another, with no internal structure.</p>
<p class="mt-2 -mb-1 text-base font-bold" dir="ltr" data-sourcepos="40:1-40:45;3593-3637"><strong>Word2vec (2013): meaning as neighborhood</strong></p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="42:1-42:355;3639-3993">The solution was already hinted at in Cardon&#8217;s paper: <strong>word2vec</strong> (Mikolov et al., 2013). The idea rests on an old linguists&#8217; intuition, summed up by John R. Firth in 1957: <em>&#8220;You shall know a word by the company it keeps.&#8221;</em> &#8220;Cat&#8221; and &#8220;dog&#8221; appear in similar contexts (&#8220;feed the ___&#8221;, &#8220;the ___ is sleeping&#8221;); they should therefore receive nearby vectors.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="44:1-44:322;3995-4316">Word2vec trains a small neural network on a simple task: guess a word from its neighbors (or the reverse). The network is not the point; what you keep are the vectors learned along the way, called <strong>embeddings</strong>. Each word becomes a point in a space of a few hundred dimensions, and this space has astonishing properties:</p>
<ul class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-disc flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr" data-sourcepos="46:1-47:119;4318-4493">
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="46:1-46:57;4318-4374">words that are close in meaning are close in distance;</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="47:1-47:119;4375-4493">some directions of the space <em>mean</em> something: the famous computation <strong>king − man + woman ≈ queen</strong> actually works.</li>
</ul>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="49:1-49:362;4495-4856">Now reread that sentence from Cardon&#8217;s paper: semantic proximity is not deduced from a symbolic categorization, but induced from statistical neighborhoods. This is exactly the symbolic-to-connectionist reversal, applied to the meaning of words. No linguist wrote a rule; meaning emerged from data. The first of our two problems — representing words — is solved.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="51:1-51:368;4858-5225">But word2vec has a serious limitation: each word gets <strong>one single, fixed vector</strong>. Yet &#8220;bank&#8221; does not mean the same thing in &#8220;I called my bank&#8221; and &#8220;I sat on the river bank.&#8221; What we would need are <strong>contextual</strong> representations that change with the sentence. Keep this limitation in mind: the Transformer is, among other things, the machine that will blow it away.</p>
</div><div class="fusion-title title fusion-title-3 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">3. Second answer: recurrent networks, a memory that reads word by word</span></h2></div><div class="fusion-text fusion-text-4 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="mt-2 -mb-1 text-base font-bold" dir="ltr" data-sourcepos="57:1-57:42;5307-5348"><strong>The RNN: a loop over time (1986–1990)</strong></p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="59:1-59:295;5350-5644">There remained the second problem: processing a variable-length sequence. The historical answer is the <strong>recurrent neural network</strong> (RNN), popularized by Jeffrey Elman in 1990 and already present in the work of the PDP group of Rumelhart and Hinton (1986) — the same group as in Cardon&#8217;s paper.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="61:1-61:258;5646-5903">The idea is elegant: the network reads the sequence <strong>one word at a time</strong>, left to right, and maintains a <strong>hidden state</strong> — a vector acting as working memory. At each word, the network combines what it reads with what it remembers, and updates its memory:</p>
<blockquote class="ml-2 border-l-4 border-&#091;hsl(var(--border-300)/0.1)&#093; pl-4 text-text-300" data-sourcepos="63:1-63:40;5905-5944">
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="63:3-63:40;5907-5944">memory(t) = f( memory(t−1), word(t) )</p>
</blockquote>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="65:1-65:229;5946-6174">Picture someone reading a sentence under their breath while keeping a mental summary they revise at every word. That is an RNN. The architecture naturally accepts sentences of any length: you simply loop for more or fewer steps.</p>
<p class="mt-2 -mb-1 text-base font-bold" dir="ltr" data-sourcepos="67:1-67:35;6176-6210"><strong>The vanishing gradient problem</strong></p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="69:1-69:540;6212-6751">In practice, simple RNNs have a crippling flaw. Remember <strong>backpropagation</strong>: to learn, the error is propagated backwards through the network. In an RNN, &#8220;backwards&#8221; means <strong>back through time</strong>, across as many steps as there are words. At each step, the error signal is multiplied by coefficients; over a long sentence it gets multiplied dozens of times by numbers that are often smaller than 1&#8230; and it melts like snow in the sun. This is the famous <strong>vanishing gradient</strong> problem, identified notably by Sepp Hochreiter as early as 1991.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="71:1-71:188;6753-6940">The concrete consequence: the network cannot learn long-range dependencies. By the time it reaches the verb &#8220;has disappeared,&#8221; it has &#8220;forgotten&#8221; the key at the beginning of the sentence.</p>
<p class="mt-2 -mb-1 text-base font-bold" dir="ltr" data-sourcepos="73:1-73:41;6942-6982"><strong>The LSTM (1997): a memory with gates</strong></p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="75:1-75:510;6984-7493">The most celebrated solution arrives in 1997: the <strong>LSTM</strong> (<em>Long Short-Term Memory</em>), by Sepp Hochreiter and Jürgen Schmidhuber. The idea: equip the memory cell with learned <strong>gates</strong> — small mechanisms that decide, at each step, what to <strong>write</strong> into memory, what to <strong>forget</strong>, and what to <strong>read</strong>. A kind of whiteboard with a doorkeeper choosing what gets noted and what gets erased. Thanks to an internal &#8220;conveyor belt&#8221; where information circulates almost untouched, the gradient survives much longer.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="77:1-77:331;7495-7825">LSTMs (and their simplified cousin, the <strong>GRU</strong>, 2014) would dominate natural language processing for twenty years. They are a perfect example of what Cardon&#8217;s paper calls the work on <strong>hyper-parameters</strong> and architecture: you do not dictate grammar rules to the network — you <em>sculpt</em> its structure so it can learn what it needs.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="79:1-79:321;7827-8147">Around 2014–2015, boosted by GPUs and large corpora, LSTMs power the speech recognition in your phone and the first neural versions of Google Translate. Connectionism, victorious over images in 2012, is winning over language. But it is precisely by pushing LSTMs to their limits that the missing link will be discovered.</p>
</div><div class="fusion-title title fusion-title-4 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">4. Translating: the seq2seq model and its bottleneck</span></h2></div><div class="fusion-text fusion-text-5 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="mt-2 -mb-1 text-base font-bold" dir="ltr" data-sourcepos="85:1-85:27;8211-8237"><strong>Encoder–decoder (2014)</strong></p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="87:1-87:313;8239-8551">Machine translation is THE queen of tasks, because it demands everything: understanding one sequence and producing another, of a different length. In 2014, two teams (Sutskever, Vinyals and Le at Google; Cho and Bengio in Montreal) propose the <strong>seq2seq</strong> (<em>sequence-to-sequence</em>) architecture, made of two RNNs:</p>
<ul class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-disc flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr" data-sourcepos="89:1-90:181;8553-8918">
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="89:1-89:185;8553-8737">an <strong>encoder</strong> reads the source sentence (&#8220;Le chat dort&#8221;) and compresses it into <strong>a single vector</strong> — a numerical summary of the whole sentence, sometimes called a &#8220;thought vector&#8221;;</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="90:1-90:181;8738-8918">a <strong>decoder</strong> starts from this vector and generates the target sentence word by word (&#8220;The,&#8221; then &#8220;cat,&#8221; then &#8220;sleeps&#8221;), each produced word being fed back in to produce the next.</li>
</ul>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="92:1-92:232;8920-9151">Note this word-by-word generation mechanism, each word conditioned on the previous ones: it is called <strong>auto-regressive</strong>, and it is <em>exactly</em> how ChatGPT or Claude write their answers today. This point from 2014 has never changed.</p>
<p class="mt-2 -mb-1 text-base font-bold" dir="ltr" data-sourcepos="94:1-94:19;9153-9171"><strong>The bottleneck</strong></p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="96:1-96:364;9173-9536">But seq2seq has an obvious Achilles heel: the <strong>entire</strong> source sentence, whether 5 or 60 words long, must fit into a single fixed-size vector. It is like asking a translator to read a whole paragraph, close the book, then translate from memory without ever reopening it. On long sentences, quality collapses. Researchers call this the information <strong>bottleneck</strong>.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="98:1-98:68;9538-9605">The solution will give its name to the paper you are about to read.</p>
</div><div class="fusion-title title fusion-title-5 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">5. Attention: letting the network look wherever it wants</span></h2></div><div class="fusion-text fusion-text-6 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="104:1-104:244;9685-9928">In 2014, Dzmitry Bahdanau, Kyunghyun Cho and Yoshua Bengio (him again — find him in the connectionist core of Cardon&#8217;s figure 2) publish an idea that changes everything: what if, instead of closing the book, the decoder could <strong>keep it open</strong>?</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="106:1-106:550;9930-10479">Concretely: the encoder no longer produces a single vector, but keeps one vector <strong>per word</strong> of the source sentence. Then, for every word it generates, the decoder computes a <strong>relevance score</strong> between what it is currently doing and each of the source words. These scores, turned into percentages (via a function called <em>softmax</em>), are used to build a weighted average of the source vectors: the decoder &#8220;focuses&#8221; on the words that are useful at that instant. To produce &#8220;sleeps,&#8221; it looks mostly at &#8220;dort.&#8221; This mechanism is called <strong>attention</strong>.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="108:1-108:88;10481-10568">Three things to remember, because they directly prepare your reading of Vaswani et al.:</p>
<ol class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-decimal flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr" data-sourcepos="110:1-112:223;10570-11193">
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="110:1-110:266;10570-10835">Attention is <strong>learned</strong>, not programmed. Nobody wrote a French–English alignment rule: the network discovers by itself that &#8220;dort&#8221; explains &#8220;sleeps.&#8221; Once again, the inductive move described by Cardon — empty the calculator, let the world provide the structure.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="111:1-111:135;10836-10970">Attention is <strong>soft</strong>: it is not a binary choice but a weighting, therefore it is differentiable, therefore backprop applies to it.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="112:1-112:223;10971-11193">Attention creates <strong>shortcuts</strong>: every generated word is directly connected to every source word, without going through the fragile chain of recurrent memory. Long-range dependencies become connections&#8230; of distance 1.</li>
</ol>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="114:1-114:548;11195-11742">Between 2015 and 2017, the &#8220;LSTM + attention&#8221; recipe becomes the world state of the art in translation (Google deploys it at the end of 2016). But an irritation grows among engineers, and it is a <em>hardware</em> one — remember the role of GPUs in Cardon&#8217;s story. An RNN reads a sentence <strong>word after word</strong>: computing word 50 must wait for word 49. Yet GPUs are massively <strong>parallel</strong> machines, built to perform millions of operations <em>at the same time</em>. Recurrence wastes the hardware and forbids training on truly gigantic corpora in reasonable time.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="116:1-116:116;11744-11859">Hence a question asked ever more insistently in the labs: now that we have attention, what is recurrence still for?</p>
</div><div class="fusion-title title fusion-title-6 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">6. &#8220;Attention Is All You Need&#8221;</span></h2></div><div class="fusion-text fusion-text-7 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="122:1-122:384;11907-12290">The answer from eight Google researchers (Vaswani, Shazeer, Parmar, Uszkoreit, Jones, Gomez, Kaiser, Polosukhin) fits in their slightly provocative title: <em>attention is all you need</em>. Their architecture, the <strong>Transformer</strong>, purely and simply removes recurrence (and convolution). Here, as a preview, are the ideas you will meet in the paper — consider this section your reading map.</p>
<p class="mt-2 -mb-1 text-base font-bold" dir="ltr" data-sourcepos="124:1-124:19;12292-12310"><strong>Self-attention</strong></p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="126:1-126:452;12312-12763">Until now, attention connected the decoder to the encoder (target to source). The Transformer&#8217;s stroke of genius is to apply it <strong>inside a single sentence</strong>: every word looks at every other word of its own sentence — including itself — to build its representation. In &#8220;The animal didn&#8217;t cross the street because <strong>it</strong> was tired,&#8221; the word &#8220;it&#8221; learns to attend to &#8220;the animal.&#8221; In one operation, each word enriches its meaning with its whole context.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="128:1-128:208;12765-12972">Notice what this solves: a word&#8217;s vector is no longer fixed as in word2vec — it is <strong>contextual</strong>, recomputed for every sentence. &#8220;Bank&#8221; will not have the same representation at the counter and by the river.</p>
<p class="mt-2 -mb-1 text-base font-bold" dir="ltr" data-sourcepos="130:1-130:44;12974-13017"><strong>Query, Key, Value: the library metaphor</strong></p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="132:1-132:560;13019-13578">The paper formalizes attention with three vectors per word, obtained through three learned transformations: a <strong>Query</strong>, a <strong>Key</strong>, and a <strong>Value</strong>. The classic image: in a library, you arrive with a question (your <em>query</em>), you compare it to the labels on the spines of the books (the <em>keys</em>), and you leave with the content of the relevant books (the <em>values</em>), in proportion to their relevance. Mathematically: dot products between Q and K → scores → softmax → weighted average of the V. That is equation (1) of the paper, the only truly indispensable one:</p>
<blockquote class="ml-2 border-l-4 border-&#091;hsl(var(--border-300)/0.1)&#093; pl-4 text-text-300" data-sourcepos="134:1-134:45;13580-13624">
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="134:3-134:45;13582-13624">Attention(Q, K, V) = softmax(QKᵀ / √d) · V</p>
</blockquote>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="136:1-136:166;13626-13791">The √d in the denominator is a mere numerical safeguard (keeping the scores from growing too large). Everything else in the paper is engineering around this formula.</p>
<p class="mt-2 -mb-1 text-base font-bold" dir="ltr" data-sourcepos="138:1-138:52;13793-13844"><strong>Multi-head, position, and the full architecture</strong></p>
<ul class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-disc flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr" data-sourcepos="140:1-143:298;13846-15137">
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="140:1-140:256;13846-14101"><strong>Multi-head attention</strong>: rather than one attention, you run 8 (or 16, or 96&#8230;) in parallel, each with its own Q, K, V. Each &#8220;head&#8221; specializes — one tracks syntax, another coreference, and so on — without anyone asking it to. Emergent behavior, again.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="141:1-141:378;14102-14479"><strong>Positional encoding</strong>: removing recurrence has a cost — the network no longer knows in what order the words are! (Attention is a set computation, insensitive to order.) The fix: add to each embedding a small vector encoding its position, built from sines and cosines of varying frequencies. Do not drown in the formulas: just remember that we <em>re-inject</em> the order we lost.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="142:1-142:360;14480-14839"><strong>Stacking</strong>: one Transformer &#8220;block&#8221; = self-attention + a small classical network (<em>feed-forward</em>), with two training stabilizers (residual connections and normalization). These blocks are stacked: 6 in the 2017 paper, 96 and more in today&#8217;s LLMs. The original paper keeps the encoder–decoder structure inherited from seq2seq, since it targets translation.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="143:1-143:298;14840-15137"><strong>Parallelism</strong>: all words are processed <strong>at the same time</strong>. No more sequential waiting: the Transformer fits GPUs perfectly. This is the decisive argument — less training time, more data ingested. Reread Cardon: the connectionist victory has always been as much about hardware as about ideas.</li>
</ul>
</div><div class="fusion-title title fusion-title-7 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">7. From the Transformer to the LLM</span></h2></div><div class="fusion-text fusion-text-8 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="149:1-149:130;15215-15344">The 2017 paper is about translation. How do we get to ChatGPT? Through an idea of disarming simplicity: <strong>next-word prediction</strong>.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="151:1-151:426;15346-15771">Take a text, hide the next word, ask the model to guess it, correct it with backprop, repeat — trillions of times, over the whole web. No human labels are needed: the text is its own correction. This is called <strong>self-supervised</strong> learning. Think back to Cardon&#8217;s &#8220;horizon&#8221;: here, the horizon of the computation (the next word) is supplied by the world itself, for free, at unlimited scale. It is the ultimate cybernetic loop.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="153:1-153:20;15773-15792">The key milestones:</p>
<ul class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-disc flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr" data-sourcepos="155:1-159:481;15794-17724">
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="155:1-155:350;15794-16143"><strong>2018 — GPT-1</strong> (OpenAI): keep only the <strong>decoder</strong> of the Transformer, trained to predict the next word, then fine-tuned on specific tasks. <strong>BERT</strong> (Google) makes the opposite choice — keep only the <strong>encoder</strong>, trained to guess masked words — and crushes the comprehension benchmarks. The two descendants of the 2017 paper divide up the world.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="156:1-156:275;16144-16418"><strong>2019 — GPT-2</strong>: same recipe, ×10 on size (1.5 billion parameters). Surprise: the model can summarize, translate, answer questions <em>without having been trained to</em> — simply because predicting the next word across the whole Internet forces it to learn a bit of everything.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="157:1-157:396;16419-16814"><strong>2020 — GPT-3 and the scaling laws</strong> (Kaplan et al.): performance grows in a regular, <em>predictable</em> way with three ingredients — parameters, data, compute. The message: bigger is better, and nobody sees the ceiling yet. GPT-3 (175 billion parameters) reveals <strong>in-context learning</strong>: give it two examples inside the question (the <em>prompt</em>), and it picks up the pattern without any retraining.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="158:1-158:429;16815-17243"><strong>2022 — ChatGPT and RLHF</strong>: a raw LLM completes text; it does not &#8220;answer.&#8221; It is aligned in two stages: fine-tuning on dialogues written by humans, then <strong>reinforcement learning from human feedback</strong> (RLHF) — annotators rank answers, the model learns to aim for the best-ranked ones. Note the historical irony: the horizon of the computation, expelled from symbolic rules, comes back through the door of <em>human preferences</em>.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="159:1-159:481;17244-17724"><strong>2023–today</strong>: the era of open models (Meta&#8217;s Llama, Mistral, Alibaba&#8217;s <strong>Qwen</strong>), of multimodal models (text + image + sound), and of the Transformer&#8217;s extension to&#8230; image and video generation. Modern diffusion models (Stable Diffusion 3, Flux, Qwen-Image, Sora, Wan) have also replaced their old internal networks with Transformers (the so-called <strong>DiT</strong>, <em>Diffusion Transformer</em>, architecture). The 2017 architecture has become the universal architecture of connectionism.</li>
</ul>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="161:1-161:806;17726-18531">One last thing, to close the loop with Cardon. The final table of his paper describes deep learning as: <em>world</em> = vectors of massive data, <em>calculator</em> = deep network, <em>horizon</em> = error optimization on an objective. The LLM is its most extreme culmination: the world is <strong>all the text ever written</strong>, the calculator is a Transformer with hundreds of billions of coefficients, and the horizon fits in three words — <strong>predict the next word</strong>. That capacities for reasoning, translation and dialogue <em>emerge</em> from such a poor objective is perhaps the most beautiful posthumous victory of the connectionist camp — and the philosophical question remains wide open, exactly where Cardon left it: what does a machine &#8220;understand&#8221; when its entire thought is, in Hinton&#8217;s phrase, &#8220;a big vector of neural activity&#8221;?</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="167:1-167:45;18585-18629">The paper is 11 dense pages. Reading advice:</p>
<ol class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-decimal flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr" data-sourcepos="169:1-172:138;18631-19387">
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="169:1-169:260;18631-18890"><strong>Read in this order</strong>: the abstract → the introduction (§1) → figure 1 (the architecture — keep it in view at all times) → §3.2 (attention, the heart of the paper) → the conclusion. The rest (§5–6, training details and translation results) can be skimmed.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="170:1-170:167;18891-19057"><strong>Do not get stuck on the math.</strong> Only one equation matters (Attention(Q,K,V), equation 1), and you already know its intuition: query → labels → weighted contents.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="171:1-171:192;19058-19249"><strong>Spot the words you now own</strong>: <em>recurrent</em>, <em>sequential computation</em>, <em>long-range dependencies</em>, <em>parallelizable</em> — you now know why they are there and what the paper is fighting against.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="172:1-172:138;19250-19387"><strong>A question to hold onto as you close the paper</strong>: why does removing recurrence make the addition of positional encoding <em>mandatory</em>?</li>
</ol>
</div><div class="fusion-title title fusion-title-8 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:20px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">Timeline</span></h2></div>
<div class="table-1">
<table width="100%">
<thead>
<tr>
<th align="left">Year</th>
<th align="left">Event</th>
<th align="left">Why it matters</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">1943</td>
<td align="left"> Formal neuron (McCulloch &amp; Pitts)</td>
<td align="left">The atom of connectionism</td>
</tr>
<tr>
<td align="left">1957</td>
<td align="left">Perceptron (Rosenblatt)</td>
<td align="left">First learning machine</td>
</tr>
<tr>
<td align="left">1969</td>
<td align="left"><em>Perceptrons</em> (Minsky &amp; Papert)</td>
<td align="left">The &#8220;excommunication,&#8221; connectionist winter</td>
</tr>
<tr>
<td align="left">1986</td>
<td align="left">Backprop popularized (Rumelhart, Hinton, Williams)</td>
<td align="left">Deep networks become trainable</td>
</tr>
<tr>
<td align="left">1990</td>
<td align="left">Elman&#8217;s RNN</td>
<td align="left">The network that reads sequences</td>
</tr>
<tr>
<td align="left">1991/1997</td>
<td align="left">Vanishing gradient identified / LSTM (Hochreiter &amp; Schmidhuber)</td>
<td align="left">A memory that goes the distance</td>
</tr>
<tr>
<td align="left">2012</td>
<td align="left">AlexNet wins ImageNet</td>
<td align="left">The &#8220;earthquake&#8221; — where the previous lecture ends</td>
</tr>
<tr>
<td align="left">2013</td>
<td align="left">word2vec (Mikolov)</td>
<td align="left">Words become vectors of meaning</td>
</tr>
<tr>
<td align="left">2014</td>
<td align="left"> seq2seq (Sutskever; Cho)</td>
<td align="left">Encoder–decoder, auto-regressive generation</td>
</tr>
<tr>
<td align="left">2014–15</td>
<td align="left">Attention (Bahdanau, Cho, Bengio)</td>
<td align="left">The decoder keeps the book open</td>
</tr>
<tr>
<td align="left">2016</td>
<td align="left">Google Translate goes neural</td>
<td align="left">Connectionism conquers mainstream language</td>
</tr>
<tr>
<td align="left">2017</td>
<td align="left">&#8220;Attention Is All You Need&#8221; (Vaswani et al.)</td>
<td align="left">The Transformer: attention alone, parallel compute</td>
</tr>
<tr>
<td align="left">2018</td>
<td align="left">GPT-1 (decoder) and BERT (encoder)</td>
<td align="left">The two lineages of the Transformer</td>
</tr>
<tr>
<td align="left">2019</td>
<td align="left">GPT-2</td>
<td align="left">The &#8220;free&#8221; capabilities of next-word prediction</td>
</tr>
<tr>
<td align="left">2020</td>
<td align="left">GPT-3, scaling laws</td>
<td align="left"> Bigger = predictably better</td>
</tr>
<tr>
<td align="left">2022</td>
<td align="left">ChatGPT (RLHF)</td>
<td align="left">The LLM becomes a conversational assistant</td>
</tr>
<tr>
<td align="left">2023</td>
<td align="left">Llama, Mistral, Qwen, DiT</td>
<td align="left">Open models; the Transformer invades image/video diffusion</td>
</tr>
</tbody>
</table>
</div>
<div class="fusion-title title fusion-title-9 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">Glossary</span></h2></div><div class="fusion-text fusion-text-9 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="202:1-202:205;20890-21094"><strong>Attention</strong> — A learned mechanism that computes, for a given element, relevance weights over a set of other elements, then takes their weighted average. Enables direct connections between distant words.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="204:1-204:161;21096-21256"><strong>Auto-regressive</strong> — Generation mode in which each new word is produced conditioned on all previous ones, one by one. This is how GPT, Claude and Qwen &#8220;write.&#8221;</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="206:1-206:123;21258-21380"><strong>Backpropagation</strong> — The algorithm (1986) that propagates the error from output to input to adjust the network&#8217;s weights.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="208:1-208:129;21382-21510"><strong>BERT</strong> (2018) — Encoder-only model, trained to guess masked words; excellent at <em>understanding</em> text (classification, search).</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="210:1-210:115;21512-21626"><strong>Context window</strong> — The maximum number of tokens the model can consider at once. A major practical limit of LLMs.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="212:1-212:124;21628-21751"><strong>Decoder</strong> — The half of the Transformer that <em>generates</em> the output sequence. GPT and most current LLMs are decoder-only.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="214:1-214:147;21753-21899"><strong>Embedding</strong> — Representation of an object (word, image, molecule&#8230;) as a dense vector in a space where geometric proximity reflects similarity.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="216:1-216:90;21901-21990"><strong>Encoder</strong> — The half of the Transformer that <em>reads</em> and represents the input sequence.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="218:1-218:105;21992-22096"><strong>GRU / LSTM</strong> — Gated recurrent cells (1997 for the LSTM) that protect information over long stretches.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="220:1-220:150;22098-22247"><strong>Hyper-parameters</strong> — Architecture and training choices set by humans (number of layers, heads, learning rate&#8230;), as opposed to learned parameters.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="222:1-222:154;22249-22402"><strong>In-context learning</strong> — A model&#8217;s ability (from GPT-3 onwards) to pick up a task from a few examples placed directly in the prompt, without retraining.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="224:1-224:151;22404-22554"><strong>LLM (Large Language Model)</strong> — A Transformer (usually decoder-only) with billions of parameters, trained by next-word prediction on immense corpora.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="226:1-226:157;22556-22712"><strong>Long-range dependency</strong> — A grammatical or semantic link between words that are far apart in a sentence. Achilles heel of RNNs, strong point of attention.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="228:1-228:138;22714-22851"><strong>Multi-head attention</strong> — Running several independent attentions (&#8220;heads&#8221;) in parallel, each free to specialize in one type of relation.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="230:1-230:160;22853-23012"><strong>Parameters (weights)</strong> — The coefficients adjusted during learning. GPT-3: 175 billion. These are what you download when you fetch a model from Hugging Face.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="232:1-232:114;23014-23127"><strong>Positional encoding</strong> — Vectors added to the embeddings to re-inject word order, which attention alone ignores.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="234:1-234:123;23129-23251"><strong>Prompt</strong> — The input text given to the LLM; since GPT-3, you &#8220;program&#8221; the model in natural language through the prompt.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="236:1-236:177;23253-23429"><strong>Query / Key / Value (Q, K, V)</strong> — The three learned projections of each word used in the attention computation: the question asked, the label compared, the content retrieved.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="238:1-238:142;23431-23572"><strong>RLHF</strong> — Reinforcement Learning from Human Feedback: aligning an LLM with human preferences via reinforcement (the basis of ChatGPT, 2022).</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="240:1-240:164;23574-23737"><strong>RNN (recurrent neural network)</strong> — A network that processes a sequence step by step while maintaining a hidden state (memory). Dominant in NLP from 1990 to 2017.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="242:1-242:131;23739-23869"><strong>Scaling laws</strong> — Empirical relations (2020) showing that LLM performance improves predictably with model size, data and compute.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="244:1-244:128;23871-23998"><strong>Self-supervised learning</strong> — Learning without human labels: the data provides its own target (e.g., the next word of a text).</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="246:1-246:169;24000-24168"><strong>Seq2seq</strong> — The encoder–decoder architecture (2014) turning one sequence into another; the framework of neural translation and the direct ancestor of the Transformer.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="248:1-248:237;24170-24406"><strong>Softmax</strong> — The function that converts a list of scores into a probability distribution (percentages summing to 100%). Used inside attention and to pick the next word — it is on this function that the <em>temperature</em> parameter operates.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="250:1-250:110;24408-24517"><strong>Token</strong> — The elementary unit of text a model manipulates (often a word fragment, ~¾ of a word on average).</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="252:1-252:190;24519-24708"><strong>Transformer</strong> (2017) — Architecture based solely on attention (no recurrence, no convolution), massively parallelizable; the foundation of all LLMs and, by now, of diffusion models (DiT).</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr" data-sourcepos="254:1-254:161;24710-24870"><strong>Word2vec</strong> (2013) — A method producing static word embeddings from their contexts of occurrence; the ancestor of the Transformer&#8217;s contextual representations.</p>
</div><div class="fusion-title title fusion-title-10 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:-10px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">Watch After Reading</span></h2></div><div class="fusion-text fusion-text-10" style="--awb-content-alignment:justify;"><ul>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="260:1-260:199;24901-25099"><strong>3Blue1Brown — the Transformer chapters of the Deep Learning series</strong>: &#8220;<a class="keychainify-checked" href="https://www.youtube.com/watch?v=wjZofJX0v4M">But what is a GPT?</a>&#8221; then &#8220;<a class="keychainify-checked" href="https://www.youtube.com/watch?v=eMlx5fFNoYc">Attention in transformers, step-by-step.</a>&#8221; The finest visualization of equation (1) in existence.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2" data-sourcepos="261:1-261:198;25100-25297"><strong>&#8220;<a class="keychainify-checked" href="https://www.youtube.com/watch?v=kCc8FmEb1nY">Let&#8217;s build GPT: from scratch, in code, spelled out</a>&#8221; — Andrej Karpathy</strong> (~2 h): building a GPT by following the paper, from tokenization to multi-head attention. Watch it with the code open.</li>
</ul>
</div></div></div><div class="fusion-layout-column fusion_builder_column fusion-builder-column-2 awb-sticky awb-sticky-medium awb-sticky-large fusion_builder_column_1_4 1_4 fusion-flex-column" style="--awb-padding-top:20px;--awb-padding-right:20px;--awb-padding-bottom:20px;--awb-padding-left:20px;--awb-bg-size:cover;--awb-border-color:var(--awb-color6);--awb-border-style:solid;--awb-width-large:25%;--awb-margin-top-large:0px;--awb-spacing-right-large:7.68%;--awb-margin-bottom-large:20px;--awb-spacing-left-large:7.68%;--awb-width-medium:25%;--awb-order-medium:0;--awb-spacing-right-medium:7.68%;--awb-spacing-left-medium:7.68%;--awb-width-small:100%;--awb-order-small:0;--awb-spacing-right-small:1.92%;--awb-spacing-left-small:1.92%;--awb-sticky-offset:150px;" data-scroll-devices="small-visibility,medium-visibility,large-visibility"><div class="fusion-column-wrapper fusion-column-has-shadow fusion-flex-justify-content-flex-start fusion-content-layout-column"><div class="fusion-text fusion-text-11"><p><span style="color: #143c4e;"><strong>Table of contents</strong></span></p>
</div><div class="awb-toc-el awb-toc-el--1" data-awb-toc-id="1" data-awb-toc-options="{&quot;allowed_heading_tags&quot;:{&quot;h2&quot;:0},&quot;ignore_headings&quot;:&quot;&quot;,&quot;ignore_headings_words&quot;:&quot;&quot;,&quot;enable_cache&quot;:&quot;no&quot;,&quot;highlight_current_heading&quot;:&quot;yes&quot;,&quot;hide_hidden_titles&quot;:&quot;no&quot;,&quot;limit_container&quot;:&quot;page_content&quot;,&quot;select_custom_headings&quot;:&quot;.contenu H2, .contenu H3&quot;,&quot;icon&quot;:&quot;fa-flag fas&quot;,&quot;counter_type&quot;:&quot;none&quot;}" style="--awb-item-padding-right:5px;--awb-item-padding-left:5px;"><div class="awb-toc-el__content"></div></div><div class="fusion-separator fusion-full-width-sep" style="align-self: center;margin-left: auto;margin-right: auto;margin-top:20px;margin-bottom:20px;width:100%;"><div class="fusion-separator-border sep-single sep-solid" style="--awb-height:20px;--awb-amount:20px;--awb-sep-color:var(--awb-color6);border-color:var(--awb-color6);border-top-width:1px;"></div></div><div class="fusion-image-element " style="--awb-margin-top:25px;--awb-margin-bottom:25px;--awb-caption-title-font-family:var(--h2_typography-font-family);--awb-caption-title-font-weight:var(--h2_typography-font-weight);--awb-caption-title-font-style:var(--h2_typography-font-style);--awb-caption-title-size:var(--h2_typography-font-size);--awb-caption-title-transform:var(--h2_typography-text-transform);--awb-caption-title-line-height:var(--h2_typography-line-height);--awb-caption-title-letter-spacing:var(--h2_typography-letter-spacing);--awb-filter:saturate(100%);--awb-filter-transition:filter 0.3s ease;--awb-filter-hover:saturate(0%);"><span class=" fusion-imageframe imageframe-none imageframe-2 hover-type-zoomout"><img decoding="async" width="1536" height="1024" title="blog lvl1" src="https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1.png" alt class="img-responsive wp-image-1685" srcset="https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1-200x133.png 200w, https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1-400x267.png 400w, https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1-600x400.png 600w, https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1-800x533.png 800w, https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1-1200x800.png 1200w, https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1.png 1536w" sizes="(max-width: 640px) 100vw, 400px" /></span></div></div></div></div></div><div class="fusion-fullwidth fullwidth-box fusion-builder-row-2 fusion-flex-container has-pattern-background has-mask-background nonhundred-percent-fullwidth non-hundred-percent-height-scrolling" style="--awb-border-radius-top-left:0px;--awb-border-radius-top-right:0px;--awb-border-radius-bottom-right:0px;--awb-border-radius-bottom-left:0px;--awb-flex-wrap:wrap;" ><div class="fusion-builder-row fusion-row fusion-flex-align-items-flex-start fusion-flex-content-wrap" style="max-width:1248px;margin-left: calc(-4% / 2 );margin-right: calc(-4% / 2 );"></div></div></p>
<p>The post <a href="https://urbangeoanalytics.com/from-neurons-to-llms-transformer-history/">The AI Reading Series · From Perceptrons to Agents · Lecture 2: From Neurons to Machines That Talk: How Connectionism Conquered Language</a> appeared first on <a href="https://urbangeoanalytics.com">Urban Geo Analytics</a>.</p>
]]></content:encoded>
					
					<wfw:commentRss>https://urbangeoanalytics.com/from-neurons-to-llms-transformer-history/feed/</wfw:commentRss>
			<slash:comments>0</slash:comments>
		
		
			</item>
		<item>
		<title>The AI Reading Series · From Perceptrons to Agents · Lecture 1: Where Machine Learning Came From — Reading Cardon&#8217;s Neurons Spike Back</title>
		<link>https://urbangeoanalytics.com/understanding-modern-ai-lecture-1-cardon-neurons-spike-back/</link>
					<comments>https://urbangeoanalytics.com/understanding-modern-ai-lecture-1-cardon-neurons-spike-back/#respond</comments>
		
		<dc:creator><![CDATA[Joan Perez]]></dc:creator>
		<pubDate>Fri, 24 Jul 2026 02:53:32 +0000</pubDate>
				<category><![CDATA[Getting Started]]></category>
		<category><![CDATA[theory]]></category>
		<category><![CDATA[AI]]></category>
		<category><![CDATA[lecture]]></category>
		<guid isPermaLink="false">https://urbangeoanalytics.com/?p=2540</guid>

					<description><![CDATA[<p>The first in a reading series taking you from the artificial neuron of 1943 to today's transformers, mixture-of-experts models and agents. Lecture 1 is a preparatory guide to Cardon, Cointet and Mazières' sociological history of AI, with reading strategy and glossary.</p>
<p>The post <a href="https://urbangeoanalytics.com/understanding-modern-ai-lecture-1-cardon-neurons-spike-back/">The AI Reading Series · From Perceptrons to Agents · Lecture 1: Where Machine Learning Came From — Reading Cardon&#8217;s Neurons Spike Back</a> appeared first on <a href="https://urbangeoanalytics.com">Urban Geo Analytics</a>.</p>
]]></description>
										<content:encoded><![CDATA[<p><div class="fusion-fullwidth fullwidth-box fusion-builder-row-3 fusion-flex-container has-pattern-background has-mask-background nonhundred-percent-fullwidth non-hundred-percent-height-scrolling" style="--awb-border-radius-top-left:0px;--awb-border-radius-top-right:0px;--awb-border-radius-bottom-right:0px;--awb-border-radius-bottom-left:0px;--awb-flex-wrap:wrap;" id="contenu" ><div class="fusion-builder-row fusion-row fusion-flex-align-items-flex-start fusion-flex-content-wrap" style="max-width:1248px;margin-left: calc(-4% / 2 );margin-right: calc(-4% / 2 );"><div class="fusion-layout-column fusion_builder_column fusion-builder-column-3 fusion_builder_column_1_1 1_1 fusion-flex-column" style="--awb-bg-size:cover;--awb-width-large:100%;--awb-margin-top-large:0px;--awb-spacing-right-large:1.92%;--awb-margin-bottom-large:20px;--awb-spacing-left-large:1.92%;--awb-width-medium:100%;--awb-order-medium:0;--awb-spacing-right-medium:1.92%;--awb-spacing-left-medium:1.92%;--awb-width-small:100%;--awb-order-small:0;--awb-spacing-right-small:1.92%;--awb-spacing-left-small:1.92%;"><div class="fusion-column-wrapper fusion-column-has-shadow fusion-flex-justify-content-flex-start fusion-content-layout-column"><div class="fusion-image-element " style="--awb-caption-title-font-family:var(--h2_typography-font-family);--awb-caption-title-font-weight:var(--h2_typography-font-weight);--awb-caption-title-font-style:var(--h2_typography-font-style);--awb-caption-title-size:var(--h2_typography-font-size);--awb-caption-title-transform:var(--h2_typography-text-transform);--awb-caption-title-line-height:var(--h2_typography-line-height);--awb-caption-title-letter-spacing:var(--h2_typography-letter-spacing);"><span class=" fusion-imageframe imageframe-none imageframe-3 hover-type-none"><img decoding="async" width="1802" height="872" title="lecture 1 &#8211; long illustration" src="https://urbangeoanalytics.com/wp-content/uploads/2026/07/lecture-1-long-illustration.png" alt class="img-responsive wp-image-2546" srcset="https://urbangeoanalytics.com/wp-content/uploads/2026/07/lecture-1-long-illustration-200x97.png 200w, https://urbangeoanalytics.com/wp-content/uploads/2026/07/lecture-1-long-illustration-400x194.png 400w, https://urbangeoanalytics.com/wp-content/uploads/2026/07/lecture-1-long-illustration-600x290.png 600w, https://urbangeoanalytics.com/wp-content/uploads/2026/07/lecture-1-long-illustration-800x387.png 800w, https://urbangeoanalytics.com/wp-content/uploads/2026/07/lecture-1-long-illustration-1200x581.png 1200w, https://urbangeoanalytics.com/wp-content/uploads/2026/07/lecture-1-long-illustration.png 1802w" sizes="(max-width: 640px) 100vw, 1200px" /></span></div></div></div><div class="fusion-layout-column fusion_builder_column fusion-builder-column-4 fusion_builder_column_3_4 3_4 fusion-flex-column" style="--awb-bg-size:cover;--awb-width-large:75%;--awb-margin-top-large:0px;--awb-spacing-right-large:2.56%;--awb-margin-bottom-large:20px;--awb-spacing-left-large:2.56%;--awb-width-medium:75%;--awb-order-medium:0;--awb-spacing-right-medium:2.56%;--awb-spacing-left-medium:2.56%;--awb-width-small:100%;--awb-order-small:0;--awb-spacing-right-small:1.92%;--awb-spacing-left-small:1.92%;" id="contenu" data-scroll-devices="small-visibility,medium-visibility,large-visibility"><div class="fusion-column-wrapper fusion-column-has-shadow fusion-flex-justify-content-flex-start fusion-content-layout-column"><div class="fusion-text fusion-text-12 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr">This is the first post in a series of lectures intended to bring a reader with no formal background in machine learning up to the state of the field as it stands today. The trajectory runs from the artificial neuron of 1943 to the systems currently deployed in research and industry: deep convolutional networks, word and image embeddings, transformer architectures, large language models, vision-language models, mixture-of-experts routing, retrieval-augmented generation, and autonomous agents. Each lecture pairs a primary reading with a preparatory guide of this kind, plus a glossary and a set of questions to keep in mind while reading.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">The series begins somewhere unexpected: <a class="keychainify-checked" href="https://hal.science/hal-02190026v1/file/NeuronsSpikeBack.pdf">a forty-page article written by sociologists</a>. This is deliberate. Before diving into LLM and agents, it is worth understanding where all of it came from, and above all understanding that none of these techniques was obvious or inevitable. Modern artificial intelligence is the outcome of a seventy-year scientific battle, with winners, losers, public humiliations, wilderness years, and spectacular comebacks. That is the story told by Dominique Cardon, Jean-Philippe Cointet and Antoine Mazières.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">The article opens on a scene that reads like a western: October 2012, a scientific conference, a largely unknown student walks on stage and announces a result that pulverises ten years of work by an entire research community. The room is stunned. That scene, the earthquake of 2012, is the article&#8217;s destination rather than its starting point. Everything else explains how the field arrived there.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">The thesis in one sentence: the history of AI is a war between two visions of the intelligent machine, a machine that is given rules and a machine that learns from examples, and after fifty years of symbolic domination it is the connectionists, long mocked and marginalised, who won.</p>
</div><div class="fusion-title title fusion-title-11 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">The Two Camps</span></h2></div><div class="fusion-text fusion-text-13 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr">Suppose you want to build a machine that recognises cats in photographs.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">The symbolic approach says: sit down and write the rules. A cat has two pointed ears, whiskers, four legs; IF pointed ears AND whiskers THEN cat. This is intelligence as reasoning, the manipulation of symbols such as &#8220;ear&#8221; and &#8220;cat&#8221; by means of logic. It is intuitive, explainable and elegant, and it is how computers have always been programmed: the human writes the program, the machine executes it.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">The connectionist approach says: rules are hopeless. A cat seen from behind, at night, half hidden behind a curtain, matches no rule at all. Instead, show the machine a hundred thousand photographs labelled &#8220;cat&#8221; or &#8220;not cat&#8221;, and let a network of small interconnected computing units, artificial neurons very loosely inspired by the brain, adjust the strength of its own connections until its answers are good. Nobody writes a rule; the rule emerges from the examples. This is intelligence as learning.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">The strength of the article is that it shows this technical choice to be simultaneously a philosophical choice (is thinking reasoning, or perceiving?), an economic choice (who receives the funding?) and a social choice (which research communities dominate?).</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">The article&#8217;s central image, figure 1, is worth studying closely:</p>
<ul class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-disc flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr">
<li class="font-claude-response-body whitespace-normal break-words pl-2"><strong>Classical machine, hypothetico-deductive:</strong> inputs plus program produce outputs. The human supplies the program.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2"><strong>Inductive machine:</strong> inputs plus outputs produce the program. The human supplies examples, photographs paired with correct answers, and the program is what comes out of the machine.</li>
</ul>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">This reversal is the single most important idea in the entire series. If you retain one thing from Lecture 1, retain that one.</p>
</div><div class="fusion-title title fusion-title-12 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">The Authors’ Analytical Grid: World, Calculator, Horizon</span></h2></div><div class="fusion-text fusion-text-14 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr">The authors analyse each era of AI using three notions, announced early and reused all the way to the final synthesis table (table 1, page 28, an excellent summary of the whole article):</p>
<ul class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-disc flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr">
<li class="font-claude-response-body whitespace-normal break-words pl-2"><strong>The world:</strong> what enters the machine. Data? Rules? Expert knowledge? An environment?</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2"><strong>The calculator:</strong> what does the processing. A logic engine? A neural network? A black box?</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2"><strong>The horizon:</strong> the goal of the computation. Solving a problem? Minimising an error? Imitating examples?</li>
</ul>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">Their key formula is slightly cryptic on first reading but becomes clear by the end. The symbolists wanted to put everything into the calculator, both the world and the goal, whereas the connectionists empty the calculator so that the world gives itself its own horizon. In other words, massive data supplies both the material and the correction: the labelled examples are what tell the network when it is wrong.</p>
</div><div class="fusion-title title fusion-title-13 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">The Four Eras and the Cast</span></h2></div><div class="fusion-text fusion-text-15 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr">The article follows four major periods. The characters introduced here reappear throughout the series.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Cybernetics and the first connectionism, 1943 to 1969.</strong> The age of the pioneers. Warren McCulloch and Walter Pitts invent the artificial neuron in 1943. Norbert Wiener founds cybernetics, the science of machines that self-correct through feedback. Frank Rosenblatt builds the Perceptron in 1957, the first machine that learns to recognise patterns; the press goes wild and conscious machines are promised.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Symbolic AI, 1956 to 1970, then expert systems in the 1980s.</strong> In 1956 John McCarthy and Marvin Minsky coin the term &#8220;artificial intelligence&#8221;, explicitly against cybernetics. With Herbert Simon and Allen Newell they capture the bulk of military funding and impose the symbolic vision. In 1969 Minsky publishes a book that proves neural networks have no future; this is the excommunication. Funding dries up, Rosenblatt dies in 1971, and connectionism enters a long winter. In the 1980s symbolic AI enjoys a second wind with expert systems, thousands of IF-THEN rules extracted from human doctors, geologists and engineers, before a second collapse.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>The return of the neurons, 1986 to 2010.</strong> A small group of holdouts, Geoffrey Hinton, Yann LeCun and Yoshua Bengio, later nicknamed the neural conspiracy, keeps the flame alive. In 1986 backpropagation finally makes it possible to train multi-layer networks. In 1989 LeCun gets a network to read postal codes, the first industrial application. But the years 1995 to 2007 are a colossal winter of rejected papers, mockery and isolation. The article contains excellent first-hand testimony from French researchers about that period.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>The triumph of deep learning, 2010 onward.</strong> Three ingredients converge: massive data, meaning the web and ImageNet with its fourteen million hand-labelled images; GPUs, the graphics processors built for video games and perfectly suited to the massively parallel computations of neural networks; and the algorithms patiently matured during the winter. Then 2012, and the earthquake. Since then, domain after domain, image, speech and text, deep networks have swept everything away.</p>
</div><div class="fusion-title title fusion-title-14 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">How to Read the Article Without Getting Lost</span></h2></div><div class="fusion-text fusion-text-16 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr">It is an academic sociology article: dense, but well written and full of anecdotes. Some practical advice.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">Take your time with the opening, the 2012 narrative on pages 2 and 3. The entire article is contained in that scene.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">Lean on the figures. Figure 1, the two machines, and table 1, the four ages, are your two anchors. Figure 3, the timeline, shows the shifting dominations visually. Figures 4 and 5 give a first look at an artificial neuron and at backpropagation.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">Do not get stuck on the details of the Web of Science queries in notes 5 and 6, or on the philosophical references to Fodor, Smolensky and the computational theory of mind. Grasp the general idea and move on. The section on convexity is the most technical; the glossary below gives the minimum needed to get through it.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">Savour the interview quotations, set in indented blocks in spoken register. That is where the history comes alive.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">Keep three questions in mind while reading:</p>
<ol class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-decimal flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr">
<li class="font-claude-response-body whitespace-normal break-words pl-2">For each era, what is in the world, the calculator and the horizon?</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2">Why did the connectionists win when they did, rather than in 1960? The hint is that it was not the ideas that changed.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2">The title speaks of striking back. Who was humiliated, when, and by whom?</li>
</ol>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr">Budget two to two and a half hours of attentive reading. It is the longest reading of the series and probably the one that will stay with you.</p>
</div><div class="fusion-title title fusion-title-15 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">Survival Glossary</span></h2></div><div class="fusion-text fusion-text-17 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr">Terms are listed roughly in their order of appearance in the article.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Computer vision.</strong> Research field aiming to make machines see: recognising objects, faces and scenes in images.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>ImageNet.</strong> A database of fourteen million images across roughly twenty-one thousand categories, hand-labelled by thousands of micro-workers. It serves as an annual competition, and it is the benchmark on which the 2012 earthquake took place.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Benchmark.</strong> A standardised test set allowing objective comparison between the performance of different methods.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>GPU, graphics processing unit.</strong> A processor originally designed for video games, capable of performing millions of simple operations in parallel, which is exactly what a neural network requires.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Deep learning.</strong> Neural networks with many layers, hence deep. The term was coined by Hinton in 2006, partly to escape the poor reputation of the word connectionism.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Machine learning.</strong> The family of methods in which a machine learns from examples instead of being explicitly programmed. Deep learning is one branch of it.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Parameters, also weights or coefficients.</strong> The numbers adjusted during learning, encoding the strength of the connections between neurons. A hundred million parameters means a hundred million small knobs tuned automatically.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Symbolic AI, also GOFAI, Good Old-Fashioned AI.</strong> The rules-and-logic approach. Thinking equals manipulating symbols.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Connectionism.</strong> The neural network approach. Thinking equals parallel, distributed computation by simple units, with intelligence emerging from the connections.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Cybernetics.</strong> The science founded by Norbert Wiener in 1948, studying systems, whether machines or organisms, that regulate themselves through feedback. The direct ancestor of connectionism.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Feedback.</strong> Reinjecting the measured output error as a new input so that the system corrects itself. The thermostat is the canonical example. It is the founding principle of cybernetics and, in a sense, of all machine learning.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Black box.</strong> A system whose inputs and outputs are observable but whose internal workings are not understood. A recurring criticism of neural networks, and a property their defenders have claimed proudly since the cybernetic era.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Formal neuron.</strong> An ultra-simplified mathematical model of a neuron, due to McCulloch and Pitts in 1943. It sums its inputs weighted by weights and activates if the sum exceeds a threshold. Figure 4 shows that this amounts to three operations, no more.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Perceptron.</strong> The first learning machine built on a neural network, developed by Rosenblatt between 1957 and 1961, funded by the US Navy and designed for image recognition. It is the emblem of early connectionism and the target of Minsky&#8217;s attack.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Layers and hidden layers.</strong> Neurons are organised in tiers: an input layer holding the data, intermediate layers called hidden where the work happens, and an output layer holding the answer. Minsky&#8217;s 1969 book attacked a single-layer perceptron.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>XOR, exclusive OR.</strong> The elementary logical function &#8220;A or B, but not both&#8221;, which a single-layer perceptron cannot learn. This was Minsky and Papert&#8217;s decisive argument for burying connectionism. Multiple layers solve the problem.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>AI winter.</strong> A period of collapse in funding and credibility following excessive promises. There have been two general ones, in the early 1970s and the late 1980s, plus the specifically connectionist winters of 1969 to 1986 and 1995 to 2007.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Expert system.</strong> A 1980s program encoding human expert knowledge as thousands of IF-THEN rules, MYCIN for medical diagnosis being the standard example, driven by an inference engine that decides which rule to apply when. Expert systems mark both the peak and the collapse of symbolic AI.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Knowledge base.</strong> The stock of rules and facts held by an expert system, to be contrasted with a dataset: a knowledge base contains intelligible rules, a dataset contains raw examples.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Backpropagation, or backprop.</strong> The algorithm popularised in 1986 by Rumelhart, Hinton and Williams that makes learning possible. The error is measured at the output, then propagated backwards layer by layer to adjust each weight in the right direction. See figure 5. This is the central mechanism of all deep learning.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Loss function.</strong> The number measuring how wrong the network is. All of learning consists in minimising it.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Gradient descent.</strong> The minimisation method: compute the slope of the loss function and take a small step downhill, then repeat millions of times. The usual image is walking down a mountain in fog by following the slope underfoot. Stochastic means the slope is estimated on a small sample of the data at a time, which is faster.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Convolution and convolutional networks, CNNs.</strong> A technique invented by LeCun in 1989 for images. Rather than connecting every pixel to every neuron, small filters are slid across the image to detect local patterns such as edges and corners independently of their position. This is the architecture behind AlexNet in 2012.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Feature engineering.</strong> The now largely extinct art of hand-programming the relevant characteristics of the data, detecting edges, corners and contrasts, before passing them to an algorithm. Deep learning made it obsolete because the network discovers its own features. An entire scientific community lost its object of research this way, which accounts for the bitterness audible in some of the testimony in the article.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>End-to-end.</strong> Processing raw data, the pixels, all the way to the final answer, &#8220;cat&#8221;, within a single network, with no intermediate step programmed by a human.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>SVM, support vector machines, and kernel methods.</strong> A rival learning method dating from 1992, mathematically elegant and dominant between roughly 1995 and 2010. It was the great internal adversary within the learning camp, and the duel between SVMs and neural networks structures an entire section of the article.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Convexity.</strong> A mathematical criticism levelled at neural networks. A convex function is shaped like a bowl, with a single hollow, so gradient descent is guaranteed to reach the bottom, the global minimum. A neural network&#8217;s loss function is instead a landscape of mountains with countless valleys, local minima, with no guarantee of finding the best one. The SVM mathematicians treated this as a fatal flaw; LeCun&#8217;s answer amounted to saying that the theoretical guarantee matters less than the fact that it works better in practice, and experience proved him right.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Overfitting.</strong> When a network learns its training examples by heart instead of extracting general regularities, performing excellently on known data and poorly on new data. Dropout, the random switching-off of neurons during training, is one countermeasure mentioned in the article.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Hyper-parameters.</strong> All the architectural choices fixed by the human before learning begins: the number of layers, the number of neurons, the learning rate. These stand in contrast to parameters, which are learned automatically. The article shows that human labour does not disappear, it shifts from writing rules to setting these values.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Dataset.</strong> A collection of examples used to train and test a model, often as input-output pairs such as a photograph and its label.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Labelled data.</strong> Examples accompanied by the correct answer, supplied by humans. This is the fuel of supervised learning.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Crowdsourcing and Mechanical Turk.</strong> Micro-work platforms where thousands of people, paid by the task, label data by drawing a box around the dog or typing the spoken word. This is the human face, invisible and poorly paid, of so-called raw data, and ImageNet is its product.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Embedding, or vector.</strong> The transformation of an object, a word, an image, a social network, into a list of numbers so that a network can compute with it. The article cites word2vec and LeCun&#8217;s formula for putting the world into a vector, world2vec. This is the direct bridge to Lecture 2.</p>
<p class="font-claude-response-body break-words whitespace-normal" dir="ltr"><strong>Induction and deduction.</strong> Deduction starts from general rules and applies them to particular cases, which is the symbolic approach. Induction starts from particular cases, the examples, and extracts a general rule, which is the connectionist approach. The article&#8217;s full subtitle, &#8220;the invention of inductive machines&#8221;, says the essential.</p>
</div><div class="fusion-title title fusion-title-16 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">After the Reading: Videos</span></h2></div><div class="fusion-text fusion-text-18 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr">Watch these after finishing the article to go further.</p>
<ul class="&#091;li_&amp;&#093;:mb-0 &#091;li_&amp;&#093;:mt-1 &#091;li_&amp;&#093;:gap-1 &#091;&amp;:not(:last-child)_ul&#093;:pb-1 &#091;&amp;:not(:last-child)_ol&#093;:pb-1 list-disc flex flex-col gap-1 pl-8 mb-3 print:block print:space-y-1" dir="ltr">
<li class="font-claude-response-body whitespace-normal break-words pl-2"><a class="keychainify-checked" href="https://www.youtube.com/watch?v=UZDiGooFs54"><em>The moment we stopped understanding AI [AlexNet]</em></a>, Welch Labs, about eighteen minutes, in English. The 2012 earthquake seen from inside the network, with excellent visualisations of AlexNet&#8217;s layers, and the perfect complement to the article&#8217;s opening scene.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2"><em>Heroes of Deep Learning: Andrew Ng interviews Geoffrey Hinton</em>, followed by the LeCun interview in the same series, in English. Both are cited in the article&#8217;s own footnotes, notes 3 and 23. Hearing Hinton and LeCun recount the connectionist winter in their own voices is worth the time.</li>
<li class="font-claude-response-body whitespace-normal break-words pl-2"><a class="keychainify-checked" href="https://www.youtube.com/watch?v=trWrEWfhTVg"><em>Le deep learning</em></a>, ScienceEtonnante (David Louapre), in French. Neural networks, backpropagation and convolution explained in twenty minutes.</li>
</ul>
</div><div class="fusion-title title fusion-title-17 fusion-sep-none fusion-title-text fusion-title-size-two" style="--awb-margin-top:25px;--awb-margin-bottom:0px;--awb-font-size:35px;"><h2 class="fusion-title-heading title-heading-left fusion-responsive-typography-calculated" style="margin:0;font-size:1em;--fontSize:35;line-height:var(--awb-typography1-line-height);"><span style="font-weight: 400;">Reference</span></h2></div><div class="fusion-text fusion-text-19 fusion-text-no-margin" style="--awb-content-alignment:justify;--awb-margin-top:25px;--awb-margin-bottom:25px;"><p class="font-claude-response-body break-words whitespace-normal" dir="ltr">Cardon, D., Cointet, J.-P. and Mazières, A. (2018). <a class="keychainify-checked" href="https://hal.science/hal-02190026v1/file/NeuronsSpikeBack.pdf"><em>Neurons spike back. The invention of inductive machines and the artificial intelligence controversy.</em></a> Réseaux, 211(5), 173–220.</p>
</div></div></div><div class="fusion-layout-column fusion_builder_column fusion-builder-column-5 awb-sticky awb-sticky-medium awb-sticky-large fusion_builder_column_1_4 1_4 fusion-flex-column" style="--awb-padding-top:20px;--awb-padding-right:20px;--awb-padding-bottom:20px;--awb-padding-left:20px;--awb-bg-size:cover;--awb-border-color:var(--awb-color6);--awb-border-style:solid;--awb-width-large:25%;--awb-margin-top-large:0px;--awb-spacing-right-large:7.68%;--awb-margin-bottom-large:20px;--awb-spacing-left-large:7.68%;--awb-width-medium:25%;--awb-order-medium:0;--awb-spacing-right-medium:7.68%;--awb-spacing-left-medium:7.68%;--awb-width-small:100%;--awb-order-small:0;--awb-spacing-right-small:1.92%;--awb-spacing-left-small:1.92%;--awb-sticky-offset:150px;" data-scroll-devices="small-visibility,medium-visibility,large-visibility"><div class="fusion-column-wrapper fusion-column-has-shadow fusion-flex-justify-content-flex-start fusion-content-layout-column"><div class="fusion-text fusion-text-20"><p><span style="color: #143c4e;"><strong>Table of contents</strong></span></p>
</div><div class="awb-toc-el awb-toc-el--2" data-awb-toc-id="2" data-awb-toc-options="{&quot;allowed_heading_tags&quot;:{&quot;h2&quot;:0},&quot;ignore_headings&quot;:&quot;&quot;,&quot;ignore_headings_words&quot;:&quot;&quot;,&quot;enable_cache&quot;:&quot;no&quot;,&quot;highlight_current_heading&quot;:&quot;yes&quot;,&quot;hide_hidden_titles&quot;:&quot;no&quot;,&quot;limit_container&quot;:&quot;page_content&quot;,&quot;select_custom_headings&quot;:&quot;.contenu H2, .contenu H3&quot;,&quot;icon&quot;:&quot;fa-flag fas&quot;,&quot;counter_type&quot;:&quot;none&quot;}" style="--awb-item-padding-right:5px;--awb-item-padding-left:5px;"><div class="awb-toc-el__content"></div></div><div class="fusion-separator fusion-full-width-sep" style="align-self: center;margin-left: auto;margin-right: auto;margin-top:20px;margin-bottom:20px;width:100%;"><div class="fusion-separator-border sep-single sep-solid" style="--awb-height:20px;--awb-amount:20px;--awb-sep-color:var(--awb-color6);border-color:var(--awb-color6);border-top-width:1px;"></div></div><div class="fusion-image-element " style="--awb-margin-top:25px;--awb-margin-bottom:25px;--awb-caption-title-font-family:var(--h2_typography-font-family);--awb-caption-title-font-weight:var(--h2_typography-font-weight);--awb-caption-title-font-style:var(--h2_typography-font-style);--awb-caption-title-size:var(--h2_typography-font-size);--awb-caption-title-transform:var(--h2_typography-text-transform);--awb-caption-title-line-height:var(--h2_typography-line-height);--awb-caption-title-letter-spacing:var(--h2_typography-letter-spacing);--awb-filter:saturate(100%);--awb-filter-transition:filter 0.3s ease;--awb-filter-hover:saturate(0%);"><span class=" fusion-imageframe imageframe-none imageframe-4 hover-type-zoomout"><img decoding="async" width="1536" height="1024" title="blog lvl1" src="https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1.png" alt class="img-responsive wp-image-1685" srcset="https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1-200x133.png 200w, https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1-400x267.png 400w, https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1-600x400.png 600w, https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1-800x533.png 800w, https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1-1200x800.png 1200w, https://urbangeoanalytics.com/wp-content/uploads/2025/11/blog-lvl1.png 1536w" sizes="(max-width: 640px) 100vw, 400px" /></span></div></div></div></div></div><div class="fusion-fullwidth fullwidth-box fusion-builder-row-4 fusion-flex-container has-pattern-background has-mask-background nonhundred-percent-fullwidth non-hundred-percent-height-scrolling" style="--awb-border-radius-top-left:0px;--awb-border-radius-top-right:0px;--awb-border-radius-bottom-right:0px;--awb-border-radius-bottom-left:0px;--awb-flex-wrap:wrap;" ><div class="fusion-builder-row fusion-row fusion-flex-align-items-flex-start fusion-flex-content-wrap" style="max-width:1248px;margin-left: calc(-4% / 2 );margin-right: calc(-4% / 2 );"></div></div></p>
<p>The post <a href="https://urbangeoanalytics.com/understanding-modern-ai-lecture-1-cardon-neurons-spike-back/">The AI Reading Series · From Perceptrons to Agents · Lecture 1: Where Machine Learning Came From — Reading Cardon&#8217;s Neurons Spike Back</a> appeared first on <a href="https://urbangeoanalytics.com">Urban Geo Analytics</a>.</p>
]]></content:encoded>
					
					<wfw:commentRss>https://urbangeoanalytics.com/understanding-modern-ai-lecture-1-cardon-neurons-spike-back/feed/</wfw:commentRss>
			<slash:comments>0</slash:comments>
		
		
			</item>
	</channel>
</rss>
