diff --git a/lectures/_static/quant-econ.bib b/lectures/_static/quant-econ.bib
index 8174125a7..2b65618fb 100644
--- a/lectures/_static/quant-econ.bib
+++ b/lectures/_static/quant-econ.bib
@@ -2182,6 +2182,26 @@ @article{PhelanStacchetti2001
month = {November}
}
+@article{BarroGordon1983,
+ author = {Barro, Robert J. and Gordon, David B.},
+ title = {Rules, Discretion and Reputation in a Model of Monetary Policy},
+ journal = {Journal of Monetary Economics},
+ volume = {12},
+ number = {1},
+ pages = {101--121},
+ year = {1983}
+}
+
+@article{SargentVelde1995,
+ author = {Sargent, Thomas J. and Velde, Fran\c{c}ois R.},
+ title = {Macroeconomic Features of the French Revolution},
+ journal = {Journal of Political Economy},
+ volume = {103},
+ number = {3},
+ pages = {474--518},
+ year = {1995}
+}
+
@article{APS1990,
title = {Toward a Theory of Discounted Repeated Games with Imperfect Monitoring},
author = {Abreu, Dilip and David Pearce and Ennio Stacchetti},
@@ -4781,3 +4801,202 @@ @inproceedings{MurrayAdamsMacKay2010
pages = {541--548},
year = {2010}
}
+
+@article{KiyotakiWright1989,
+ author = {Kiyotaki, Nobuhiro and Wright, Randall},
+ title = {On Money as a Medium of Exchange},
+ journal = {Journal of Political Economy},
+ volume = {97},
+ number = {4},
+ pages = {927--954},
+ year = {1989},
+ doi = {10.1086/261634}
+}
+
+@article{MarimonMcGrattanSargent1990,
+ author = {Marimon, Ramon and McGrattan, Ellen and Sargent, Thomas J.},
+ title = {Money as a Medium of Exchange in an Economy with Artificially
+ Intelligent Agents},
+ journal = {Journal of Economic Dynamics and Control},
+ volume = {14},
+ number = {2},
+ pages = {329--373},
+ year = {1990},
+ doi = {10.1016/0165-1889(90)90025-C}
+}
+
+@book{Holland1975,
+ author = {Holland, John H.},
+ title = {Adaptation in Natural and Artificial Systems},
+ publisher = {University of Michigan Press},
+ address = {Ann Arbor},
+ year = {1975}
+}
+
+@book{HollandHolyoakNisbettThagard1986,
+ author = {Holland, John H. and Holyoak, Keith J. and Nisbett, Richard E.
+ and Thagard, Paul R.},
+ title = {Induction: Processes of Inference, Learning, and Discovery},
+ publisher = {MIT Press},
+ address = {Cambridge, MA},
+ year = {1986}
+}
+
+@book{Goldberg1989,
+ author = {Goldberg, David E.},
+ title = {Genetic Algorithms in Search, Optimization, and Machine Learning},
+ publisher = {Addison-Wesley},
+ address = {Reading, MA},
+ year = {1989}
+}
+
+@book{Sargent1993,
+ author = {Sargent, Thomas J.},
+ title = {Bounded Rationality in Macroeconomics},
+ publisher = {Oxford University Press},
+ address = {Oxford},
+ series = {The Arne Ryde Memorial Lectures},
+ year = {1993}
+}
+
+@article{KarekenWallace1981,
+ author = {Kareken, John and Wallace, Neil},
+ title = {On the Indeterminacy of Equilibrium Exchange Rates},
+ journal = {Quarterly Journal of Economics},
+ volume = {96},
+ number = {2},
+ pages = {207--222},
+ year = {1981},
+ doi = {10.2307/1882388}
+}
+
+@article{Samuelson1958,
+ author = {Samuelson, Paul A.},
+ title = {An Exact Consumption-Loan Model of Interest with or without the
+ Social Contrivance of Money},
+ journal = {Journal of Political Economy},
+ volume = {66},
+ number = {6},
+ pages = {467--482},
+ year = {1958},
+ doi = {10.1086/258100}
+}
+
+@article{LucasPrescott1971,
+ author = {Lucas, Robert E. and Prescott, Edward C.},
+ title = {Investment Under Uncertainty},
+ journal = {Econometrica},
+ volume = {39},
+ number = {5},
+ pages = {659--681},
+ year = {1971},
+ doi = {10.2307/1909571}
+}
+
+@incollection{MarcetSargent1989hyper,
+ author = {Marcet, Albert and Sargent, Thomas J.},
+ title = {Least Squares Learning and the Dynamics of Hyperinflation},
+ editor = {Barnett, William A. and Geweke, John and Shell, Karl},
+ booktitle = {Economic Complexity: Chaos, Sunspots, Bubbles, and Nonlinearity},
+ publisher = {Cambridge University Press},
+ address = {Cambridge},
+ pages = {119--137},
+ year = {1989}
+}
+
+@article{MarimonSunder1993,
+ author = {Marimon, Ramon and Sunder, Shyam},
+ title = {Indeterminacy of Equilibria in a Hyperinflationary World:
+ Experimental Evidence},
+ journal = {Econometrica},
+ volume = {61},
+ number = {5},
+ pages = {1073--1107},
+ year = {1993},
+ doi = {10.2307/2951494}
+}
+
+@article{BrunoFischer1990,
+ author = {Bruno, Michael and Fischer, Stanley},
+ title = {Seigniorage, Operating Rules, and the High Inflation Trap},
+ journal = {Quarterly Journal of Economics},
+ volume = {105},
+ number = {2},
+ pages = {353--374},
+ year = {1990},
+ doi = {10.2307/2937791}
+}
+
+@article{Brock1974,
+ author = {Brock, William A.},
+ title = {Money and Growth: The Case of Long Run Perfect Foresight},
+ journal = {International Economic Review},
+ volume = {15},
+ number = {3},
+ pages = {750--777},
+ year = {1974},
+ doi = {10.2307/2525739}
+}
+
+@article{Imrohoroglu1993,
+ author = {\.{I}mrohoro\u{g}lu, Selahattin},
+ title = {Testing for Sunspot Equilibria in the {German} Hyperinflation},
+ journal = {Journal of Economic Dynamics and Control},
+ volume = {17},
+ number = {1-2},
+ pages = {289--317},
+ year = {1993},
+ doi = {10.1016/S0165-1889(06)80013-0}
+}
+
+@article{ChenWhite1998,
+ author = {Chen, Xiaohong and White, Halbert},
+ title = {Nonparametric Adaptive Learning with Feedback},
+ journal = {Journal of Economic Theory},
+ volume = {82},
+ number = {1},
+ pages = {190--222},
+ year = {1998},
+ doi = {10.1006/jeth.1998.2426}
+}
+
+@incollection{MarcetMarshall1992,
+ author = {Marcet, Albert and Marshall, David A.},
+ title = {Convergence of Approximate Model Solutions to Rational
+ Expectations Equilibria Using the Method of Parameterized
+ Expectations},
+ booktitle = {Universitat Pompeu Fabra Economics Working Paper},
+ number = {13},
+ year = {1992}
+}
+
+@article{Arifovic1996,
+ author = {Arifovic, Jasmina},
+ title = {The Behavior of the Exchange Rate in the Genetic Algorithm and
+ Experimental Economies},
+ journal = {Journal of Political Economy},
+ volume = {104},
+ number = {3},
+ pages = {510--541},
+ year = {1996},
+ doi = {10.1086/262032}
+}
+
+@book{MinskyPapert1969,
+ author = {Minsky, Marvin and Papert, Seymour},
+ title = {Perceptrons: An Introduction to Computational Geometry},
+ publisher = {MIT Press},
+ address = {Cambridge, MA},
+ year = {1969}
+}
+
+@incollection{Axelrod1987,
+ author = {Axelrod, Robert},
+ title = {The Evolution of Strategies in the Iterated Prisoner's Dilemma},
+ editor = {Davis, Lawrence},
+ booktitle = {Genetic Algorithms and Simulated Annealing},
+ publisher = {Morgan Kaufmann},
+ address = {Los Altos, CA},
+ pages = {32--41},
+ year = {1987}
+}
diff --git a/lectures/_toc.yml b/lectures/_toc.yml
index e16fd96a3..91397a3c1 100644
--- a/lectures/_toc.yml
+++ b/lectures/_toc.yml
@@ -117,11 +117,22 @@ parts:
- file: lq_bewley_complete_markets
- file: lq_robust_smoothing
- file: lq_inventories
+- caption: Bounded Rationality in Macroeconomics
+ numbered: true
+ chapters:
+ - file: bounded_rationality
+ - file: olg_adaptive_money
+ - file: learning_approximation
+ - file: exchange_rate_learning
+ - file: genetic_classifier
+ - file: marimon_mcgrattan_sargent
+ - file: prospects_bounded_rationality
- caption: Phillips Curve Tradeoffs
numbered: true
chapters:
- file: phillips_two_stories
- file: phillips_credibility
+ - file: phillips_credible_policies
- file: phillips_adaptive
- file: phillips_misspecified
- file: phillips_self_confirming
diff --git a/lectures/bounded_rationality.md b/lectures/bounded_rationality.md
new file mode 100644
index 000000000..df1efc412
--- /dev/null
+++ b/lectures/bounded_rationality.md
@@ -0,0 +1,1061 @@
+---
+jupytext:
+ text_representation:
+ extension: .md
+ format_name: myst
+ format_version: 0.13
+ jupytext_version: 1.17.1
+kernelspec:
+ display_name: Python 3 (ipykernel)
+ language: python
+ name: python3
+---
+
+(bounded_rationality)=
+```{raw} jupyter
+
+```
+
+# A Peculiar Definition of Bounded Rationality
+
+```{index} single: Bounded Rationality
+```
+
+```{contents} Contents
+:depth: 2
+```
+
+## Overview
+
+This lecture opens a series based on {cite:t}`Sargent1993` about Bounded Rationality in Macroeconomics, from one point of view in the late 1980s and early 1990s.
+
+Those were years when the Soviet Union ended and formerly Warsaw Pact countries were rearranging their political economies.
+
+For those countries, those were thus truly *regime changes* in the technical sense of modern dynamic macroeconomics.
+
+Two generations of work on economic dynamics — in game theory, macroeconomics, and general
+equilibrium theory — had produced theories that embraced the rational expectations assumption, models that were designed to understand settings in which people face recurrent
+situations they have lived through many times before.
+
+The transitions across regimes underway in Eastern Europe in the early 1990s were not like that.
+
+People there were confronted with unprecedented opportunities, new and ill-defined rules, and
+a daily struggle to figure out the mechanism that would eventually govern trade and
+production.
+
+Economists with good models of a market economy had ample *equilibrium theories*
+describing how a system behaves once it has fully adjusted to a new and coherent set of rules
+and expectations.
+
+They knew much less about the dynamics of *transitions* from a Soviet system to a market economy.
+
+They might hold prejudices and anecdotes about how to manage such a transition, but no
+empirically confirmed formal theory of it.
+
+Against this background, some economists ventured into what {cite:t}`Sims1980`
+called the "wilderness" of irrational expectations and bounded rationality.
+
+The aim was partly to build theories of transition dynamics, partly to understand the
+properties of equilibrium dynamics themselves, and partly to study systems that never settle
+down.
+
+This series follows {cite:t}`Sargent1993` a little way into that wilderness.
+
+### The knowledge that rational expectations imputes
+
+To see what bounded rationality retreats *from*, start with rational expectations, which
+imposes **two** requirements:
+
+1. **Individual rationality** — each artificial agent's behavior maximizes an objective function
+ subject to perceived constraints.
+1. **Mutual consistency** — the constraints perceived by everybody in the system agree
+ with one another.
+
+The second requirement is Muth's *rational expectations* assumption.
+
+In an economy one person's decisions are part of another person's constraints, so consistency
+requires each person to hold correct beliefs about everyone else's decisions, decision
+processes, and beliefs.
+
+Consistency is also what gives rational expectations its power: without some restriction on
+perceptions, a model in which behavior depends on arbitrary assumptions about subjective beliefs can produce almost any outcome at
+all.
+
+But look at what that requirement imputes to people once a model is taken to data.
+
+The agents inside a rational expectations model evaluate their Euler equations using
+*equilibrium* probability distributions.
+
+Those are the very distributions that the econometrician studying them is still struggling to
+estimate.
+
+The agents, in other words, have somehow already solved the inference problem that the
+economist is only part way through.
+
+### Sargent's formulation of bounded rationality
+
+Sargent's **bounded rationality** program keeps individual rationality and retreats from mutual
+consistency in a particular way that was motivated by his love of time series econometrics.
+
+Sargent's version of a bounded rationality program is:
+
+> I interpret a proposal to build models with 'boundedly rational' agents as a call to
+> retreat from the second piece of rational expectations (mutual consistency of perceptions)
+> by expelling rational agents from our model environments and replacing them with
+> 'artificially intelligent' agents who behave like econometricians. These 'econometricians'
+> theorize, estimate, and adapt in attempting to learn about probability distributions
+> which, under rational expectations, they already know.
+
+The agents are made more like the people who build the models: they gather data, form
+theories, estimate, and adapt.
+
+```{note}
+After he saw Sargent's manuscript, Carnegie-Mellon's Herbert Simon wrote Sargent a letter
+saying that he objected to Sargent's formulation and recommended that Sargent not call what he
+was doing ''bounded rationality''.
+
+Simon particularly disliked Sargent's making the agents inside his model act like
+econometricians, a pretense that Simon found preposterous.
+```
+
+Sargent's proposal makes work harder for the model builder, not easier.
+
+Withdrawing the assumption of a commonly understood environment means we must put something in
+its place, and there are many plausible somethings to choose among:
+
+> This area is wilderness because the researcher faces so many choices after he decides to
+> forgo the discipline provided by equilibrium theorizing. The commitment to equilibrium
+> theorizing made many choices for him by requiring that people be modelled as optimal
+> decision-makers within a commonly understood environment. When we withdraw the assumption
+> of a commonly understood environment, we have to replace it with something, and there are
+> so many plausible possibilities.
+
+### What the program is good for
+
+The payoffs the book pursues are of three kinds.
+
+Sometimes a collection of adaptive agents learns to behave *as if* it had rational
+expectations, which lends the equilibrium a plausibility it lacked as a mere assumption.
+
+Sometimes adaptive agents converge to a *particular* equilibrium among many, turning learning
+into a device for **selecting** among rational expectations equilibria, and, relatedly, into
+a way of **computing** equilibria too complicated to solve by hand.
+
+And, more ambitiously, adaptive dynamics hold out the hope of a theory of the *transition*
+itself, the out-of-equilibrium adjustment that the Eastern European reformers had to manage
+blind.
+
+That last promise is the least fulfilled, as the series will acknowledge; the selection and
+computation payoffs are the surer ones.
+
+This lecture takes up selection, through the sharpest case: models with **too many
+equilibria**.
+
+When a rational expectations model has a continuum of equilibria, the physical description of
+the economy plus the equilibrium concept fail to pin down what happens.
+
+Something else must choose, and a plausible account of how people grope toward equilibrium is
+a natural candidate for that something.
+
+We build the two monetary examples that the rest of the series returns to repeatedly, both
+with a continuum of rational expectations equilibria:
+
+* a quantity-theory model of money and prices in which the price level is determined only up
+ to an arbitrary **bubble** term, and
+* a two-currency version in which the *exchange rate is completely unrestricted*.
+
+Along the way we set up the machinery — the fixed-point view of equilibrium, the relaxation
+algorithm, adaptive expectations, and Muth's inverse optimal-prediction problem — that later
+lectures use to expel the rational agents and put adaptive ones in their place.
+
+### The rest of the series
+
+* {doc}`olg_adaptive_money` — adaptive households in Samuelson's overlapping generations
+ monetary model.
+ - Least squares learning selects the low-inflation equilibrium that the rational expectations
+ dynamics reject, and laboratory subjects go the same way.
+ - A government learning a Phillips curve closes the lecture.
+* {doc}`learning_approximation` — what an agent must give up when the state is continuous and a
+ separate response for each contingency is out of the question, which brings in approximate
+ equilibria and the resemblance between learning algorithms and equilibrium computation
+ algorithms.
+* {doc}`exchange_rate_learning` — the two-currency model of this lecture with adaptive agents,
+ where learning pins the exchange rate down only by making it depend on history, while a
+ genetic-algorithm economy instead produces volatility that never dies.
+* {doc}`genetic_classifier` — a catalogue of candidate brains, from Holland and the
+ connectionists: perceptrons, associative memories, genetic algorithms, classifier systems.
+* {doc}`marimon_mcgrattan_sargent` — populations of classifier systems that discover, from
+ scratch, which good will serve as money.
+* {doc}`prospects_bounded_rationality` — the 1993 ledger of what the program achieved and what
+ it did not, and a postscript on the three decades since.
+
+Let's start with some imports.
+
+```{code-cell} ipython3
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+```
+
+## Rational expectations as a fixed point
+
+### A static market
+
+Take a competitive industry with a large number $n$ of identical firms.
+
+Each firm chooses output $x$ to maximize
+
+$$
+R(x, p) = p x - c(x),
+$$
+
+where $c$ is an increasing, convex cost function, and price is determined by a
+downward-sloping inverse demand curve
+
+$$
+p = p(nX)
+$$
+
+in which $X$ is the output of the *average* firm.
+
+Each firm is a price-taker and an $X$-taker: it takes $p$ as given and sets marginal cost
+equal to price, $p = c'(x)$.
+
+Write the solution as $x = g(p)$.
+
+Substituting the demand curve gives the **best-response map**
+
+```{math}
+:label: br_map
+
+x = g(p(nX)) \equiv h(X),
+```
+
+which sends a conjectured industry average $X$ to the individual output an optimizing firm
+would choose against it.
+
+That is the first component of rational expectations, and it is all that individual
+rationality delivers.
+
+The second component — consistency — requires that what each firm chooses coincides with
+what it assumed the average firm was choosing:
+
+```{math}
+:label: static_ree
+
+X = h(X).
+```
+
+A static rational expectations equilibrium is a **fixed point of the best-response map**.
+
+For a concrete example, take a quadratic cost function $c(x) = \tfrac{\gamma}{2} x^2$ and a
+linear inverse demand curve $p = a - b n X$.
+
+Then $p = c'(x) = \gamma x$ gives $x = p / \gamma$, so
+
+$$
+h(X) = \frac{a - b n X}{\gamma},
+\qquad\text{with fixed point}\qquad
+X^* = \frac{a}{\gamma + b n}.
+$$
+
+```{code-cell} ipython3
+a, b, n, γ = 10.0, 0.3, 5, 1.0
+
+def h(X):
+ "Best response of an individual firm to an industry average X."
+ return (a - b * n * X) / γ
+
+X_star = a / (γ + b * n)
+print(f"X* = {X_star}")
+print(f"h(X*) = {h(X_star)}")
+print(f"h'(X) = {-b * n / γ}")
+```
+
+### Computing the fixed point
+
+Now suppose we want to *find* $X^*$ without solving for it in closed form.
+
+A natural starting point is the **relaxation algorithm**: carry an estimate $X^*_k$ of the
+equilibrium, compute the best response to it, and move part of the way toward that best
+response,
+
+```{math}
+:label: relaxation
+
+X^*_k = X^*_{k-1} + \lambda\bigl(h(X^*_{k-1}) - X^*_{k-1}\bigr),
+```
+
+where $\lambda \in (0, 1]$ is a **relaxation parameter**.
+
+With $\lambda = 1$ this is simple iteration on $h$, the classic cobweb.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Relaxation algorithm paths, undamped and damped"
+ name: fig-br-relaxation
+---
+def relax(λ, X0=1.0, n_iter=12):
+ "Iterate the relaxation algorithm, returning the whole path."
+ path = np.empty(n_iter + 1)
+ path[0] = X0
+ for k in range(n_iter):
+ path[k + 1] = path[k] + λ * (h(path[k]) - path[k])
+ return path
+
+
+fig, axes = plt.subplots(1, 2, figsize=(11, 4))
+
+axes[0].plot(relax(1.0), 'o-', ms=4, lw=1, color='C0',
+ label=r"$\lambda = 1.0$ (cobweb)")
+axes[0].set_title("undamped")
+axes[0].set_ylabel("$X^*_k$")
+
+for i, λ in enumerate((0.8, 0.5, 0.3)):
+ axes[1].plot(relax(λ), 'o-', ms=4, lw=1, color=f'C{i+1}',
+ label=fr"$\lambda = {λ}$")
+axes[1].set_title("damped")
+axes[1].set_ylim(-1, 9)
+
+for ax in axes:
+ ax.axhline(X_star, color='k', lw=0.8, ls='--', label="$X^*$")
+ ax.set_xlabel("$k$")
+ ax.legend(frameon=False, fontsize=9)
+plt.tight_layout()
+plt.show()
+```
+
+The naive cobweb $\lambda = 1$ **diverges**, oscillating ever further away from the
+equilibrium it is trying to find; note the scale on the left panel.
+
+Since $h$ is affine, iteration {eq}`relaxation` has multiplier $1 + \lambda(h' - 1)$, so it
+converges if and only if
+
+$$
+\bigl|1 + \lambda(h' - 1)\bigr| < 1
+\qquad\Longleftrightarrow\qquad
+\lambda < \frac{2}{1 + bn/\gamma} .
+$$
+
+```{code-cell} ipython3
+λ_max = 2 / (1 + b * n / γ)
+print(f"converges iff λ < {λ_max}")
+print(f"λ = 1.0 : multiplier = {1 + 1.0 * (-b*n/γ - 1):+.2f} (diverges)")
+print(f"λ = 0.8 : multiplier = {1 + 0.8 * (-b*n/γ - 1):+.2f} (period-2 cycle)")
+print(f"λ = 0.5 : multiplier = {1 + 0.5 * (-b*n/γ - 1):+.2f} (converges)")
+```
+
+At $\lambda = 0.8$ the multiplier is exactly $-1$: the scheme maps $X \mapsto 8 - X$ and so
+bounces forever between $1$ and $7$ without either converging or diverging, which is the
+undamped zigzag in the right panel.
+
+Damping the adjustment enough makes the algorithm find the equilibrium; not damping it
+enough makes the algorithm chase its own tail.
+
+Hold on to that observation.
+
+A recurring theme of this series is that *how* you adapt determines where, and whether, you end
+up.
+
+### Adaptive expectations
+
+Equation {eq}`relaxation` was introduced as an algorithm running in iteration count $k$.
+
+Reinterpret $k$ as **calendar time** $t$, read $X^*_t$ as the value people *expect*, and
+$X_t = h(X^*_{t-1})$ as what actually happens, and it becomes a theory of expectation
+formation:
+
+```{math}
+:label: adaptive_exp
+
+X^*_t = (1 - \lambda) X^*_{t-1} + \lambda X_t
+ = \lambda \sum_{j=0}^{\infty} (1 - \lambda)^j X_{t-j}.
+```
+
+This is the **adaptive expectations** scheme that Cagan {cite:p}`Cagan` used to study
+hyperinflations and that Friedman used to study consumption.
+
+Expectations are a geometrically declining distributed lag of past observations, with a
+single free parameter $\lambda$ describing beliefs.
+
+Cagan and Friedman took $\lambda$ as a free parameter and left open the question of *why*
+anyone would form expectations this way.
+
+### Muth's inverse problem
+
+{cite:t}`Muth1960` set out to eliminate $\lambda$ as a free parameter by turning the
+question around.
+
+Instead of asking what forecasts a given environment implies, he asked: **for what
+environment would exponential smoothing be the optimal forecast?**
+
+His answer is that {eq}`adaptive_exp` is the least-squares forecast of $X_{t+k}$ at every
+horizon $k$ if and only if $X_t$ follows
+
+```{math}
+:label: muth_process
+
+X_t = X_{t-1} + \epsilon_t - \theta \epsilon_{t-1},
+```
+
+with $\{\epsilon_t\}$ a martingale difference sequence, and the smoothing weight tied to the
+moving-average coefficient by
+
+$$
+\lambda = 1 - \theta .
+$$
+
+```{note}
+Sargent writes both the smoothing weight in {eq}`adaptive_exp` and the moving-average
+coefficient in {eq}`muth_process` as $\lambda$.
+
+We use $\theta$ for the second to keep the relationship $\lambda = 1 - \theta$ visible.
+
+Two other symbols work double shifts in this lecture, both following the book.
+
+In the money model below, $\lambda$ is no longer a gain but the gross growth rate $w_1/w_2$ of
+the bubble term, and $\gamma$ is no longer the curvature of a cost function but the coefficient
+linking the price level to the money supply.
+
+The code distinguishes them as `λ_m` and `γ_m`.
+```
+
+The forecasting rule inherits its one parameter from the stochastic process being forecast.
+
+Let's confirm this numerically: simulate {eq}`muth_process`, run exponential smoothing at a
+range of weights, and see which weight forecasts best.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Forecast MSE of exponential smoothing"
+ name: fig-br-smoothing-mse
+---
+def smoothing_mse(x, λ):
+ "One-step-ahead forecast MSE of exponential smoothing with weight λ."
+ f = np.empty_like(x)
+ f[0] = x[0]
+ for i in range(1, len(x)):
+ f[i] = λ * x[i] + (1 - λ) * f[i - 1]
+ return np.mean((x[1:] - f[:-1]) ** 2)
+
+
+rng = np.random.default_rng(0)
+T = 200_000
+ε = rng.standard_normal(T + 1) # one shock path, shared across all λ
+grid = np.linspace(0.02, 0.98, 97)
+
+fig, ax = plt.subplots(figsize=(7, 4))
+for θ in (0.2, 0.5, 0.8):
+ X = np.cumsum(ε[1:] - θ * ε[:-1]) # Muth's process
+ mse = np.array([smoothing_mse(X, λ) for λ in grid])
+ line, = ax.plot(grid, mse, lw=1.2, label=fr"$\theta = {θ}$")
+ ax.axvline(1 - θ, color=line.get_color(), lw=0.8, ls='--')
+ print(f"θ = {θ}: best λ = {grid[mse.argmin()]:.2f}, 1 - θ = {1 - θ:.2f},"
+ f" min MSE = {mse.min():.4f}")
+ax.set_xlabel(r"smoothing weight $\lambda$")
+ax.set_ylabel("forecast MSE")
+ax.set_ylim(0.9, 3)
+ax.legend(frameon=False)
+plt.show()
+```
+
+Each curve bottoms out exactly at its dashed line $\lambda = 1 - \theta$, and the minimized
+mean squared error is $\mathbb{V}[\epsilon_t] = 1$ up to simulation noise —
+exponential smoothing at that weight *is* the conditional expectation, and nothing can beat
+it.
+
+Muth's exercise was the first application of the rational expectations idea in the form that
+became standard: find the restrictions that link a forecasting scheme to the environment in
+which it is used.
+
+It also pushed later researchers to treat the **forecasting scheme itself** as the object in
+terms of which equilibrium is defined.
+
+### The dynamic analogue
+
+That is exactly what happens when we move from static to dynamic models.
+
+Let each agent choose a sequence rather than a single action, taking as given a **perceived
+law of motion** for the aggregate state,
+
+$$
+X_t = H(X_{t-1}, u_t),
+$$
+
+where $\{u_t\}$ is IID
+
+Solving the agent's dynamic program yields an individual decision rule
+$x_t = h(x_{t-1}, X_{t-1}, u_t)$.
+
+Imposing that the representative agent is representative, $x_t = X_t$, delivers the
+**actual law of motion**
+
+$$
+X_t = h(X_{t-1}, X_{t-1}, u_t) \equiv H^*(X_{t-1}, u_t),
+$$
+
+and hence a map from perceived to actual laws of motion,
+
+```{math}
+:label: t_map
+
+H^* = T(H).
+```
+
+A dynamic rational expectations equilibrium is a fixed point $H = T(H)$, the same idea as
+{eq}`static_ree`, but the fixed point now lives in a space of *functions* rather than a space
+of numbers.
+
+The relaxation algorithm carries over unchanged,
+
+```{math}
+:label: t_relaxation
+
+H^*_k = H^*_{k-1} + \lambda\bigl(T(H^*_{k-1}) - H^*_{k-1}\bigr),
+```
+
+except that it now revises entire expectations-generating functions in response to the gap
+between what they predicted and what happened.
+
+Every learning model in this series is some version of {eq}`t_relaxation` with the gain
+$\lambda$ made to decline over time and the revision driven by data rather than by an
+exact evaluation of $T$.
+
+```{seealso}
+{doc}`rational_expectations` develops the $T$ map in detail for a Lucas–Prescott industry model
+and computes its fixed point.
+
+{doc}`ls_learning` studies what happens when agents estimate the perceived law of motion by
+least squares while living inside the system their estimates help determine.
+```
+
+## Money and prices
+
+We now build the first of the two models that motivate everything that follows.
+
+A representative money-holder chooses nominal balances $m_t$ to carry from $t$ to $t+1$ to
+maximize
+
+```{math}
+:label: money_objective
+
+\ln\left(2 w_1 - \frac{m_t}{p_t}\right) + \ln\left(2 w_2 + \frac{m_t}{p^*_{t+1}}\right),
+\qquad w_1 > w_2 > 0,
+```
+
+where $p_t$ is the current price level and $p^*_{t+1}$ is the price level expected next
+period.
+
+The first term is consumption today, reduced by the real resources $m_t / p_t$ given up to
+acquire currency; the second is consumption tomorrow, augmented by the goods
+$m_t / p^*_{t+1}$ that the currency is expected to command.
+
+Differentiating {eq}`money_objective` with respect to $m_t$ and rearranging gives the demand
+for money
+
+```{math}
+:label: money_demand
+
+\frac{m_t}{p_t} = w_1 - w_2 \frac{p^*_{t+1}}{p_t} .
+```
+
+Real balances fall when currency is expected to lose value faster.
+
+This is a version of the demand function Cagan {cite:p}`Cagan` used to study hyperinflations,
+and it also arises in Samuelson's {cite:p}`Samuelson1958` overlapping generations model.
+
+To close the model we need a theory of $p^*_{t+1}$.
+
+Suppose the money supply grows at a constant rate,
+
+```{math}
+:label: money_supply
+
+M_{t+1} = \mu M_t ,
+```
+
+and that the household believes the price level is related to the money supply by
+
+```{math}
+:label: price_belief
+
+p_t = \gamma M_t + \lambda^t c ,
+```
+
+for constants $(\gamma, \lambda, c)$ that summarize its beliefs, all positive so that the
+price level stays positive.
+
+Knowing $\mu$, the household forecasts
+$p^*_{t+1} = \gamma \mu M_t + \lambda^{t+1} c$.
+
+Substituting the forecast and {eq}`price_belief` into {eq}`money_demand` gives money demand
+as a function of the current money supply,
+
+```{math}
+:label: money_demand_solved
+
+m_t = \gamma(w_1 - w_2 \mu) M_t + \lambda^t (w_1 - w_2 \lambda) c .
+```
+
+Notice a feature common to models in which expectations matter: the *demand* for money
+depends on its *supply*, because today's demand depends on tomorrow's expected price level,
+which is believed to depend on tomorrow's money supply.
+
+### Equilibrium
+
+Setting demand equal to supply, $m_t = M_t$, turns {eq}`money_demand_solved` into a
+functional equation,
+
+$$
+M_t = \gamma (w_1 - w_2 \mu) M_t + \lambda^t (w_1 - w_2 \lambda) c ,
+$$
+
+which must hold at every date.
+
+Matching the two terms gives
+
+```{math}
+:label: money_ree
+
+\gamma = (w_1 - \mu w_2)^{-1},
+\qquad
+\lambda = \frac{w_1}{w_2},
+\qquad
+c \geq 0 \ \text{ arbitrary} ,
+```
+
+so the equilibrium price level is
+
+```{math}
+:label: money_price_level
+
+p_t = (w_1 - \mu w_2)^{-1} M_t + \left(\frac{w_1}{w_2}\right)^t c .
+```
+
+The parameters $\gamma$ and $\lambda$ are pinned down.
+
+The constant $c$ is not.
+
+*Every $c \geq 0$ is a rational expectations equilibrium*, and expectations formed from
+{eq}`price_belief` are always exactly right in each of them.
+
+Let's verify that the residual really does vanish for any $c$ we care to try.
+
+```{code-cell} ipython3
+w1, w2, μ, M0 = 2.0, 1.0, 1.5, 1.0
+γ_m, λ_m = 1 / (w1 - μ * w2), w1 / w2
+
+t = np.arange(12)
+M = M0 * μ ** t
+
+def price_path(c):
+ "Equilibrium price level for bubble constant c."
+ return γ_m * M + λ_m ** t * c
+
+for c in (0.0, 0.5, 3.0, 25.0):
+ p = price_path(c)
+ p_star = γ_m * μ * M + λ_m ** (t + 1) * c # forecast of next period's price
+ m = w1 * p - w2 * p_star # money demand
+ print(f"c = {c:5}: max |demand - supply| = {np.max(np.abs(m - M)):.2e}")
+```
+
+Money demand equals money supply exactly, at every date, for every $c$.
+
+### The bubble
+
+Since $w_1 > w_2$, the second term in {eq}`money_price_level` grows at the gross rate
+$\lambda = w_1 / w_2 > 1$.
+
+In the equilibrium with $c = 0$ the price level is proportional to the money supply: the
+quantity theory in its textbook form.
+
+In every other equilibrium the price level carries a component that has nothing to do with
+the money supply and grows exponentially: a purely speculative **bubble**.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Price levels and real balances under bubbles"
+ name: fig-br-bubbles
+---
+fig, axes = plt.subplots(1, 2, figsize=(11, 4))
+
+for c in (0.0, 0.5, 3.0):
+ lab = f"$c = {c}$" + (" (quantity theory)" if c == 0 else "")
+ axes[0].plot(t, price_path(c), 'o-', ms=3, lw=1, label=lab)
+ axes[1].plot(t, M / price_path(c), 'o-', ms=3, lw=1, label=lab)
+
+axes[0].set_yscale('log')
+axes[0].set_ylabel("$p_t$ (log scale)")
+axes[1].set_ylabel("real balances $M_t / p_t$")
+axes[1].axhline(w1 - μ * w2, color='k', lw=0.8, ls='--')
+axes[1].set_ylim(0, 0.6)
+for ax in axes:
+ ax.set_xlabel("$t$")
+ ax.legend(frameon=False, fontsize=9)
+plt.tight_layout()
+plt.show()
+```
+
+The left panel shows the price level diverging further and further from the quantity-theory
+path as $c$ rises.
+
+The right panel shows what that does to the real value of the money stock: with $c = 0$ real
+balances are constant at $w_1 - \mu w_2$, while with $c > 0$ the bubble drives them steadily
+toward zero.
+
+Along a bubble path the gross inflation rate climbs toward $w_1 / w_2$, which from
+{eq}`money_demand` is precisely the rate at which the demand for real balances vanishes.
+
+The economy demonetizes itself, purely because everyone expects it to.
+
+```{code-cell} ipython3
+infl = pd.DataFrame(
+ {f"c = {c}": price_path(c)[1:] / price_path(c)[:-1] for c in (0.0, 0.5, 3.0)},
+ index=pd.Index(t[1:], name="t"),
+).round(3)
+infl
+```
+
+There is nothing in the model — no preference, no technology, no policy — that says which
+$c$ we are in.
+
+Rational expectations, as the paper puts it, "is not a sufficiently restrictive principle to
+determine outcomes."
+
+## Two currencies
+
+The indeterminacy gets worse when we allow more than one currency.
+
+Following {cite:t}`KarekenWallace1981`, keep {eq}`money_demand` as the demand for
+currency *in total*, and suppose there are two fiat currencies in supplies $M_{1t}$ and
+$M_{2t}$ that are perfect substitutes as long as their rates of return are equal:
+
+```{math}
+:label: equal_returns
+
+\frac{p^*_{1,t+1}}{p_{1t}} = \frac{p^*_{2,t+1}}{p_{2t}} .
+```
+
+This indifference about *which* currency to hold is what makes the exchange rate
+indeterminate.
+
+Let people believe the price levels are given by
+
+```{math}
+:label: two_currency_belief
+
+p_{1t} = \gamma_1 M_{1t} + \gamma_2 e M_{2t} + c \lambda^t ,
+\qquad
+p_{2t} = e^{-1} p_{1t} ,
+```
+
+where $e$ is a constant exchange rate.
+
+Requiring that the demand for currency, valued in units of currency 1, equal the total
+supply $M_{1t} + e M_{2t}$ gives
+
+```{math}
+:label: two_currency_ree
+
+\gamma_1 = (w_1 - \mu_1 w_2)^{-1},
+\quad
+\gamma_2 = (w_1 - \mu_2 w_2)^{-1},
+\quad
+\lambda = \frac{w_1}{w_2},
+\quad
+c \geq 0,
+\quad
+e \in [0, \infty) .
+```
+
+These equations are remarkable for what they leave out.
+
+The exchange rate $e$ is *entirely unrestricted* — if the equations have a solution for one
+$e$, they have a solution for every other — and the formulas for $\gamma_1$ and $\gamma_2$ do
+not involve $e$ at all.
+
+Take the simplest case: two currencies in fixed supply, $\mu_1 = \mu_2 = 1$, and set $c = 0$.
+
+Then $\gamma_1 = \gamma_2 = (w_1 - w_2)^{-1}$ and the price levels are constant.
+
+```{code-cell} ipython3
+w1, w2 = 2.0, 1.0
+H1, H2 = 100.0, 120.0 # fixed supplies of the two currencies
+γ_e = 1 / (w1 - w2)
+
+def two_currency(e):
+ "Price levels and the real allocation at exchange rate e."
+ p1 = γ_e * (H1 + e * H2)
+ p2 = p1 / e
+ supply = H1 + e * H2 # total currency, in units of currency 1
+ demand = (w1 - w2) * p1 # money demand, with p* = p since prices are constant
+ return p1, p2, supply / p1, demand - supply
+
+pd.DataFrame(
+ [two_currency(e) for e in (0.25, 0.5, 1.0, 2.0, 4.0)],
+ index=pd.Index([0.25, 0.5, 1.0, 2.0, 4.0], name="e"),
+ columns=["$p_1$", "$p_2$", "real balances", "excess demand"],
+).round(4)
+```
+
+Every row is an equilibrium.
+
+The nominal price levels move around a great deal as $e$ varies.
+
+Since $e$ is the value of a unit of currency 2 in units of currency 1, a larger $e$ means
+currency 2 is worth more, which raises $p_1$ and lowers $p_2$.
+
+But the last two columns tell the real story: total real balances are $w_1 - w_2 = 1$
+regardless of $e$, and markets clear exactly.
+
+*The real allocation is identical in every one of these equilibria.* The model determines
+what people consume and how much purchasing power the currency stock commands; it says
+nothing whatever about the rate at which the two monies exchange.
+
+This is a sharp version of a problem that has haunted international monetary theory, and it
+is not a knife-edge case: it is a continuum.
+
+## Where this leaves us
+
+We now have two models in which rational expectations is silent about something we would
+very much like to predict.
+
+There are three ways to respond.
+
+The first is to add restrictions to the environment until the equilibrium becomes unique.
+
+The second is to declare the indeterminacy a genuine feature of the world.
+
+The third — the one this series pursues — is to ask what happens when we **replace the
+rational agents with adaptive ones** and watch where the system goes.
+
+This is a substantive change, not a technicality.
+
+An adaptive agent is not endowed with the equilibrium; it has beliefs, a rule for revising
+them, and initial conditions.
+
+Those extra objects are exactly what the rational expectations equilibrium conditions failed
+to pin down, so a system of adaptive agents can select an outcome where rational expectations
+could not.
+
+The rest of this series takes that idea seriously, and the results are mixed in an instructive
+way.
+
+In the overlapping generations monetary economy, least squares learning selects the *opposite*
+equilibrium from the one the rational expectations dynamics converge to, and human experimental
+subjects side with the adaptive model.
+
+In the two-currency model above, adaptive agents do pin the exchange rate down, but only by
+making it depend on initial conditions.
+
+The rest points of the learning algorithm reproduce the indeterminacy exactly, and what selects
+an outcome is the dead hand of history.
+
+In a Kiyotaki–Wright search economy, adaptive agents learn to use a medium of exchange and
+select the *fundamental* equilibrium over the speculative one, even at parameters where theory
+says the speculative equilibrium is the only one.
+
+In each case the algorithm supplies what the equilibrium concept did not.
+
+Whether that is a discovery about economies or an artifact of the algorithm is the question the
+series keeps returning to, and {doc}`prospects_bounded_rationality` renders a verdict.
+
+## Exercises
+
+```{exercise-start}
+:label: br_ex1
+```
+
+The relaxation algorithm {eq}`relaxation` converges for the static market if and only if
+$\lambda < 2 / (1 + bn/\gamma)$.
+
+The ratio $bn/\gamma$ measures the slope of demand relative to the curvature of costs, so a
+market with steep demand and near-linear costs is one in which the naive cobweb
+($\lambda = 1$) is badly behaved.
+
+Verify the stability boundary numerically: for a grid of values of $bn/\gamma$, find the
+largest $\lambda$ (on a fine grid) for which the algorithm converges, and compare with the
+analytical prediction.
+
+```{exercise-end}
+```
+
+```{solution-start} br_ex1
+:class: dropdown
+```
+
+```{code-cell} ipython3
+def converges(slope, λ, n_iter=400, tol=1e-8):
+ """Does the relaxation algorithm converge when h'(X) = -slope?"""
+ a_, X = 10.0, 1.0
+ for _ in range(n_iter):
+ X_new = X + λ * ((a_ - slope * X) - X)
+ if not np.isfinite(X_new) or abs(X_new) > 1e12:
+ return False
+ X, X_prev = X_new, X
+ return abs(X - X_prev) < tol
+
+
+λ_grid = np.linspace(0.01, 1.5, 300)
+rows = []
+for slope in (0.5, 1.0, 1.5, 2.0, 4.0):
+ ok = [λ for λ in λ_grid if converges(slope, λ)]
+ rows.append((slope, max(ok) if ok else np.nan, 2 / (1 + slope)))
+
+pd.DataFrame(rows, columns=["$bn/\\gamma$", "largest $\\lambda$ found",
+ "$2/(1 + bn/\\gamma)$"]).round(3)
+```
+
+The numerical boundary sits consistently a little *below* the analytical one, and that gap is
+not grid spacing; it is a real feature of the test.
+
+As $\lambda$ approaches the boundary the multiplier approaches $-1$, so convergence becomes
+arbitrarily slow, and a test with a fixed iteration count and tolerance declares failure just
+before the true boundary.
+
+The size of the gap is predictable: with 400 iterations and a tolerance of $10^{-8}$, the test
+passes only while $|1 - \lambda(1 + bn/\gamma)|^{400} \lesssim 10^{-8}$, i.e. while the
+multiplier is below about $\exp(-18.4/400) = 0.955$ in absolute value.
+
+```{code-cell} ipython3
+predicted = [(1 + 0.955) / (1 + s_) for s_ in (0.5, 1.0, 1.5, 2.0, 4.0)]
+pd.DataFrame({"$bn/\\gamma$": [0.5, 1.0, 1.5, 2.0, 4.0],
+ "found": [r[1] for r in rows],
+ "predicted by the tolerance": predicted,
+ "true boundary": [r[2] for r in rows]}).round(3)
+```
+
+Note the first row: when $bn/\gamma < 1$ the boundary exceeds one, so the naive cobweb
+converges on its own and no damping is needed.
+
+Damping is what buys convergence in the steep markets, and the steeper the market the more
+damping is required.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: br_ex2
+```
+
+The equilibrium {eq}`money_ree` was derived without checking that it makes economic sense.
+
+Show that a monetary equilibrium requires $\mu < w_1 / w_2$, by finding what goes wrong with
+the demand for real balances when money grows faster than that.
+
+```{exercise-end}
+```
+
+```{solution-start} br_ex2
+:class: dropdown
+```
+
+Along the $c = 0$ equilibrium, real balances are constant at
+$M_t / p_t = \gamma^{-1} = w_1 - \mu w_2$.
+
+This is positive only when $\mu < w_1 / w_2$.
+
+If money grows faster, {eq}`money_ree` still "solves" the functional equation, but it asks
+the household to hold a negative quantity of currency.
+
+```{code-cell} ipython3
+w1, w2 = 2.0, 1.0
+μ_grid = np.array([0.5, 1.0, 1.5, 1.9, 2.0, 2.5])
+
+with np.errstate(divide='ignore'): # γ is infinite exactly at μ = w1/w2
+ table = pd.DataFrame({
+ "$\\mu$": μ_grid,
+ "$\\gamma = (w_1 - \\mu w_2)^{-1}$": 1 / (w1 - μ_grid * w2),
+ "real balances $w_1 - \\mu w_2$": w1 - μ_grid * w2,
+ "monetary equilibrium?": np.where(μ_grid < w1 / w2, "yes", "no"),
+ }).round(3)
+table
+```
+
+At $\mu = w_1 / w_2 = 2$ the demand for real balances hits zero and $\gamma$ blows up; beyond
+it, both are negative.
+
+The intuition runs through {eq}`money_demand`: the household holds currency only if the
+expected loss of purchasing power, $p^*_{t+1} / p_t$, is smaller than $w_1 / w_2$.
+
+Money growing at rate $\mu$ produces inflation at rate $\mu$, so $\mu \geq w_1 / w_2$ drives
+the demand for money to zero, the same boundary that the *bubble* equilibria approach
+asymptotically from below.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: br_ex3
+```
+
+In the two-currency example we set $\mu_1 = \mu_2 = 1$ and found that the real allocation was
+the same at every exchange rate.
+
+That is special.
+
+Repeat the calculation with $\mu_1 \neq \mu_2$ — say $\mu_1 = 1.0$ and $\mu_2 = 1.3$, with
+$w_1 = 2$, $w_2 = 1$, $M_{1,0} = M_{2,0} = 100$, and $c = 0$ — and compute total real
+balances at several exchange rates over the first several periods.
+
+Does the choice of $e$ still leave the real allocation untouched?
+
+```{exercise-end}
+```
+
+```{solution-start} br_ex3
+:class: dropdown
+```
+
+```{code-cell} ipython3
+w1, w2 = 2.0, 1.0
+μ1, μ2 = 1.0, 1.3
+γ1, γ2 = 1 / (w1 - μ1 * w2), 1 / (w1 - μ2 * w2)
+t = np.arange(10)
+M1, M2 = 100.0 * μ1 ** t, 100.0 * μ2 ** t
+
+def real_balances(e):
+ p1 = γ1 * M1 + γ2 * e * M2
+ return (M1 + e * M2) / p1
+
+pd.DataFrame({f"e = {e}": real_balances(e) for e in (0.25, 1.0, 4.0)},
+ index=pd.Index(t, name="t")).round(4)
+```
+
+```{code-cell} ipython3
+fig, ax = plt.subplots(figsize=(7, 4))
+for e in (0.25, 1.0, 4.0):
+ ax.plot(t, real_balances(e), 'o-', ms=3, lw=1, label=f"$e = {e}$")
+ax.axhline(1 / γ1, color='k', lw=0.8, ls='--', label=r"$1/\gamma_1$")
+ax.axhline(1 / γ2, color='gray', lw=0.8, ls=':', label=r"$1/\gamma_2$")
+ax.set_xlabel("$t$")
+ax.set_ylabel("total real balances")
+ax.legend(frameon=False)
+plt.show()
+```
+
+No. With unequal money growth rates the exchange rate affects the real allocation, and the
+allocation is no longer even constant over time.
+
+The reason is that $\gamma_1 \neq \gamma_2$: the two currencies are valued differently
+because they are expected to be diluted at different rates, so the *composition* of the
+currency stock matters, and $e$ is what fixes that composition.
+
+Since currency 2 grows faster, it comes to dominate the stock whatever $e$ we choose, and real
+balances converge to $1/\gamma_2$ from wherever $e$ starts them.
+
+So the pure nominal indeterminacy of the fixed-supply case is a knife-edge, but the
+indeterminacy of $e$ itself is not: every $e$ in the table is still an equilibrium.
+
+```{solution-end}
+```
diff --git a/lectures/exchange_rate_learning.md b/lectures/exchange_rate_learning.md
new file mode 100644
index 000000000..800898e8e
--- /dev/null
+++ b/lectures/exchange_rate_learning.md
@@ -0,0 +1,827 @@
+---
+jupytext:
+ text_representation:
+ extension: .md
+ format_name: myst
+ format_version: 0.13
+ jupytext_version: 1.17.1
+kernelspec:
+ display_name: Python 3 (ipykernel)
+ language: python
+ name: python3
+---
+
+(exchange_rate_learning)=
+```{raw} jupyter
+
+```
+
+# Exchange Rate Indeterminacy, Learning, and Experiments
+
+```{index} single: Bounded Rationality; Exchange Rate Indeterminacy
+```
+
+```{contents} Contents
+:depth: 2
+```
+
+## Overview
+
+In {doc}`bounded_rationality` we met a model in which rational expectations cannot pin down the
+exchange rate at all.
+
+Two fiat currencies, perfect substitutes as long as their rates of return are equal, leave the
+exchange rate $e$ *completely unrestricted*: if the equilibrium conditions have a solution for
+one $e$, they have a solution for every other, and the real allocation is identical across all
+of them.
+
+This lecture, following {cite:t}`Sargent1993`, expels the rational agents from that economy and
+puts adaptive ones in their place.
+
+Two questions organize what follows.
+
+**First**, does learning pin the exchange rate down?
+
+An adaptive agent has beliefs, a rule for revising them, and initial conditions, and those
+initial conditions are exactly what the equilibrium conditions failed to determine.
+
+So a system of adaptive agents *can* select an exchange rate where rational expectations could
+not.
+
+We shall see that it does, but in a very particular way.
+
+Newton–Raphson learners converge to a **determinate** exchange rate that depends entirely on
+where they started.
+
+The "dead hand of history" does the pinning that fundamentals refused to do.
+
+**Second**, does a ghost of the indeterminacy survive?
+
+It does.
+
+The rest points of the learning algorithm reproduce the indeterminacy exactly: the condition
+that stops the algorithm is the same arbitrage condition that made the exchange rate free in
+the first place.
+
+Sargent calls a regime that leaves the exchange rate history-dependent "a weak reed" on which
+to base exchange rate determination.
+
+Then we turn to evidence.
+
+{cite:t}`Arifovic1996` ran this economy as a laboratory experiment with paying human subjects.
+
+Their exchange rate never settled down.
+
+And when she replaced the Newton–Raphson learners with a **genetic algorithm** — a population
+of binary-string agents bred by selection, crossover, and mutation — she got an economy whose
+exchange rate wanders persistently, with a spectrum resembling real floating exchange rates.
+
+That contrast — a learning rule that converges versus one that generates permanent volatility —
+is where this lecture lands, and it is the on-ramp to {doc}`marimon_mcgrattan_sargent`, where
+genetic algorithms and classifier systems take over entirely.
+
+Let's start with some imports.
+
+```{code-cell} ipython3
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from typing import NamedTuple
+```
+
+## The two-currency economy
+
+We use the overlapping generations incarnation of the {cite:t}`KarekenWallace1981`
+indeterminacy.
+
+At each date $t$, $N$ two-period-lived agents are born, endowed with $w_1$ when young and $w_2$
+when old, with $w_1 > w_2$.
+
+There are two fiat currencies in fixed supplies $H_1$ and $H_2$.
+
+A young agent makes two decisions: how much to save, $s_t$, and what fraction $\lambda_t$ of
+that saving to hold in currency 1 (the rest going to currency 2).
+
+The realized lifetime utility of an agent who chooses $(s_t, \lambda_t)$ is
+
+```{math}
+:label: xr_utility
+
+U(s_t, \lambda_t)
+= u(w_1 - s_t)
++ u\!\left(w_2 + \lambda_t s_t \frac{p_{1t}}{p_{1,t+1}}
+ + (1 - \lambda_t) s_t \frac{p_{2t}}{p_{2,t+1}}\right),
+```
+
+where $p_{it}$ is the price level in currency $i$ and $p_{it}/p_{i,t+1}$ is the gross return on
+holding currency $i$.
+
+We take $u(c) = \ln c$ throughout.
+
+Equating each currency's supply to the demand for it gives the price levels
+
+```{math}
+:label: xr_prices
+
+p_{1t} = \frac{H_1}{\sum_i \lambda_{it} s_{it}},
+\qquad
+p_{2t} = \frac{H_2}{\sum_i (1 - \lambda_{it}) s_{it}},
+```
+
+and the exchange rate is $e_t = p_{1t}/p_{2t}$.
+
+### The indeterminacy, recalled
+
+Consider a stationary equilibrium with constant prices.
+
+Then each currency returns $p_{it}/p_{i,t+1} = 1$, so both returns are equal, and the portfolio
+return in {eq}`xr_utility` is $1$ regardless of $\lambda$.
+
+The saving decision then solves $\max_s \ln(w_1 - s) + \ln(w_2 + s)$, giving
+
+$$
+s^\star = \frac{w_1 - w_2}{2},
+$$
+
+but the portfolio share $\lambda$ is *completely undetermined*: utility does not depend on it
+when the two returns are equal.
+
+And $\lambda$ is what fixes the exchange rate.
+
+From {eq}`xr_prices`, with a common $(s, \lambda)$,
+
+$$
+e = \frac{p_{1}}{p_{2}} = \frac{H_1}{H_2}\cdot\frac{1 - \lambda}{\lambda}.
+$$
+
+Every $\lambda \in (0, 1)$ is an equilibrium, so every $e \in (0, \infty)$ is an equilibrium.
+
+This is the indeterminacy of {doc}`bounded_rationality`, now expressed through an undetermined
+portfolio choice.
+
+This lecture runs two economies with different parameters — Sargent's and Arifovic's — so we
+carry the parameters in a container and pass them explicitly rather than leaving them lying
+about as globals.
+
+```{code-cell} ipython3
+class Params(NamedTuple):
+ w1: float # endowment when young
+ w2: float # endowment when old
+ H1: float # fixed supply of currency 1
+ H2: float # fixed supply of currency 2
+
+ @property
+ def s_star(self):
+ "Rational expectations saving rate, when both currencies return 1."
+ return (self.w1 - self.w2) / 2
+
+
+kw = Params(w1=20.0, w2=15.0, H1=100.0, H2=120.0)
+
+print(f"rational expectations saving rate s* = {kw.s_star}")
+print(f"any portfolio share λ is an equilibrium, and e = (H1/H2)(1-λ)/λ is free")
+```
+
+## Newton–Raphson learning
+
+Now expel the rational agents.
+
+Following {cite:t}`Sargent1993`, we split the population into two classes — call them "even"
+and "odd" — because a cohort's lifetime utility cannot be evaluated until it is old.
+
+Each class carries its own rule $(s, \lambda)$, updated only from the experience of earlier
+agents of the same class.
+
+An agent revises $(s, \lambda)$ by a **Newton–Raphson** step against realized utility: move in
+the direction that a second-order expansion of {eq}`xr_utility` says would have raised utility,
+given the returns the agent actually experienced.
+
+Writing $g$ for the gradient and $R$ for a running estimate of the (negative-definite) Hessian,
+the recursion is
+
+```{math}
+:label: xr_newton
+
+\begin{aligned}
+R_{\tau+1} &= R_\tau + \gamma_\tau (H_\tau - R_\tau), \\
+\begin{bmatrix} s \\ \lambda \end{bmatrix}_{\tau+1}
+&= \begin{bmatrix} s \\ \lambda \end{bmatrix}_\tau
+ - \gamma_\tau R_{\tau+1}^{-1}\, g_\tau,
+\end{aligned}
+```
+
+with $H_\tau$ the realized Hessian and $\gamma_\tau$ a gain.
+
+Here is the gradient and Hessian of {eq}`xr_utility` in the returns
+$R_1 = p_{1t}/p_{1,t+1}$ and $R_2 = p_{2t}/p_{2,t+1}$.
+
+```{code-cell} ipython3
+def grad_hess(p, s, lam, R1, R2):
+ "Gradient and Hessian of realized utility U(s, λ)."
+ A = lam*R1 + (1 - lam)*R2 # portfolio gross return
+ c2 = p.w2 + s*A # old-age consumption
+ dR = R1 - R2
+ g = np.array([-1/(p.w1 - s) + A/c2,
+ s*dR/c2])
+ H_ss = -1/(p.w1 - s)**2 - A**2/c2**2
+ H_ll = -s**2 * dR**2 / c2**2
+ H_sl = dR/c2 - s*A*dR/c2**2
+ return g, np.array([[H_ss, H_sl], [H_sl, H_ll]])
+```
+
+### The indeterminacy shows up as a singular Hessian
+
+Look at the $\lambda$ block of the Hessian.
+
+When the two returns are equal, $R_1 = R_2$, every $\lambda$-derivative vanishes: the gradient
+component $s(R_1 - R_2)/c_2$ is zero, and so is the curvature $-s^2(R_1 - R_2)^2/c_2^2$.
+
+Utility is **flat in $\lambda$** at equal returns.
+
+That is the indeterminacy, seen locally: there is no force pushing $\lambda$ in any direction,
+so a Newton step — which divides the gradient by the curvature — is $0/0$ in the $\lambda$
+direction.
+
+For the algorithm to move at all we must keep $R$ invertible.
+
+We do that with a small **prior curvature** $\kappa$ in the $\lambda$ direction.
+
+Economically this is exactly the sluggishness that {cite:t}`Sargent1993` invokes: it is the
+"dead hand of history" that lets an otherwise-free exchange rate settle down.
+
+```{code-cell} ipython3
+def newton_learning(p, s0, lam0, T=400, gain=0.3, κ=0.5):
+ """
+ Two-currency OLG with Newton-Raphson learning of (s, λ), one rule per class
+ (even/odd). κ is a prior curvature in the indeterminate λ direction that
+ keeps the second-moment matrix R invertible.
+ """
+ s = np.array(s0, float)
+ lam = np.array(lam0, float)
+ R = [np.array([[-1.0, 0.0], [0.0, -κ]]) for _ in range(2)] # prior curvature
+ p1 = np.empty(T)
+ p2 = np.empty(T)
+ e = np.empty(T)
+ s_hist = np.empty((T, 2))
+ lam_hist = np.empty((T, 2))
+
+ for t in range(T):
+ j = t % 2 # class that is young at t
+ p1[t] = p.H1 / (lam[j]*s[j])
+ p2[t] = p.H2 / ((1 - lam[j])*s[j])
+ e[t] = p1[t] / p2[t]
+ if t >= 1: # class young at t-1 is now old
+ jp = 1 - j
+ R1, R2 = p1[t-1]/p1[t], p2[t-1]/p2[t]
+ g, H = grad_hess(p, s[jp], lam[jp], R1, R2)
+ H[1, 1] -= κ # retain prior λ-curvature
+ R[jp] = R[jp] + gain*(H - R[jp])
+ step = gain * np.linalg.solve(R[jp], g)
+ s[jp] = np.clip(s[jp] - step[0], 0.1, p.w1 - 0.1)
+ lam[jp] = np.clip(lam[jp] - step[1], 0.02, 0.98)
+ s_hist[t] = s
+ lam_hist[t] = lam
+
+ return e, s_hist, lam_hist
+```
+
+### Convergence to a history-dependent exchange rate
+
+Run two experiments that are identical except for the initial portfolio shares.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Newton-Raphson learning from two initial conditions"
+ name: fig-xr-learning
+---
+e1, s1, l1 = newton_learning(kw, [3.5, 2.0], [0.35, 0.40])
+e2, s2, l2 = newton_learning(kw, [2.0, 3.5], [0.62, 0.58])
+
+fig, axes = plt.subplots(1, 3, figsize=(14, 4))
+
+axes[0].plot(np.log(e1), 'C0', lw=1.3, label="experiment 1")
+axes[0].plot(np.log(e2), 'C1', lw=1.3, label="experiment 2")
+axes[0].set_title("log exchange rate")
+axes[0].set_xlabel("$t$")
+axes[0].legend(frameon=False)
+
+for s_h, c in [(s1, 'C0'), (s2, 'C1')]:
+ axes[1].plot(s_h[:, 0], color=c, lw=1.0)
+ axes[1].plot(s_h[:, 1], color=c, lw=1.0, ls=':')
+axes[1].axhline(kw.s_star, color='k', lw=0.8, ls='--')
+axes[1].set_title("saving (solid = even, dotted = odd)")
+axes[1].set_xlabel("$t$")
+
+for l_h, c in [(l1, 'C0'), (l2, 'C1')]:
+ axes[2].plot(l_h[:, 0], color=c, lw=1.0)
+ axes[2].plot(l_h[:, 1], color=c, lw=1.0, ls=':')
+axes[2].set_title(r"portfolio share $\lambda$")
+axes[2].set_xlabel("$t$")
+plt.tight_layout()
+plt.show()
+```
+
+Both economies converge.
+
+Saving climbs or falls to the rational expectations rate $s^\star = 2.5$ from either side, and
+the two classes' portfolio shares merge to a common value.
+
+But the two experiments converge to *different exchange rates*.
+
+The only thing that differed between them was where the portfolios started.
+
+```{code-cell} ipython3
+for name, e, l in [("experiment 1", e1, l1), ("experiment 2", e2, l2)]:
+ print(f"{name}: e → {e[-1]:.4f}, s → {s1[-1][0]:.3f}, λ → {l[-1][0]:.3f}")
+```
+
+The saving rate is pinned down by fundamentals; the exchange rate is pinned down by history.
+
+### The dead hand of history
+
+To see how completely history governs the outcome, sweep the initial portfolio share and record
+the exchange rate each economy settles on.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Limiting exchange rate against initial portfolio share"
+ name: fig-xr-limits
+---
+λ0_grid = np.linspace(0.15, 0.85, 15)
+e_limits = [newton_learning(kw, [4.0, 4.0], [λ0, λ0])[0][-1] for λ0 in λ0_grid]
+
+fig, ax = plt.subplots(figsize=(7, 4.5))
+ax.plot(λ0_grid, e_limits, 'C0o-', ms=5, label="limiting $e$ from learning")
+ax.plot(λ0_grid, (kw.H1/kw.H2)*(1 - λ0_grid)/λ0_grid, 'k--', lw=1,
+ label=r"$(H_1/H_2)(1-\lambda)/\lambda$")
+ax.set_xlabel(r"initial portfolio share $\lambda_0$")
+ax.set_ylabel("limiting exchange rate $e$")
+ax.legend(frameon=False)
+plt.show()
+```
+
+The limiting exchange rate traces out the *entire rational expectations continuum*.
+
+Every point on the curve is a valid rational expectations equilibrium; learning selects among
+them purely by initial condition.
+
+This is the sense in which learning renders the exchange rate determinate.
+
+It does not add a fundamental that fundamentals were missing.
+
+It converts an *indeterminacy of levels* into a *dependence on history*: the exchange rate the
+economy reaches is whatever the initial portfolio beliefs implied, frozen in place.
+
+### The ghost of indeterminacy
+
+Why does the exchange rate freeze rather than move to some particular value?
+
+Because the algorithm comes to rest wherever the gradient vanishes, and the portfolio gradient
+$s(R_1 - R_2)/c_2$ vanishes precisely when $R_1 = R_2$, the **arbitrage condition** that made the
+exchange rate free under rational expectations.
+
+```{code-cell} ipython3
+# at the limit the classes have merged, so prices are constant and R1 = R2 = 1
+s_lim, lam_lim = s1[-1], l1[-1]
+p1_lim = kw.H1 / (lam_lim*s_lim)
+p2_lim = kw.H2 / ((1 - lam_lim)*s_lim)
+print(f"classes converged to a common rule: s = {s_lim.round(4)}, λ = {lam_lim.round(4)}")
+print(f"→ prices constant across periods, so R1 = R2 = 1 (arbitrage holds at the rest point)")
+print(f"→ the λ-gradient is zero for *any* λ, so learning cannot move the exchange rate off "
+ f"wherever history left it")
+```
+
+The rest points of the learning dynamics *are* the rational expectations equilibria, exchange
+rate and all.
+
+The indeterminacy is not resolved; it is displaced into the initial conditions.
+
+Sargent is blunt about how much weight this can bear:
+
+> Put differently, a regime that allows the exchange rate to be history-dependent seems to be an
+> ill-formed mechanism.
+
+An exchange rate determined by nothing but the accident of initial beliefs is a weak reed.
+
+If the economy is a shade different — if the learning rule keeps experimenting rather than
+settling — the whole construction can come apart.
+
+That is exactly what the next two exhibits show.
+
+## Evidence from the laboratory
+
+{cite:t}`Arifovic1996` implemented this two-currency economy as an experiment with paying human
+subjects, using $w_1 = 11$, $w_2 = 1$, $H_1 = H_2 = 10$.
+
+Each young subject chose a saving rate and a fraction to allocate between the two currencies; the
+experimenters cleared the two money markets against those choices, exactly as in {eq}`xr_prices`.
+
+The result did not look like the Newton–Raphson simulations at all.
+
+The exchange rate fluctuated persistently, roughly within a band from $0.5$ to $2$, and showed
+no sign of settling down.
+
+If anything the amplitude grew across sessions.
+
+The simple Newton–Raphson model, which always converges to a constant, does a poor job of
+matching that.
+
+So Arifovic built a different model of the same economy, one in which the agents are a
+**population** bred by a genetic algorithm.
+
+## A genetic algorithm economy
+
+Now each class is a population of $N = 30$ agents, and each agent is a **binary string** of
+length $30$: the first $20$ bits encode its saving rate, the last $10$ its portfolio share.
+
+Every period the young population's strings are decoded into $(s_i, \lambda_i)$ pairs, the two
+money markets clear against the aggregates via {eq}`xr_prices`, and the exchange rate is read
+off.
+
+A generation later, when the cohort is old, each string's realized utility {eq}`xr_utility` is
+its **fitness**, and the genetic algorithm breeds the next generation.
+
+```{code-cell} ipython3
+N, L_s, L_lam = 30, 20, 10 # population size; bits for saving and portfolio
+
+arifovic = Params(w1=11.0, w2=1.0, H1=10.0, H2=10.0)
+
+def decode(p, pop):
+ "Binary strings → (s, λ), with s in (0, w1) and λ in (0, 1)."
+ ints_s = pop[:, :L_s] @ (1 << np.arange(L_s)[::-1])
+ ints_l = pop[:, L_s:] @ (1 << np.arange(L_lam)[::-1])
+ s = 0.05 + (p.w1 - 0.10) * ints_s / (2**L_s - 1)
+ lam = 0.02 + 0.96 * ints_l / (2**L_lam - 1)
+ return s, lam
+
+def market_prices(p, pop):
+ s, lam = decode(p, pop)
+ return p.H1 / np.sum(lam*s), p.H2 / np.sum((1 - lam)*s)
+
+def fitness(p, pop, R1, R2):
+ "Realized lifetime utility of each string given the returns it experienced."
+ s, lam = decode(p, pop)
+ c2 = p.w2 + s*(lam*R1 + (1 - lam)*R2)
+ return np.log(p.w1 - s) + np.log(np.maximum(c2, 1e-9))
+```
+
+The genetic algorithm has the classic three operators — fitness-proportional **selection** of
+parents, single-point **crossover**, and bit-flip **mutation** — plus one that Arifovic
+introduced, the **election operator**: a child is admitted to the next generation only if it
+would have been fitter than its parent on the most recent returns, otherwise the parent survives.
+
+```{code-cell} ipython3
+def genetic_step(p, pop, R1, R2, rng, p_mut=0.033, election=True):
+ "Breed one new generation from the current one."
+ fit = fitness(p, pop, R1, R2)
+ weight = fit - fit.min() + 1e-6 # shift to positive for roulette wheel
+ new = np.empty_like(pop)
+ for k in range(0, N, 2):
+ i, j = np.searchsorted(np.cumsum(weight), rng.random(2) * weight.sum())
+ parents = np.array([pop[i], pop[j]])
+ cut = rng.integers(1, L_s + L_lam) # single-point crossover
+ kids = parents.copy()
+ kids[0, cut:], kids[1, cut:] = parents[1, cut:], parents[0, cut:]
+ for c in kids: # mutation
+ c[rng.random(L_s + L_lam) < p_mut] ^= 1
+ if election: # Arifovic's election operator
+ f_kids = fitness(p, kids, R1, R2)
+ f_par = fitness(p, parents, R1, R2)
+ for m in range(2):
+ new[k + m] = kids[m] if f_kids[m] > f_par[m] else parents[m]
+ else:
+ new[k], new[k + 1] = kids
+ return new
+
+
+def genetic_economy(p, T=3000, seed=0, election=True):
+ rng = np.random.default_rng(seed)
+ pops = [rng.integers(0, 2, (N, L_s + L_lam)) for _ in range(2)] # even, odd
+ p1 = np.empty(T)
+ p2 = np.empty(T)
+ e = np.empty(T)
+ for t in range(T):
+ j = t % 2
+ p1[t], p2[t] = market_prices(p, pops[j])
+ e[t] = p1[t] / p2[t]
+ if t >= 1:
+ jp = 1 - j
+ R1, R2 = p1[t-1]/p1[t], p2[t-1]/p2[t]
+ pops[jp] = genetic_step(p, pops[jp], R1, R2, rng, election=election)
+ return e
+```
+
+### Volatility that never dies
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Exchange rate in the genetic-algorithm economy"
+ name: fig-xr-genetic
+---
+e_ga = genetic_economy(arifovic, T=4000, seed=1)
+log_e = np.log(e_ga[500:]) # drop a burn-in
+
+fig, ax = plt.subplots(figsize=(9, 4))
+ax.plot(log_e, lw=0.5)
+ax.set_xlabel("$t$")
+ax.set_ylabel("$\\log e_t$")
+plt.show()
+```
+
+The exchange rate wanders and keeps wandering.
+
+Unlike the Newton–Raphson learner, the genetic population never settles: mutation keeps
+injecting new strings, and the market keeps repricing them.
+
+The volatility is a permanent feature, not a transient.
+
+```{code-cell} ipython3
+early = np.log(e_ga[500:1500]).std()
+late = np.log(e_ga[3000:]).std()
+print(f"standard deviation of log e, early window: {early:.3f}")
+print(f"standard deviation of log e, late window: {late:.3f}")
+print("→ the volatility does not damp out over time")
+```
+
+### The shape of the volatility
+
+Arifovic reported that the exchange rate in her genetic economy behaves almost like a random walk
+— but with **mean reversion**, showing up as a dip in the spectrum of its first difference at
+zero frequency.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Spectrum and autocorrelation of the exchange rate"
+ name: fig-xr-spectrum
+---
+def averaged_spectrum(x, n_seg=16):
+ "Bartlett-averaged periodogram, for a readable spectral estimate."
+ x = x - x.mean()
+ seg = len(x) // n_seg
+ acc = np.zeros(seg // 2 + 1)
+ for k in range(n_seg):
+ acc += np.abs(np.fft.rfft(x[k*seg:(k+1)*seg]))**2 / seg
+ return np.fft.rfftfreq(seg), acc / n_seg
+
+
+d_log_e = np.diff(np.log(e_ga[500:]))
+d_log_e = np.clip(d_log_e, *np.percentile(d_log_e, [1, 99])) # winsorize tails
+freq, spec = averaged_spectrum(d_log_e)
+
+def acf(x, K):
+ x = x - x.mean()
+ return np.array([np.sum(x[k:]*x[:len(x)-k]) / np.sum(x*x) for k in range(K)])
+
+
+fig, axes = plt.subplots(1, 2, figsize=(12, 4))
+axes[0].plot(freq[1:], spec[1:], lw=1.2)
+axes[0].axhline(spec[1:].mean(), color='k', ls=':', lw=0.8)
+axes[0].set_title(r"spectrum of $\Delta \log e$")
+axes[0].set_xlabel("frequency")
+
+axes[1].bar(range(25), acf(np.log(e_ga[500:]), 25))
+axes[1].set_title(r"autocorrelation of $\log e$")
+axes[1].set_xlabel("lag")
+plt.tight_layout()
+plt.show()
+```
+
+The spectrum of the first difference is low near zero frequency and rises toward the higher
+frequencies, the signature of a series whose *level* is close to a random walk, but with enough
+mean reversion to pull the low-frequency power down.
+
+The autocorrelation of the level decays slowly, as a near-unit-root series would, but it *does*
+decay, which a pure random walk's would not.
+
+```{code-cell} ipython3
+band = slice(1, len(spec)//2)
+print(f"spectral power of Δlog e near zero frequency: {spec[1]:.4f}")
+print(f"average spectral power over low-to-mid band: {spec[band].mean():.4f}")
+print(f"→ pronounced dip at zero frequency (mean reversion): {spec[1] < spec[band].mean()}")
+```
+
+The book's assessment: real floating exchange rates for hard-currency pairs have spectra of log
+differences that look much like this, except *without* the dip at zero frequency; actual
+exchange rates are even closer to pure random walks.
+
+Arifovic's genetic economy, with no fundamental shocks at all, manufactures most of the way
+there out of nothing but a population of adapting agents repricing two intrinsically identical
+currencies.
+
+## Concluding remarks
+
+Two models of the same indeterminate economy gave two very different verdicts.
+
+The Newton–Raphson learner **converges** — and in converging, exposes the indeterminacy rather
+than curing it.
+
+Its rest point is an arbitrage condition that leaves the exchange rate free, so the limiting
+exchange rate is whatever history's initial portfolio beliefs implied.
+
+Determinacy by dead hand, which Sargent judges a weak reed.
+
+The genetic-algorithm economy **does not converge**.
+
+A whole population of adapting strings, constantly refreshed by mutation and repriced by the
+market, generates exchange rate volatility that never dies and that mimics the low-frequency
+behavior of real floating rates — from an economy with no fundamental disturbances whatsoever.
+
+The gap between them is not about the economics, which is identical, but about the *adaptive
+machinery*.
+
+A device that stabilizes a single agent's learning — the election operator, which keeps only
+improving offspring — turns out, inside a self-referential market, to sustain volatility rather
+than damp it ({ref}`xr_ex2`).
+
+Which learning model an economist reaches for is, once again, one of the choices that the
+bounded-rationality program forces into the open.
+
+That machinery — Holland's genetic algorithm, and its richer cousin the classifier system — is
+the subject of {doc}`marimon_mcgrattan_sargent`, where a population of adaptive agents must not
+merely tune a portfolio but *discover from scratch* which good will serve as money.
+
+## Exercises
+
+```{exercise-start}
+:label: xr_ex1
+```
+
+The Newton–Raphson economy converges to a rest point where the two classes share a common rule
+and the arbitrage condition $R_1 = R_2$ holds.
+
+Verify the claim that the limiting exchange rate depends on the *whole* initial condition, not
+just the initial $\lambda$, by starting the two classes **asymmetrically**.
+
+Run the learner from several initial conditions in which the even and odd classes begin with
+different portfolio shares, and confirm that (a) the classes' shares merge to a common value,
+(b) saving converges to $s^\star$, and (c) the limiting exchange rate depends on the starting
+point.
+
+Does the common limiting $\lambda$ equal the average of the two initial shares?
+
+```{exercise-end}
+```
+
+```{solution-start} xr_ex1
+:class: dropdown
+```
+
+```{code-cell} ipython3
+rows = []
+starts = [([3.0, 2.2], [0.30, 0.50]),
+ ([2.2, 3.0], [0.50, 0.30]),
+ ([3.5, 2.0], [0.35, 0.55]),
+ ([2.5, 2.5], [0.40, 0.60])]
+for s0, lam0 in starts:
+ e, s_h, l_h = newton_learning(kw, s0, lam0, T=600)
+ rows.append([str(lam0), l_h[-1][0], l_h[-1][1], s_h[-1].mean(),
+ e[-1], np.mean(lam0)])
+
+pd.DataFrame(rows, columns=["initial λ", "λ even (limit)", "λ odd (limit)",
+ "s (limit)", "e (limit)", "mean of initial λ"]).round(4)
+```
+
+The two classes' shares always merge (columns 2 and 3 agree), saving always converges to
+$s^\star = 2.5$, and the limiting exchange rate differs across starting points — so it is
+genuinely history-dependent.
+
+But the common limiting $\lambda$ is **not** simply the average of the two initial shares:
+compare the merged value with the last column.
+
+The transient path matters, because while the classes still differ the returns differ, and
+those transient return gaps push $\lambda$ around before the arbitrage condition shuts the
+motion off.
+
+The exchange rate records the whole history of the adjustment, not just its starting average.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: xr_ex2
+```
+
+Arifovic's **election operator** admits a child to the next generation only if it beats its parent
+on the latest returns.
+
+In single-agent optimization such a filter can only help — it never lets a worse rule replace a
+better one.
+
+But this economy is *self-referential*: the returns against which fitness is measured are
+themselves determined by the population's choices.
+
+Investigate what the election operator does to exchange rate volatility.
+
+Run the genetic economy with and without it across several seeds, and report the volatility of
+the log exchange rate.
+
+Which way does the operator push volatility, and can you explain why?
+
+```{exercise-end}
+```
+
+```{solution-start} xr_ex2
+:class: dropdown
+```
+
+```{code-cell} ipython3
+rows = []
+for election in (True, False):
+ sds = [np.log(genetic_economy(arifovic, T=2500, seed=sd,
+ election=election)[500:]).std()
+ for sd in range(6)]
+ rows.append([election, np.mean(sds), np.min(sds), np.max(sds)])
+
+pd.DataFrame(rows, columns=["election operator", "mean s.d. of log e",
+ "min", "max"]).round(3)
+```
+
+The election operator **raises** volatility — the opposite of its stabilizing role in
+single-agent search.
+
+The mechanism is the self-reference.
+
+With the filter on, the population keeps only offspring that beat their parents *on last
+period's returns*, so it chases the recently profitable portfolio.
+
+But "recently profitable" depends on the exchange rate, which the population's own chasing
+moves — so the whole population piles into the recent winner, overshoots, and the exchange rate
+lurches to make a different portfolio profitable, whereupon the population chases that.
+
+Without the filter, mutation keeps the population diffuse, and diverse portfolios partly cancel
+in the aggregates {eq}`xr_prices`, damping the swings.
+
+A device that unambiguously improves an isolated learner can destabilize a market of learners.
+
+It is a compact warning about reading single-agent intuitions into multi-agent systems — a
+theme that returns in {doc}`marimon_mcgrattan_sargent`.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: xr_ex3
+```
+
+The lecture claimed the genetic economy's exchange rate is "close to a random walk, but with mean
+reversion."
+
+Make the claim quantitative.
+
+Treating $\log e_t$ as data, estimate the first-order autoregressive coefficient in $\log e_t = \mu + \phi \log e_{t-1} + \varepsilon_t$,
+and compare it with $1$ (a pure random walk).
+
+Do this for several seeds.
+
+Is $\phi$ close to but below $1$, consistent with a near-unit-root, mean-reverting series?
+
+```{exercise-end}
+```
+
+```{solution-start} xr_ex3
+:class: dropdown
+```
+
+```{code-cell} ipython3
+def ar1_coefficient(x):
+ "OLS estimate of φ in x_t = μ + φ x_{t-1} + ε."
+ x0, x1 = x[:-1], x[1:]
+ X = np.column_stack([np.ones_like(x0), x0])
+ μ, φ = np.linalg.lstsq(X, x1, rcond=None)[0]
+ return φ
+
+rows = []
+for sd in range(6):
+ le = np.log(genetic_economy(arifovic, T=3000, seed=sd)[500:])
+ rows.append([sd, ar1_coefficient(le)])
+
+table = pd.DataFrame(rows, columns=["seed", "AR(1) coefficient φ"])
+print(table.round(4).to_string(index=False))
+print(f"\nmean φ across seeds: {table['AR(1) coefficient φ'].mean():.4f}")
+```
+
+The estimated $\phi$ is consistently close to $1$ but strictly below it — a highly persistent
+series that nonetheless reverts to its mean rather than drifting forever.
+
+That is precisely the "near random walk with mean reversion" the spectrum showed: a pure random
+walk would have $\phi = 1$ exactly (and no dip at zero frequency), while a value a little under
+$1$ produces the slow autocorrelation decay and the low-frequency dip we saw.
+
+The genetic economy lands in that near-unit-root region on its own, with no fundamental shocks
+driving it — the persistence is manufactured entirely by the population dynamics repricing two
+identical currencies.
+
+```{solution-end}
+```
diff --git a/lectures/genetic_classifier.md b/lectures/genetic_classifier.md
new file mode 100644
index 000000000..40151d3a0
--- /dev/null
+++ b/lectures/genetic_classifier.md
@@ -0,0 +1,975 @@
+---
+jupytext:
+ text_representation:
+ extension: .md
+ format_name: myst
+ format_version: 0.13
+ jupytext_version: 1.17.1
+kernelspec:
+ display_name: Python 3 (ipykernel)
+ language: python
+ name: python3
+---
+
+(genetic_classifier)=
+```{raw} jupyter
+
+```
+
+# Genetic Algorithms and Classifier Systems
+
+```{index} single: Bounded Rationality; Genetic Algorithms
+```
+
+```{contents} Contents
+:depth: 2
+```
+
+## Overview
+
+The lectures so far gave adaptive agents a fairly econometric brain.
+
+In {doc}`olg_adaptive_money` and {doc}`learning_approximation` they ran recursive least
+squares; in {doc}`exchange_rate_learning` they took Newton steps against realized utility.
+
+In each case an agent held a *parametric* rule and tuned its coefficients.
+
+This lecture surveys a different toolbox, the one {cite:t}`Sargent1993` draws from John
+Holland's work on artificial intelligence and from the connectionist literature on neural
+networks.
+
+These devices do not tune the coefficients of a fixed rule.
+
+They **discover the rule itself**, out of a large space of possibilities, from experience.
+
+Sargent's framing is that the bounded-rationality program asks us to populate models with agents
+who "behave like econometricians," and that the artificial-intelligence literature is a catalogue
+of candidate brains for them:
+
+> It is from the store of methods built up in these literatures that we shall select the 'brains'
+> to give our boundedly rational agents.
+
+We tour four of them:
+
+1. the **perceptron**, the simplest neural network, which turns out to be a linear discriminant
+ function, a direct link to the "agents as econometricians" reading;
+1. the **Hopfield network**, an associative memory that recovers stored patterns from corrupted
+ inputs by rolling downhill on an energy function;
+1. the **genetic algorithm**, Holland's population-based search, which we set loose on Axelrod's
+ iterated prisoner's dilemma and watch discover cooperation;
+1. the **classifier system**, Holland's "brain as a competitive economy" of if-then rules, whose
+ simplest instance — a two-armed bandit — already exposes a subtle limitation.
+
+We close with **evolutionary programming**: the idea that a population of adaptive agents,
+plodding toward an equilibrium, can be used to *compute* equilibria that we cannot solve for
+directly.
+
+That is the idea {doc}`marimon_mcgrattan_sargent` carries out in full, so this lecture is the
+machinery on which the next one is built.
+
+Let's start with some imports.
+
+```{code-cell} ipython3
+import numpy as np
+import matplotlib.pyplot as plt
+```
+
+## The perceptron: an agent as a discriminant function
+
+The simplest neural network is a single **perceptron**: $k$ inputs $x_i$, weights $w_i$, and one
+output
+
+$$
+y = S\!\left(\sum_{i=1}^k w_i x_i\right) = S(w^\top x),
+$$
+
+where $S$ is a "squasher" mapping $\mathbb{R}$ onto $[0, 1]$: a step function, or the sigmoid
+$S(z) = 1/(1 + e^{-z})$, or any cumulative distribution function.
+
+For fixed weights the perceptron is a **classifier**: it fires ($y = 1$) when $w^\top x > 0$ and stays
+quiet otherwise, so the boundary $w^\top x = 0$ is a hyperplane separating two classes.
+
+Training the perceptron — choosing $w$ to minimize $\sum_t (y_t - S(w^\top x_t))^2$ — is a nonlinear
+least squares problem, solved by the familiar stochastic-gradient recursion
+
+$$
+w_t = w_{t-1} + \gamma_t \nabla S(w_{t-1}, x_t)\,(y_t - S(w_{t-1}^\top x_t)),
+$$
+
+which is the same $1/t$-flavored update that ran the learning models of the previous lectures.
+
+Take the book's example: separate football players from economists on two standardized features.
+
+```{code-cell} ipython3
+rng = np.random.default_rng(0)
+n = 100
+economists = rng.multivariate_normal([-1.0, -0.3], [[0.5, 0.1], [0.1, 0.5]], n)
+players = rng.multivariate_normal([1.2, 0.8], [[0.6, 0.0], [0.0, 0.6]], n)
+X = np.vstack([economists, players])
+y = np.r_[np.zeros(n), np.ones(n)]
+X_aug = np.column_stack([np.ones(len(X)), X]) # prepend an intercept
+
+def sigmoid(z):
+ return 1 / (1 + np.exp(-z))
+
+w = np.zeros(3)
+for _ in range(200): # train by stochastic gradient
+ for t in rng.permutation(len(X)):
+ w += 0.1 * (y[t] - sigmoid(X_aug[t] @ w)) * X_aug[t]
+
+print(f"training accuracy: {np.mean((sigmoid(X_aug @ w) > 0.5) == y):.3f}")
+```
+
+The point {cite:t}`Sargent1993` stresses is that this is nothing exotic to an econometrician:
+the perceptron's decision boundary is essentially a **linear discriminant function**.
+
+Compare the perceptron's boundary normal to Fisher's linear discriminant direction.
+
+```{code-cell} ipython3
+m0, m1 = X[y == 0].mean(0), X[y == 1].mean(0)
+S_within = np.cov(X[y == 0].T) * n + np.cov(X[y == 1].T) * n
+lda_direction = np.linalg.solve(S_within, m1 - m0)
+lda_direction /= np.linalg.norm(lda_direction)
+perceptron_direction = w[1:] / np.linalg.norm(w[1:])
+
+print(f"perceptron boundary normal : {perceptron_direction.round(3)}")
+print(f"Fisher discriminant direction: {lda_direction.round(3)}")
+print(f"cosine similarity: {abs(perceptron_direction @ lda_direction):.4f}")
+```
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Perceptron boundary separating the two groups"
+ name: fig-gc-perceptron
+---
+fig, ax = plt.subplots(figsize=(6.5, 5))
+ax.scatter(*economists.T, s=18, c='C0', label="economists ($y=0$)")
+ax.scatter(*players.T, s=18, c='C1', label="football players ($y=1$)")
+xs = np.linspace(X[:, 0].min(), X[:, 0].max(), 2)
+ax.plot(xs, -(w[0] + w[1]*xs) / w[2], 'k-', lw=1.5, label="perceptron boundary")
+ax.set_xlabel("weight (standardized)")
+ax.set_ylabel("salary (standardized)")
+ax.legend(frameon=False)
+plt.show()
+```
+
+The two directions are all but identical.
+
+Minsky and Papert's famous critique {cite:p}`MinskyPapert1969` was precisely that a single
+perceptron can represent *only* a linear discriminant, and so cannot separate classes that a
+line cannot.
+
+The revival of the field came with the recognition that **layering** perceptrons — feeding
+their outputs into further perceptrons — approximates any nonlinear discriminant, the subject
+of {doc}`back_prop`.
+
+For our purposes the perceptron is the entry in the catalogue closest to what the previous
+lectures already did: a parametric rule, trained by gradient descent, that an econometrician
+would recognize on sight.
+
+The remaining three brains are stranger.
+
+## Associative memory: the Hopfield network
+
+The second device stores patterns and recalls them from corrupted fragments.
+
+Represent a pattern as a vector of $\pm 1$ of length $N$.
+
+We want to store $p$ patterns $\sigma^1, \ldots, \sigma^p$ so that each is a fixed point of a
+dynamical system, and so that feeding in a corrupted version converges quickly to the nearest
+stored pattern.
+
+The **Hopfield network** does this with the dynamics
+
+$$
+s(t+1) = \operatorname{sgn}(w\, s(t)),
+$$
+
+and a weight matrix built from the patterns themselves.
+
+When the patterns are orthogonal, Hebb's rule $w = \tfrac{1}{N}\sigma\sigma^\top$ suffices; for
+merely linearly independent (correlated) patterns one uses the projection rule $w = \tfrac{1}{N}\sigma V^{-1}\sigma^\top$
+with $V = \tfrac{1}{N}\sigma^\top \sigma$.
+
+Both make each stored pattern an exact fixed point.
+
+We store five letters on a $5\times5$ pixel grid, and because letters share many pixels, they are
+correlated, so the projection rule is the right one.
+
+```{code-cell} ipython3
+PATTERNS = {
+ 'A': ["01110", "10001", "11111", "10001", "10001"],
+ 'E': ["11111", "10000", "11110", "10000", "11111"],
+ 'I': ["11111", "00100", "00100", "00100", "11111"],
+ 'O': ["01110", "10001", "10001", "10001", "01110"],
+ 'T': ["11111", "00100", "00100", "00100", "00100"],
+}
+
+def to_vector(rows):
+ return np.array([1 if c == '1' else -1 for r in rows for c in r])
+
+letters = list(PATTERNS)
+σ = np.array([to_vector(PATTERNS[c]) for c in letters]) # (p, N)
+
+def projection_rule(patterns):
+ "Weight matrix making each (correlated) pattern an exact fixed point."
+ N = patterns.shape[1]
+ Σ = patterns.T # (N, p)
+ V = Σ.T @ Σ / N
+ return Σ @ np.linalg.inv(V) @ Σ.T / N
+
+def energy(w, s):
+ "Hopfield energy; stored patterns are local minima."
+ return -0.5 * s @ w @ s
+
+def recall(w, s0, max_iter=30):
+ "Iterate sgn(w s) to a fixed point."
+ s = s0.copy()
+ for _ in range(max_iter):
+ s_new = np.sign(w @ s)
+ s_new[s_new == 0] = 1
+ if np.array_equal(s_new, s):
+ break
+ s = s_new
+ return s
+
+w_hop = projection_rule(σ)
+fixed = all(np.array_equal(np.sign(w_hop @ σ[i]), σ[i]) for i in range(len(letters)))
+print(f"all stored letters are fixed points: {fixed}")
+print(f"energy of each stored letter: {[round(energy(w_hop, σ[i]), 1) for i in range(len(letters))]}")
+```
+
+Every stored letter sits at the same energy $-N/2$ and is a fixed point.
+
+Now corrupt a few pixels and let the network fall back to the nearest memory.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Hopfield recall of letters from corrupted inputs"
+ name: fig-gc-hopfield
+---
+rng = np.random.default_rng(3)
+show = ['E', 'O', 'T']
+
+fig, axes = plt.subplots(len(show), 3, figsize=(6, 6))
+for row, letter in enumerate(show):
+ i = letters.index(letter)
+ corrupt = σ[i].copy()
+ corrupt[rng.choice(25, 4, replace=False)] *= -1 # flip 4 pixels
+ recovered = recall(w_hop, corrupt)
+ for col, (img, title) in enumerate([(σ[i], "stored"),
+ (corrupt, "corrupted"),
+ (recovered, "recovered")]):
+ axes[row, col].imshow(img.reshape(5, 5), cmap='binary', vmin=-1, vmax=1)
+ axes[row, col].set_xticks([])
+ axes[row, col].set_yticks([])
+ if row == 0:
+ axes[row, col].set_title(title)
+plt.tight_layout()
+plt.show()
+```
+
+The corrupted letters snap back to their originals.
+
+How reliably depends on how much is corrupted and on how correlated the patterns are.
+
+A corrupted letter can occasionally land in a *different* letter's basin, or in a spurious
+mixture the designer never stored.
+
+```{code-cell} ipython3
+for n_flip in (3, 5, 7):
+ correct = 0
+ for i in range(len(letters)):
+ for _ in range(40):
+ corrupt = σ[i].copy()
+ corrupt[rng.choice(25, n_flip, replace=False)] *= -1
+ if np.array_equal(recall(w_hop, corrupt), σ[i]):
+ correct += 1
+ print(f"{n_flip} pixels flipped: recovered exactly {correct}/{len(letters)*40}")
+```
+
+The mechanism is worth naming because it recurs throughout this program.
+
+Recall is **descent on an energy function** whose local minima are the stored patterns.
+
+Feeding in a corrupted pattern places the system on the energy surface near the right minimum,
+and the dynamics roll it down.
+
+This is exactly the geometry behind **simulated annealing**, the method the book invokes for
+escaping *unwanted* local minima: add a controlled amount of random "shaking," large at first
+and declining over time, so the system can hop out of shallow spurious basins before settling
+into a deep one.
+
+It is the stochastic cousin of Newton's method, and it reappears as the "genetic"
+experimentation in classifier systems.
+
+## The genetic algorithm
+
+The third brain does not descend a smooth surface at all.
+
+Holland's **genetic algorithm** searches "rugged landscapes" that lack the smoothness Newton's
+method needs, by evolving a *population* of candidate solutions encoded as binary strings.
+
+Given a fitness function $f$ to maximize, and a population of $N$ strings of length $S$, one
+generation applies four operators:
+
+1. **Evaluation**: compute each string's fitness $f(x_i)$.
+1. **Reproduction**: copy strings into the next generation with probability proportional to
+ fitness (a "biased roulette wheel").
+1. **Crossover**: pair strings up and swap their tails at a random cut point, forming children
+ that recombine their parents' segments.
+1. **Mutation**: flip each bit independently with a small probability $p_m$.
+
+Reproduction concentrates the population on what already works; crossover and mutation inject
+new candidates.
+
+Crossover is the workhorse: it "preserves long sections of genetic structure" — Holland's
+*schemata* — while still exploring, provided the population is diverse enough to recombine.
+
+Crucially, the individual strings do not learn: each lives one generation and dies.
+
+Only the *society*, the sequence of populations, learns.
+
+That is what makes the genetic algorithm awkward as a model of one brain, and a better fit as a
+model of a population or a market, a point that matters when we reach
+{doc}`marimon_mcgrattan_sargent`.
+
+### Axelrod's iterated prisoner's dilemma
+
+{cite:t}`Axelrod1987` used the genetic algorithm to evolve strategies for the **iterated prisoner's
+dilemma**, the game whose round-robin tournament, famously, was won by tit-for-tat.
+
+Two players each choose to Cooperate or Defect; the payoffs reward defection against a cooperator
+but punish mutual defection.
+
+```{code-cell} ipython3
+# 1 = Cooperate, 0 = Defect; T > R > P > S is the prisoner's dilemma ranking
+T_pay, R_pay, P_pay, S_pay = 5, 3, 1, 0
+
+def payoff(a, b):
+ "My payoff when I play a against opponent's b."
+ if a and b: return R_pay # both cooperate
+ if not a and not b: return P_pay # both defect
+ if not a and b: return T_pay # I defect, they cooperate
+ return S_pay # I cooperate, they defect
+```
+
+Following Axelrod, a strategy is a **70-bit string**: a strategy conditions on the outcomes of
+the last three rounds.
+
+Each round is one of four outcomes (my move, opponent's move), so three rounds give $4^3 = 64$
+possible histories, and 64 bits specify the action in each.
+
+The remaining 6 bits encode a presumed pre-game history to get the first moves going.
+
+```{code-cell} ipython3
+def play(gene, opponent, n_rounds=150):
+ "Play a 70-bit genetic strategy against an opponent; return average payoffs."
+ action, premise = gene[:64], gene[64:]
+ hist = list(premise) # last 3 rounds as [my, opp] pairs
+ my_moves, opp_moves = [], []
+ my_total = opp_total = 0
+ for t in range(n_rounds):
+ idx = 0
+ for bit in hist:
+ idx = (idx << 1) | bit # 6 history bits -> index 0..63
+ a = action[idx] # my move from the lookup table
+ b = opponent(my_moves, opp_moves, t)
+ my_total += payoff(a, b)
+ opp_total += payoff(b, a)
+ my_moves.append(a)
+ opp_moves.append(b)
+ hist = [a, b] + hist[:4] # roll the 3-round window
+ return my_total / n_rounds, opp_total / n_rounds
+```
+
+The strategy is bred to do well against a fixed panel of opponents.
+
+Each panel member is a function of the histories *as that member sees them*: its first
+argument is what its opponent has played and its second is what it has played itself.
+
+```{code-cell} ipython3
+panel_rng = np.random.default_rng(2024)
+
+def all_cooperate(opp_moves, own_moves, t): return 1
+def all_defect(opp_moves, own_moves, t): return 0
+def tit_for_tat(opp_moves, own_moves, t): return 1 if t == 0 else opp_moves[-1]
+def grudger(opp_moves, own_moves, t): return 0 if 0 in opp_moves else 1
+def random_play(opp_moves, own_moves, t): return int(panel_rng.random() < 0.5)
+
+PANEL = {'AllC': all_cooperate, 'AllD': all_defect, 'TFT': tit_for_tat,
+ 'Grudger': grudger, 'Random': random_play}
+
+def fitness(gene):
+ "Average payoff against the whole panel."
+ return np.mean([play(gene, opp)[0] for opp in PANEL.values()])
+```
+
+Tit-for-tat copies what its opponent last did, and the grudger defects forever once its
+opponent has defected once.
+
+Now evolve.
+
+```{code-cell} ipython3
+def evolve(N=60, generations=80, p_mut=0.01, seed=0):
+ rng = np.random.default_rng(seed)
+ pop = rng.integers(0, 2, (N, 70))
+ best_hist, mean_hist = [], []
+ for _ in range(generations):
+ fit = np.array([fitness(pop[i]) for i in range(N)])
+ best_hist.append(fit.max())
+ mean_hist.append(fit.mean())
+ weight = fit - fit.min() + 1e-6 # shift positive for roulette
+ nxt = np.empty_like(pop)
+ for k in range(0, N, 2):
+ i, j = np.searchsorted(np.cumsum(weight), rng.random(2) * weight.sum())
+ cut = rng.integers(1, 70) # single-point crossover
+ c1 = np.concatenate([pop[i, :cut], pop[j, cut:]])
+ c2 = np.concatenate([pop[j, :cut], pop[i, cut:]])
+ for c in (c1, c2):
+ c[rng.random(70) < p_mut] ^= 1 # mutation
+ nxt[k], nxt[k+1] = c1, c2
+ pop = nxt
+ fit = np.array([fitness(pop[i]) for i in range(N)])
+ return pop[fit.argmax()], np.array(best_hist), np.array(mean_hist)
+
+
+panel_rng = np.random.default_rng(0) # reset the panel's randomizer
+champion, best_hist, mean_hist = evolve()
+print(f"generation 0: best fitness {best_hist[0]:.3f}")
+print(f"generation {len(best_hist)-1}: best fitness {best_hist[-1]:.3f}")
+```
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Population fitness across generations"
+ name: fig-gc-fitness
+---
+fig, ax = plt.subplots(figsize=(7.5, 4))
+ax.plot(best_hist, label="best in population", lw=2)
+ax.plot(mean_hist, label="population mean", lw=2)
+ax.set_xlabel("generation")
+ax.set_ylabel("panel fitness")
+ax.legend(frameon=False)
+plt.show()
+```
+
+Fitness climbs, then holds.
+
+What kind of strategy did evolution find?
+
+```{code-cell} ipython3
+print("evolved champion's average payoff against each opponent:")
+for name, opp in PANEL.items():
+ me, them = play(champion, opp)
+ print(f" vs {name:8s}: me = {me:.2f}, opponent = {them:.2f}")
+```
+
+The evolved strategy is *nice and retaliatory*, in Axelrod's language.
+
+It reaches mutual cooperation with the cooperative opponents (AllC, TFT, Grudger), refuses to
+be suckered by AllD (settling into mutual defection near the punishment payoff), and exploits
+Random.
+
+Nobody told it to cooperate; cooperation emerged because it pays against a panel that contains
+cooperators.
+
+And it does all this *better than tit-for-tat itself* would against the same panel.
+
+```{code-cell} ipython3
+def play_function(strategy, opponent, n=150):
+ "Average payoff of a stateful strategy function against an opponent."
+ mine, theirs, total = [], [], 0
+ for t in range(n):
+ # each side is passed its opponent's history first, then its own
+ a, b = strategy(theirs, mine, t), opponent(mine, theirs, t)
+ total += payoff(a, b)
+ mine.append(a)
+ theirs.append(b)
+ return total / n
+
+panel_rng = np.random.default_rng(0)
+tft_fitness = np.mean([play_function(tit_for_tat, opp) for opp in PANEL.values()])
+print(f"tit-for-tat's own panel fitness: {tft_fitness:.3f}")
+print(f"evolved champion's panel fitness: {best_hist[-1]:.3f}")
+```
+
+This reproduces Axelrod's headline finding: the genetic algorithm "produced a strategy that
+would have won the tournament, and in particular would have outplayed the 'tit-for-tat'
+strategy that won the tournament."
+
+It behaves much like tit-for-tat against cooperators, but its extra memory lets it squeeze more
+out of exploitable opponents that tit-for-tat leaves alone.
+
+The margin is not large, and it should not be.
+
+Tit-for-tat is a strong strategy against this panel; what the extra memory buys is a modest
+edge against the opponents that tit-for-tat handles adequately but not optimally.
+
+## Classifier systems
+
+The genetic algorithm evolves a population in which no individual learns.
+
+Holland's **classifier system** puts the same evolutionary machinery *inside a single agent*,
+as a model of one brain.
+
+Sargent describes it as Holland's vision of the mind as a **competitive economy**:
+
+> The statements compete with one another for the opportunity to decide. The classifier system
+> incorporates elements of the genetic algorithm with other aspects in a way that represents a
+> brain in terms that Holland describes as a competitive economy.
+
+A classifier system has:
+
+* **Classifiers**, if-then rules encoded as strings over the trinary alphabet $\{0, 1, \#\}$
+ whose condition part matches states and whose action part prescribes a move, with $\#$ a
+ wildcard ("I don't care") that lets general rules coexist with specific ones.
+* **A decoder** that, given the current state, finds which classifiers' conditions match.
+* **An auction** that selects one matching classifier to act: the strongest, or one chosen with
+ probability proportional to strength.
+* **An accounting system** that updates each classifier's **strength** — a running average of the
+ net rewards its decisions earn — and, in sequential problems, passes reward *backward* from
+ rules that collect payoffs to the rules that set them up.
+* **Genetic operators** that create new classifiers, generalize (add $\#$'s) and specialize (remove
+ them), so the system's very vocabulary evolves.
+
+### A two-armed bandit
+
+The simplest classifier system, due to Brian Arthur and Carl Simon, plays a **two-armed
+bandit**.
+
+Arm $i$ pays a random reward with mean $\mu_i$, and $\mu_1 > \mu_2$, but the player knows
+neither distribution.
+
+The classifier system holds two rules — "pull arm 1" and "pull arm 2" — whose conditions are
+always met.
+
+Each rule's strength is the running average of the payoff that arm has delivered, and the arm
+to pull is chosen with probability proportional to strength.
+
+The two rules start with **equal** strengths: the system is told nothing about either arm and
+must build its estimates out of the rewards it experiences.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "The classifier bandit converges to probability matching"
+ name: fig-gc-bandit
+---
+def two_armed_bandit(μ, σ=0.5, T=20_000, seed=0):
+ "Strength = running average of an arm's payoff; pull ∝ strength."
+ rng = np.random.default_rng(seed)
+ S = np.ones(2) # equal strengths: no prior knowledge
+ τ = np.array([1, 1]) # pull counters
+ pulls = np.empty(T, int)
+ for t in range(T):
+ w = np.maximum(S, 1e-9) # strengths can dip below zero early on
+ i = 0 if rng.random() < w[0] / w.sum() else 1
+ reward = μ[i] + σ * rng.standard_normal()
+ τ[i] += 1
+ S[i] += (reward - S[i]) / τ[i] # running average
+ pulls[t] = i
+ return pulls
+
+pulls = two_armed_bandit([3.0, 1.0])
+frac_best = np.cumsum(pulls == 0) / np.arange(1, len(pulls) + 1)
+
+fig, ax = plt.subplots(figsize=(7.5, 4))
+ax.plot(frac_best, lw=1)
+ax.axhline(0.75, color='k', ls='--', label="probability match $\\mu_1/(\\mu_1+\\mu_2)$")
+ax.axhline(1.0, color='C3', ls=':', label="optimal (always best arm)")
+ax.set_xlabel("$t$")
+ax.set_ylabel("fraction of pulls on the best arm")
+ax.set_ylim(0.5, 1.05)
+ax.legend(frameon=False)
+plt.show()
+```
+
+The fraction of pulls on the better arm converges, but **not to one**.
+
+It converges to $\mu_1/(\mu_1 + \mu_2)$, the arm's share of total expected reward.
+
+```{code-cell} ipython3
+for μ in ([1.0, 0.5], [2.0, 1.0], [3.0, 1.0]):
+ frac = np.mean(two_armed_bandit(μ)[-5000:] == 0)
+ print(f"μ = {μ}: fraction on best arm = {frac:.3f}, "
+ f"probability match = {μ[0]/(μ[0]+μ[1]):.3f} (optimal = 1.0)")
+```
+
+Arthur and Simon proved this: the strength-proportional classifier **probability-matches**.
+
+It plays the arms in proportion to their expected rewards rather than concentrating on the best
+one, so it leaves reward on the table forever.
+
+That is not a bug to paper over; it is a lesson about accounting.
+
+A classifier system is only as good as the scheme by which strength is assigned and passed
+around.
+
+The naive rule here matches probabilities; better rules do better.
+
+And in a *sequential* problem — where a rule's payoff comes only much later, through a chain of
+intermediate decisions — the accounting must do something harder still: reward a rule for
+merely *setting up* a profitable future decision.
+
+Holland's device for this is the **bucket brigade**: each acting classifier pays part of its
+strength to the classifier that acted just before it, the one that moved the system into the
+state where the current rule could act.
+
+Reward paid at the end of a chain seeps backward, rule by rule, until early setup rules that
+never touch a payoff directly acquire strength for enabling it.
+
+Designing that backward flow is the central craft of building a classifier system, and it is
+exactly what {doc}`marimon_mcgrattan_sargent` must get right to make agents learn to accept
+money today for the sake of a trade tomorrow.
+
+## Evolutionary programming
+
+We have surveyed four brains.
+
+The last idea is about what to *do* with them.
+
+A recurring finding of this whole section is that systems of adaptive agents, however plodding,
+tend to converge on rational expectations equilibria.
+
+{doc}`olg_adaptive_money` and {doc}`learning_approximation` showed least squares learners
+finding one; {doc}`exchange_rate_learning` showed Newton learners settling on one (of many).
+
+**Evolutionary programming** turns that tendency into a tool.
+
+If a population of adaptive agents reliably converges to an equilibrium, we can run the
+population as a *method for computing* the equilibrium, especially in models too complicated to
+solve by hand.
+
+Sargent is careful about what is and isn't going on:
+
+> The adaptive agents are 'teaching' the economist in the same sense that any numerical algorithm
+> for solving nonlinear equations 'teaches' a mathematician. When these agents can 'teach' us
+> something, it is because we designed them to do so.
+
+This is the same duality that ran through {doc}`learning_approximation`: a learning economy is
+a decentralized equilibrium computation, and an equilibrium computation is a centralized
+learning algorithm.
+
+The genetic algorithm and the classifier system are simply richer computational engines than
+recursive least squares, able to search rugged landscapes and to discover the structure of a
+good rule, not just tune the coefficients of a fixed one.
+
+The showcase is {cite:t}`KiyotakiWright1989`'s search-theoretic model of money, in which the
+medium of exchange is not assumed but must **emerge** from how agents choose to trade.
+
+The equilibrium is a set of trading strategies and matching probabilities, and for enriched
+versions of the model it is hard to characterize analytically.
+
+{cite:t}`MarimonMcGrattanSargent1990` put populations of Holland classifier systems into that environment and
+watched them converge to an equilibrium, then built a five-good version, with no known analytical
+solution, and let the classifier systems suggest what the equilibrium looked like.
+
+## Concluding remarks
+
+The four brains in this catalogue differ in what they take as given.
+
+The perceptron is handed a functional form and asked only for its coefficients, which is why an
+econometrician recognizes it immediately.
+
+The Hopfield network is handed the patterns themselves and asked only to recall them.
+
+The genetic algorithm is handed nothing but a fitness function and must search a space in which
+it cannot compute a gradient.
+
+The classifier system is handed a vocabulary in which rules can be written and must discover
+which rules are worth holding.
+
+Moving down that list, we hand the agent less and ask it to find more, which is precisely the
+direction the bounded-rationality program pushes us.
+
+Two cautions emerged along the way, and both return in the next lecture.
+
+The genetic algorithm's individuals do not learn: only the population does, which makes it a
+model of a society rather than of a mind.
+
+And the bandit showed that a classifier system's performance is decided by its *accounting* —
+by how strength is assigned, bid, and passed along — rather than by the fact of having
+classifiers at all.
+
+{doc}`marimon_mcgrattan_sargent` assembles this machinery into agents who learn, from scratch, to
+use money, and both cautions bear directly on what those agents turn out to be able to learn.
+
+## Exercises
+
+```{exercise-start}
+:label: gc_ex1
+```
+
+The Hopfield network's recall is descent on the energy $E(s) = -\tfrac{1}{2}s^\top ws$.
+
+Verify the descent directly.
+
+Take a stored letter, corrupt several pixels, and record the energy at each step of the recall
+dynamics.
+
+Confirm that energy never increases and that recall halts at (or below) the stored pattern's
+energy.
+
+Then corrupt patterns more heavily and classify the failures: does recall land on a *different*
+stored letter, or on a *spurious* state that was never stored?
+
+Compare the energies in each case, and use them to explain why the error happens.
+
+```{exercise-end}
+```
+
+```{solution-start} gc_ex1
+:class: dropdown
+```
+
+```{code-cell} ipython3
+def recall_with_energy(w, s0, max_iter=30):
+ s = s0.copy()
+ trace = [energy(w, s)]
+ for _ in range(max_iter):
+ s_new = np.sign(w @ s)
+ s_new[s_new == 0] = 1
+ trace.append(energy(w, s_new))
+ if np.array_equal(s_new, s):
+ break
+ s = s_new
+ return s, trace
+
+rng = np.random.default_rng(1)
+i = letters.index('E')
+corrupt = σ[i].copy()
+corrupt[rng.choice(25, 5, replace=False)] *= -1
+final, trace = recall_with_energy(w_hop, corrupt)
+
+print(f"energy along the recall path: {[round(e, 2) for e in trace]}")
+print(f"monotonically non-increasing: {all(np.diff(trace) <= 1e-9)}")
+print(f"recovered the intended letter 'E': {np.array_equal(final, σ[i])}")
+```
+
+The energy falls monotonically along every recall path: the dynamics can only move downhill, which
+is why the network always halts at a local minimum.
+
+Now corrupt more heavily and sort the outcomes into three kinds: the intended letter, a *different*
+stored letter, and a spurious state that is a fixed point but was never taught.
+
+```{code-cell} ipython3
+def classify_outcome(final, i):
+ if np.array_equal(final, σ[i]):
+ return "intended"
+ if any(np.array_equal(final, σ[k]) for k in range(len(letters))):
+ return "wrong letter"
+ return "spurious"
+
+counts = {"intended": 0, "wrong letter": 0, "spurious": 0}
+wrong_energy, spurious_energy = [], []
+for i in range(len(letters)):
+ for seed in range(200):
+ r = np.random.default_rng(1000*i + seed)
+ corrupt = σ[i].copy()
+ corrupt[r.choice(25, 7, replace=False)] *= -1
+ final = recall(w_hop, corrupt)
+ kind = classify_outcome(final, i)
+ counts[kind] += 1
+ if kind == "wrong letter":
+ wrong_energy.append(energy(w_hop, final))
+ elif kind == "spurious":
+ spurious_energy.append(energy(w_hop, final))
+
+print(f"7-pixel corruptions ({sum(counts.values())} trials): {counts}")
+print(f"stored patterns all have energy {energy(w_hop, σ[0]):.1f}")
+print(f"wrong-letter results: energy in [{min(wrong_energy):.1f}, {max(wrong_energy):.1f}]")
+print(f"spurious results: energy in [{min(spurious_energy):.1f}, {max(spurious_energy):.1f}]")
+```
+
+Both failure modes occur, and at this level of corruption the spurious states are the more
+common of the two by some margin.
+
+A **wrong-letter** result sits at exactly the stored energy $-N/2$: the corruption pushed the
+starting point clear across a basin boundary, into the domain of attraction of a *different*
+stored memory that is just as deep.
+
+Energy descent converges to *a* minimum but cannot guarantee the *nearest* one when correlated
+patterns have interlocking basins.
+
+A **spurious** result is a local minimum the designer never intended, typically a blend of
+stored patterns, created as an artifact of the storage rule.
+
+Its energy ranges from shallower than the stored patterns' all the way down to exactly their
+depth, so depth alone does not identify a memory as one we asked for.
+
+Some of the deepest spurious states are sign reversals: because $\operatorname{sgn}(w(-s)) = -\operatorname{sgn}(ws)$,
+the vector $-\sigma$ is a fixed point of the same energy whenever $\sigma$ is, and the network
+stores each letter's photographic negative whether we wanted it or not.
+
+Because all of these are genuine local minima, downhill dynamics cannot escape any of them.
+
+Both are why the network is imperfect, and both are why **simulated annealing** matters: adding
+declining random shaking lets the system hop out of a shallow spurious basin, or across a basin
+boundary, before settling, something pure energy descent can never do.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: gc_ex2
+```
+
+The genetic algorithm's exploration comes from two operators, crossover and mutation.
+
+The book argues that crossover "lies at the heart of the algorithm," while mutation alone "is a
+poor mechanism for injecting diversity."
+
+Test the claim on Axelrod's game.
+
+Write a mutation-only variant (each child is a mutated copy of one selected parent, with no
+crossover) and compare the fitness it reaches against the full algorithm.
+
+The book is careful to say *when* mutation alone is weak: "when the mutation rate is set to a
+very low value, mutation alone is a poor mechanism for injecting diversity."
+
+So run the comparison at a **low** mutation rate, where crossover has to carry the exploration.
+
+```{exercise-end}
+```
+
+```{solution-start} gc_ex2
+:class: dropdown
+```
+
+```{code-cell} ipython3
+def evolve_no_crossover(N=60, generations=60, p_mut=0.005, seed=0):
+ "Mutation-only variant: each child is a mutated copy of one selected parent."
+ rng = np.random.default_rng(seed)
+ pop = rng.integers(0, 2, (N, 70))
+ best_hist = []
+ for _ in range(generations):
+ fit = np.array([fitness(pop[i]) for i in range(N)])
+ best_hist.append(fit.max()) # recorded exactly as in `evolve`
+ weight = fit - fit.min() + 1e-6
+ parents = np.searchsorted(np.cumsum(weight), rng.random(N) * weight.sum())
+ nxt = pop[parents].copy()
+ nxt[rng.random((N, 70)) < p_mut] ^= 1
+ pop = nxt
+ return best_hist[-1]
+
+rows = []
+for seed in range(5):
+ panel_rng = np.random.default_rng(seed)
+ with_x = evolve(N=60, generations=60, p_mut=0.005, seed=seed)[1][-1]
+ panel_rng = np.random.default_rng(seed)
+ without_x = evolve_no_crossover(seed=seed)
+ rows.append((seed, with_x, without_x))
+
+print(f"{'seed':>4} {'with crossover':>16} {'mutation only':>16}")
+for sd, a, b in rows:
+ print(f"{sd:>4} {a:>16.3f} {b:>16.3f}")
+print(f"{'mean':>4} {np.mean([r[1] for r in rows]):>16.3f} "
+ f"{np.mean([r[2] for r in rows]):>16.3f}")
+```
+
+At a low mutation rate crossover clearly wins: it reaches a higher fitness than the mutation-only
+variant on almost every seed.
+
+The reason is the one the book gives.
+
+With little mutation, a mutation-only population can only inch forward one rare bit-flip at a
+time, and quickly loses diversity as selection copies its few best strings.
+
+Crossover instead recombines whole *segments* that have already proved useful in different
+strings: a good way of handling one opponent spliced onto a good way of handling another.
+
+It injects large, structured variation while preserving the schemata that fitness has already
+favored, which is why Holland put it at the center of the algorithm.
+
+```{note}
+The margin narrows, and can even vanish, at higher mutation rates: when mutation is generating
+plenty of diversity on its own, crossover's contribution is less pivotal.
+
+Try re-running the comparison at `p_mut=0.02` to see the effect shrink.
+
+Whether crossover is decisive depends on the mutation rate *and* on whether the encoding lines
+up useful building blocks with contiguous bit segments, which, for this history-indexed
+strategy table, it only partly does.
+```
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: gc_ex3
+```
+
+The two-armed bandit classifier probability-matches, which is suboptimal: it keeps pulling the worse
+arm a fixed fraction of the time forever.
+
+A natural fix is to make the auction **greedier** as the system gains confidence.
+
+Replace the strength-proportional choice rule with a softmax,
+
+$$
+\pi_1 = \frac{e^{\beta S_1}}{e^{\beta S_1} + e^{\beta S_2}},
+$$
+
+where $\beta$ controls greediness ($\beta \to \infty$ always picks the stronger arm).
+
+Implement it and show how the long-run fraction on the best arm depends on $\beta$.
+
+How large must $\beta$ be to move the classifier from probability matching toward the optimal
+policy?
+
+```{exercise-end}
+```
+
+```{solution-start} gc_ex3
+:class: dropdown
+```
+
+```{code-cell} ipython3
+def bandit_softmax(μ, β, σ=0.5, T=20_000, seed=0):
+ rng = np.random.default_rng(seed)
+ S = np.ones(2) # equal strengths, as before
+ τ = np.array([1, 1])
+ pulls = np.empty(T, int)
+ for t in range(T):
+ p1 = 1 / (1 + np.exp(-β * (S[0] - S[1])))
+ i = 0 if rng.random() < p1 else 1
+ reward = μ[i] + σ * rng.standard_normal()
+ τ[i] += 1
+ S[i] += (reward - S[i]) / τ[i]
+ pulls[t] = i
+ return np.mean(pulls[-5000:] == 0)
+
+μ = [3.0, 1.0]
+print(f"probability match target = {μ[0]/(μ[0]+μ[1]):.3f}, optimal = 1.000\n")
+for β in (0.5, 1.0, 2.0, 4.0, 8.0):
+ frac = bandit_softmax(μ, β)
+ print(f"β = {β:>4}: fraction on best arm = {frac:.3f}")
+```
+
+As $\beta$ rises the classifier abandons probability matching and concentrates on the better arm,
+approaching the optimal policy of always pulling it.
+
+The exercise makes concrete why the *accounting and auction rules* — not just the fact of
+having classifiers — determine how well a classifier system performs.
+
+The strength-proportional rule of Arthur and Simon is one point on a spectrum; a greedy rule
+sits at the other end.
+
+Real classifier systems, including the one in {doc}`marimon_mcgrattan_sargent`, choose their
+auction and strength-update rules deliberately, precisely because the choice is what separates
+a system that merely matches probabilities from one that learns to act well.
+
+```{solution-end}
+```
diff --git a/lectures/learning_approximation.md b/lectures/learning_approximation.md
new file mode 100644
index 000000000..f48684481
--- /dev/null
+++ b/lectures/learning_approximation.md
@@ -0,0 +1,850 @@
+---
+jupytext:
+ text_representation:
+ extension: .md
+ format_name: myst
+ format_version: 0.13
+ jupytext_version: 1.17.1
+kernelspec:
+ display_name: Python 3 (ipykernel)
+ language: python
+ name: python3
+---
+
+(learning_approximation)=
+```{raw} jupyter
+
+```
+
+# Learning, Approximation, and Equilibrium Computation
+
+```{index} single: Bounded Rationality; Parameterized Expectations
+```
+
+```{contents} Contents
+:depth: 2
+```
+
+## Overview
+
+In {doc}`olg_adaptive_money` a system of adaptive agents found a rational expectations
+equilibrium it was never told about.
+
+The agents there learned a *separate* saving rate for each level of the government deficit —
+a **non-parametric** rule, one number per state.
+
+That works when there are two states.
+
+It runs into two walls otherwise, both noted in {cite:t}`Sargent1993`:
+
+* a state that occurs rarely is learned about slowly, because its observations arrive slowly;
+* when the number of states is large — and especially when the state is *continuous* — one
+ parameter per state is hopeless.
+
+The econometrician's response is to impose a **parametric** form on the decision rule,
+$s = f(G, \theta)$, with $\theta$ of low dimension, and use every observation to estimate
+$\theta$.
+
+This lecture works through what that buys and what it costs.
+
+The costs and benefits are two sides of one idea.
+
+A parametric family can be learned quickly because every observation informs every state.
+
+But a learning scheme confined to a family of functions can converge to a rational expectations
+equilibrium *only if some member of the family supports one*.
+
+Otherwise the best it can reach is an **approximate equilibrium**.
+
+Working this out surfaces a theme that runs through the whole bounded-rationality program:
+
+> Learning algorithms and equilibrium computation algorithms look like each other.
+
+We will make that concrete.
+
+**Marcet's method of parameterized expectations** — a standard tool for *computing* rational
+expectations equilibria — turns out, when written recursively, to be exactly a model of
+adaptive agents learning.
+
+An equilibrium computation is a centralized learning algorithm run by the modeller; a learning
+economy is a decentralized equilibrium computation run by the agents.
+
+The plan:
+
+1. Put a *continuous* deficit into the overlapping generations model of the previous lecture,
+ turning the equilibrium condition into a **functional equation** in the saving rule $f(G)$.
+1. Compute a benchmark solution, so we have ground truth to measure against.
+1. Solve it by **parameterized expectations** with families of increasing richness, and watch
+ approximate equilibria give way to the exact one.
+1. Show that the recursive, real-time version of the same algorithm **is** a learning economy.
+1. Learn the saving rule **non-parametrically** with a kernel estimator, and weigh the
+ trade-off against the parametric route.
+
+Let's start with some imports.
+
+```{code-cell} ipython3
+import numpy as np
+import pandas as pd
+import quantecon as qe
+import matplotlib.pyplot as plt
+from scipy.optimize import brentq
+```
+
+## A continuous-state functional equation
+
+We keep the overlapping generations monetary economy of {doc}`olg_adaptive_money` — two-period
+lived agents, log utility, endowments $w_1$ when young and $w_2$ when old, currency the only
+store of value, and a government financing a deficit by printing money.
+
+The one change: the deficit $G_t$ now follows a **continuous** Markov process, with transition
+kernel $F(G', G) = \operatorname{Prob}\{G_{t+1} \leq G' \mid G_t = G\}$.
+
+We look for a stationary equilibrium in which saving is a *function* of the current deficit,
+$s_t = f(G_t)$.
+
+The household's first-order condition, evaluated at $s_t = f(G_t)$, is
+
+$$
+u'(w_1 - f(G_t)) = \mathbb{E}_t\bigl[u'(w_2 + f(G_t) R_t)\, R_t\bigr],
+\qquad \mathbb{E}_t(\cdot) = \mathbb{E}(\cdot \mid G_t),
+$$
+
+and the government budget constraint with market clearing gives the return on currency in
+terms of the saving rule, exactly as before (with $N = 1$),
+
+$$
+R_t = \frac{f(G_{t+1}) - G_{t+1}}{f(G_t)} .
+$$
+
+Substituting $R_t$, and using $f(G_t) R_t = f(G_{t+1}) - G_{t+1}$ for old-age consumption,
+turns the first-order condition into a **functional equation** in $f$:
+
+```{math}
+:label: functional_equation
+
+u'\bigl(w_1 - f(G_t)\bigr)
+= \mathbb{E}_t\!\left[
+u'\!\bigl(w_2 + f(G_{t+1}) - G_{t+1}\bigr)
+\cdot \frac{f(G_{t+1}) - G_{t+1}}{f(G_t)}
+\right].
+```
+
+Under log utility, $u'(c) = 1/c$, and because $f(G_t)$ is known at $t$ it pulls out of the
+expectation.
+
+Writing
+
+```{math}
+:label: psi_definition
+
+\psi(G_t) \equiv \mathbb{E}_t\bigl[u'(w_2 + f(G_{t+1}) - G_{t+1})\,(f(G_{t+1}) - G_{t+1})\bigr],
+```
+
+equation {eq}`functional_equation` becomes
+
+$$
+\frac{1}{w_1 - f(G_t)} = \frac{\psi(G_t)}{f(G_t)},
+\qquad\text{i.e.}\qquad
+f(G_t) = \frac{w_1\,\psi(G_t)}{1 + \psi(G_t)} .
+$$
+
+So the whole problem reduces to finding the **conditional expectation** $\psi(G)$: once we
+have it, the saving rule follows in closed form.
+
+That observation is the hinge of the entire lecture.
+
+Every method below — parametric or not — is a way of estimating the one object $\psi(G)$.
+
+```{code-cell} ipython3
+w1, w2 = 20.0, 10.0
+
+def s_from_psi(ψ):
+ "Recover the saving rule from the conditional expectation ψ."
+ return w1 * ψ / (1 + ψ)
+```
+
+### The deficit process
+
+We take $\log G_t$ to follow an AR(1),
+$\log G_{t+1} = (1-\rho)\log \bar G + \rho \log G_t + \sigma \varepsilon_{t+1}$, and discretize
+it with the method of {cite:t}`Tauchen1986` to get a fine grid for the benchmark.
+
+The parameters keep the deficit modest enough that a monetary equilibrium exists at every
+state.
+
+```{code-cell} ipython3
+ρ, σ, G_bar = 0.7, 0.30, 0.6
+
+mc = qe.tauchen(21, ρ, σ, mu=np.log(G_bar) * (1 - ρ), n_std=2.5)
+G, P = np.exp(mc.state_values), mc.P
+ergodic = mc.stationary_distributions[0] # where the deficit spends its time
+
+print(f"deficit grid: {len(G)} points from {G.min():.3f} to {G.max():.3f}")
+print(f"mean deficit under the ergodic distribution: {ergodic @ G:.3f}")
+```
+
+## A benchmark solution
+
+To measure any learning scheme, we first need the true $f(G)$.
+
+On the discrete grid, {eq}`functional_equation` is a system of equations that we solve by
+**iterating the perceived-to-actual map**, precisely the $T$ map of {doc}`bounded_rationality`.
+
+Guess a saving rule; use it on the right-hand side to compute the actual saving each state
+would call forth; repeat until the two agree.
+
+Because $\psi$ is monotone we can invert the first-order condition state by state with a
+bracketing root-finder, which makes the iteration robust.
+
+```{code-cell} ipython3
+def solve_benchmark(G, P, damp=0.5, tol=1e-12, max_iter=5000):
+ "Fixed point of the perceived-to-actual saving map on the deficit grid."
+ n = len(G)
+ s = np.full(n, (w1 - w2) / 2)
+ for it in range(max_iter):
+ s_new = np.empty(n)
+ for i in range(n):
+ # E_t[ u'(c2) (s' - G') ] under the perceived rule s
+ expect = sum((s[j] - G[j]) / (w2 + (s[j] - G[j])) * P[i, j] for j in range(n))
+ # solve 1/(w1 - s_i) = expect / s_i for s_i in (G_i, w1)
+ foc = lambda si: 1/(w1 - si) - expect/si
+ s_new[i] = brentq(foc, G[i] + 1e-9, w1 - 1e-9)
+ if np.max(np.abs(s_new - s)) < tol:
+ return s, it
+ s = damp * s_new + (1 - damp) * s
+ return s, it
+
+
+s_benchmark, iters = solve_benchmark(G, P)
+print(f"converged in {iters} iterations")
+
+
+def foc_residual(s_rule):
+ "Max |first-order-condition error| of a saving rule under the true kernel P."
+ n = len(G)
+ err = []
+ for i in range(n):
+ rhs = sum((s_rule[j] - G[j]) / (w2 + (s_rule[j] - G[j])) / s_rule[i] * P[i, j]
+ for j in range(n))
+ err.append(1/(w1 - s_rule[i]) - rhs)
+ return np.max(np.abs(err))
+
+
+print(f"benchmark FOC residual: {foc_residual(s_benchmark):.2e}")
+print(f"saving rule f(G) runs from {s_benchmark.max():.3f} (low deficit) "
+ f"to {s_benchmark.min():.3f} (high deficit)")
+```
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Benchmark equilibrium saving rule"
+ name: fig-la-benchmark
+---
+fig, ax = plt.subplots(figsize=(6.5, 4))
+ax.plot(G, s_benchmark, 'k-', lw=2)
+ax.set_xlabel("deficit $G$")
+ax.set_ylabel("saving $f(G)$")
+plt.show()
+```
+
+Saving falls as the deficit rises: a larger deficit means faster money creation, a poorer
+return on currency, and less saving.
+
+The rule is gently curved, which will matter in a moment.
+
+This $f(G)$ is the object no adaptive agent gets to see.
+
+Everything below tries to recover it.
+
+## Parameterized expectations
+
+Marcet's **method of parameterized expectations** attacks {eq}`functional_equation` by putting
+a parametric form on the conditional expectation $\psi(G)$ from {eq}`psi_definition`,
+
+$$
+\psi(G) \approx \psi(G, \theta) = \phi(G)^\top \theta ,
+$$
+
+where $\phi(G)$ is a vector of basis functions and $\theta$ a short vector of coefficients.
+
+The algorithm is a fixed-point iteration on $\theta$:
+
+1. Given $\theta$, the saving rule is $f(G) = w_1 \psi(G,\theta) / (1 + \psi(G,\theta))$.
+1. Simulate a long path of the deficit and compute the implied saving each period.
+1. Form the *realized* value of the object inside {eq}`psi_definition`,
+ $y_t = u'(w_2 + s_{t+1} - G_{t+1})(s_{t+1} - G_{t+1})$, and **regress** it on $\phi(G_t)$ to
+ get a new $\theta$.
+1. Iterate to convergence.
+
+The regression in step 3 is doing the work of the conditional expectation: least squares
+projects the realized $y_t$ onto functions of the current state, which is exactly what
+$\mathbb{E}_t[\cdot \mid G_t]$ is.
+
+We use monomials in a rescaled $\log G$ as the basis, and vary the degree.
+
+```{code-cell} ipython3
+G_min, G_max = G.min(), G.max()
+
+def simulate_deficit(T, seed=0):
+ "Simulate the AR(1) deficit, clipped to the benchmark support."
+ rng = np.random.default_rng(seed)
+ μ = np.log(G_bar) * (1 - ρ)
+ lg = np.empty(T)
+ lg[0] = np.log(G_bar)
+ for t in range(1, T):
+ lg[t] = μ + ρ * lg[t-1] + σ * rng.standard_normal()
+ return np.clip(np.exp(lg), G_min, G_max)
+
+def basis(G_vals, degree):
+ "Monomials in log-deficit rescaled to [-1, 1]."
+ x = 2 * (np.log(G_vals) - np.log(G_min)) / (np.log(G_max) - np.log(G_min)) - 1
+ return np.vstack([x**k for k in range(degree + 1)]).T
+
+
+def parameterized_expectations(degree, T=40_000, n_iter=200, seed=0):
+ "Batch parameterized expectations; returns the implied saving rule on the grid."
+ G_sim = simulate_deficit(T + 1, seed)
+ Φ = basis(G_sim, degree)
+ θ = np.zeros(degree + 1)
+ θ[0] = 2.0
+ for _ in range(n_iter):
+ ψ = np.clip(Φ @ θ, 0.05, 50)
+ s = np.clip(s_from_psi(ψ), G_sim + 0.05, w1 - 0.05)
+ y = (s[1:] - G_sim[1:]) / (w2 + (s[1:] - G_sim[1:])) # realized regressand
+ θ_new = np.linalg.lstsq(Φ[:-1], y, rcond=None)[0]
+ if np.max(np.abs(θ_new - θ)) < 1e-11:
+ θ = θ_new
+ break
+ θ = 0.4 * θ + 0.6 * θ_new
+ ψ_grid = np.clip(basis(G, degree) @ θ, 0.05, 50)
+ return s_from_psi(ψ_grid)
+```
+
+### Approximate versus exact equilibria
+
+Now run it for families of increasing richness: a constant $\psi$ (saving independent of the
+deficit), a linear one, and a quadratic one.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Parameterized expectations rules of increasing degree"
+ name: fig-la-pea
+---
+rules = {deg: parameterized_expectations(deg) for deg in (0, 1, 2)}
+
+fig, ax = plt.subplots(figsize=(7, 4.5))
+ax.plot(G, s_benchmark, 'k-', lw=2.2, label="benchmark $f(G)$")
+labels = {0: "degree 0 (constant)", 1: "degree 1 (linear)", 2: "degree 2 (quadratic)"}
+for deg, colour in [(0, 'C0'), (1, 'C1'), (2, 'C2')]:
+ ax.plot(G, rules[deg], '--', color=colour, lw=1.4, label=labels[deg])
+ax.set_xlabel("deficit $G$")
+ax.set_ylabel("saving $s$")
+ax.legend(frameon=False)
+plt.show()
+```
+
+The **constant** family cannot represent a saving rule that depends on the deficit at all, so
+the best it can do is a flat line through the middle of the data.
+
+That flat line *is* an approximate equilibrium — a fixed point of the learning scheme — but it
+satisfies the first-order condition only on average, not state by state.
+
+The **linear** family does much better but bends the wrong way at the extremes.
+
+The **quadratic** family is visually indistinguishable from the benchmark: because the true
+$f(G)$ is very nearly quadratic here, a three-parameter rule essentially nails it.
+
+We can quantify "how close" two ways: by the distance to the benchmark, and by the residual
+in the first-order condition under the true kernel, the honest measure of how far an
+approximate equilibrium is from an exact one.
+
+```{code-cell} ipython3
+rows = []
+for deg in (0, 1, 2):
+ s_hat = rules[deg]
+ sup = np.max(np.abs(s_hat - s_benchmark))
+ rms = np.sqrt(ergodic @ (s_hat - s_benchmark)**2)
+ rows.append([labels[deg], sup, rms, foc_residual(s_hat)])
+
+pd.DataFrame(rows, columns=["family", "sup $|s - f|$",
+ "ergodic RMS", "FOC residual"]).set_index("family").round(4)
+```
+
+Every column falls as the family grows richer.
+
+The first-order-condition residual — the quantity that is exactly zero at a rational
+expectations equilibrium and positive at an approximate one — drops by more than an order of
+magnitude from the constant family to the quadratic.
+
+```{note}
+Richer is not *always* better with finite data.
+
+Push the degree higher and the extra terms start chasing sampling noise in the tails of the
+deficit distribution, where observations are scarce, and the sup-norm error can tick back up
+even as the fit improves where agents actually spend their time.
+
+That is the same phenomenon as the slowly-learned rare state in {doc}`olg_adaptive_money`, seen
+from the approximation side: a scheme fits well where the data is, and the cost of misfitting a
+rarely-visited region is small in terms of expected utility.
+
+{ref}`lae_ex1` explores this.
+```
+
+## Learning is equilibrium computation
+
+So far parameterized expectations is an algorithm *we* run to compute an equilibrium.
+
+Now comes the point of the whole lecture.
+
+Write the same algorithm **recursively** — updating $\theta$ once per period as a single new
+observation arrives, rather than re-regressing a whole simulated panel — and it becomes a model
+of *adaptive agents learning in real time*.
+
+The recursive form is ordinary recursive least squares:
+
+```{math}
+:label: recursive_pea
+
+\begin{aligned}
+R_{t+1} &= R_t + \tfrac{1}{t}\bigl(\phi(G_t)\phi(G_t)^\top - R_t\bigr), \\
+\theta_{t+1} &= \theta_t + \tfrac{1}{t} R_{t+1}^{-1} \phi(G_t)\bigl(y_t - \phi(G_t)^\top \theta_t\bigr),
+\end{aligned}
+```
+
+where $y_t$ is the same realized regressand as before and $R_t$ tracks the second moment of
+the regressors.
+
+This is the identical object we met in {doc}`olg_adaptive_money`: a stochastic-approximation
+recursion with a $1/t$ gain.
+
+The only difference from the state-by-state learning there is that $\theta$ indexes a
+*parametric* rule, so a single observation updates the saving rule everywhere at once.
+
+```{code-cell} ipython3
+def online_pea(degree=2, T=500_000, seed=0, ridge=1e-3):
+ """
+ Recursive-least-squares parameterized expectations: the learning economy.
+
+ Young agents at t save according to the current θ; next period the Euler
+ equation's realized value arrives and θ is updated once. Returns the θ
+ path and the final implied saving rule.
+ """
+ G_sim = simulate_deficit(T, seed)
+ θ = np.array([2.0] + [0.0] * degree)
+ R = ridge * np.eye(degree + 1)
+
+ def saving(g, θ):
+ ψ = np.clip(basis(np.array([g]), degree)[0] @ θ, 0.05, 50)
+ return np.clip(s_from_psi(ψ), g + 0.05, w1 - 0.05)
+
+ x_prev = None
+ θ_path = np.empty((T, degree + 1))
+ for t in range(T):
+ s_t = saving(G_sim[t], θ)
+ if x_prev is not None: # date t-1 Euler realizes now
+ y = (s_t - G_sim[t]) / (w2 + (s_t - G_sim[t]))
+ gain = 1.0 / (t + 1)
+ R += gain * (np.outer(x_prev, x_prev) - R)
+ θ = θ + gain * np.linalg.solve(R, x_prev * (y - x_prev @ θ))
+ x_prev = basis(np.array([G_sim[t]]), degree)[0]
+ θ_path[t] = θ
+
+ s_grid = np.array([saving(g, θ) for g in G])
+ return θ_path, s_grid
+
+
+θ_path, s_online = online_pea()
+print(f"online PEA: sup error = {np.max(np.abs(s_online - s_benchmark)):.4f}, "
+ f"ergodic RMS = {np.sqrt(ergodic @ (s_online - s_benchmark)**2):.4f}")
+```
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Online learning of the saving rule"
+ name: fig-la-online
+---
+fig, axes = plt.subplots(1, 2, figsize=(11, 4.2))
+
+axes[0].plot(θ_path[:, 0], label=r"$\theta_0$", lw=2)
+axes[0].plot(θ_path[:, 1], label=r"$\theta_1$", lw=2)
+axes[0].plot(θ_path[:, 2], label=r"$\theta_2$", lw=2)
+axes[0].set_xscale('log')
+axes[0].set_xlabel("$t$ (log scale)")
+axes[0].set_ylabel(r"$\theta_t$")
+axes[0].set_title("coefficients learned in real time")
+axes[0].legend(frameon=False, fontsize=9)
+
+axes[1].plot(G, s_benchmark, 'k-', lw=2.2, label="benchmark $f(G)$")
+axes[1].plot(G, s_online, 'C3--', lw=1.5, label="online PEA")
+axes[1].set_xlabel("deficit $G$")
+axes[1].set_ylabel("saving $s$")
+axes[1].set_title("saving rule after learning")
+axes[1].legend(frameon=False)
+plt.tight_layout()
+plt.show()
+```
+
+The coefficients settle down, and the saving rule they imply tracks the benchmark closely.
+
+The small remaining gap is the signature of a scheme that has not quite finished converging.
+
+Convergence is *slow*: the $1/t$ gain guarantees it but does not hurry it, and the online
+scheme takes hundreds of thousands of periods to reach an accuracy the batch algorithm attains
+in a handful of passes.
+
+That gap is exactly the difference between an agent constrained to learn from experience as it
+arrives and a modeller free to re-use a whole simulated history at once.
+
+But the limit is the same object.
+
+Sargent's summary:
+
+> Learning algorithms and equilibrium computation algorithms look like each other. Equilibrium
+> computation algorithms often have interpretations as centralized learning algorithms whereby
+> the model builder, acting in a role of 'social planner,' gropes for a set of pricing
+> functions for markets and decision rules for agents that will satisfy all of the individual
+> optimum conditions and market-clearing conditions. We have also seen that learning systems
+> with boundedly rational agents sometimes have interpretations as decentralized equilibrium
+> computation algorithms.
+
+This is why Marcet arrived at parameterized expectations as a *computational* method by way of
+earlier work on the *dynamics of least squares learning*.
+
+The two are the same recursion read in two directions.
+
+## Learning without a functional form
+
+The parametric route is fast but stakes everything on the family.
+
+If no member of $\{f(\cdot, \theta)\}$ supports an equilibrium, the scheme converges to an
+approximate equilibrium and stops.
+
+The **non-parametric** alternative imposes no functional form.
+
+Following the recursive kernel estimators of {cite:t}`ChenWhite1998`, an agent estimates the
+conditional expectation $\psi(G)$ directly from a kernel-weighted average of past realized
+values, letting the data choose the shape.
+
+The recursive kernel density estimator is itself a stochastic-approximation recursion,
+
+$$
+\hat F_t(x) = \hat F_{t-1}(x) + \tfrac{1}{t}\!\left[K\!\left(\tfrac{x - x_t}{h_t}\right) - \hat F_{t-1}(x)\right],
+$$
+
+with a bandwidth $h_t \searrow 0$, the same $1/t$-gain shape as everything else in this
+lecture, now updating an entire estimated density rather than a finite parameter vector.
+
+What we implement below is the **batch** counterpart of that recursion: a
+**Nadaraya–Watson** estimate of $\psi(G)$, formed as a kernel-weighted average of the realized
+regressand $y$ over a whole simulated history, with weights that fall off with distance in
+$\log G$, and iterated to a fixed point.
+
+Reading it in batch form keeps the comparison clean, because the parametric scheme we are
+measuring it against is also in batch form.
+
+The recursive display above stands to this smoother exactly as {eq}`recursive_pea` stands to
+the batch algorithm of the previous section: same estimator, read as a real-time learning rule
+rather than as a computation.
+
+```{code-cell} ipython3
+def kernel_smoother(T=400_000, seed=0, h=0.06, n_iter=40, damp=0.5):
+ """
+ Non-parametric parameterized expectations. ψ(G) is estimated by a
+ Nadaraya-Watson smoother of the realized regressand -- no functional form.
+ The rule is iterated to a self-consistent fixed point.
+ """
+ G_sim = simulate_deficit(T, seed)
+ log_sim = np.log(G_sim[:-1])
+ # kernel weights of each grid point against the whole simulated history
+ W = np.array([np.exp(-0.5 * ((np.log(g) - log_sim) / h)**2) for g in G])
+ W /= W.sum(axis=1, keepdims=True)
+
+ s_grid = np.full(len(G), (w1 - w2) / 2)
+ for _ in range(n_iter):
+ s = np.clip(np.interp(np.log(G_sim), np.log(G), s_grid), G_sim + 0.05, w1 - 0.05)
+ y = (s[1:] - G_sim[1:]) / (w2 + (s[1:] - G_sim[1:]))
+ ψ = W @ y
+ s_new = np.clip(s_from_psi(ψ), G + 0.05, w1 - 0.05)
+ if np.max(np.abs(s_new - s_grid)) < 1e-8:
+ s_grid = s_new
+ break
+ s_grid = damp * s_new + (1 - damp) * s_grid
+ return s_grid
+
+
+s_kernel = kernel_smoother()
+print(f"kernel smoother: sup error = {np.max(np.abs(s_kernel - s_benchmark)):.4f}, "
+ f"ergodic RMS = {np.sqrt(ergodic @ (s_kernel - s_benchmark)**2):.4f}")
+```
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Kernel smoother approximation to the saving rule"
+ name: fig-la-kernel
+---
+fig, ax = plt.subplots(figsize=(6.5, 4.2))
+ax.plot(G, s_benchmark, 'k-', lw=2.2, label="benchmark $f(G)$")
+ax.plot(G, s_kernel, 'C3o', ms=4, label="kernel smoother")
+ax.set_xlabel("deficit $G$")
+ax.set_ylabel("saving $s$")
+ax.legend(frameon=False)
+plt.show()
+```
+
+With no functional form imposed, the kernel learner recovers the benchmark to about the same
+accuracy as the quadratic parametric rule.
+
+That the two do equally well here is not a coincidence to lean on.
+
+It is because the true $f(G)$ *happens* to be almost exactly quadratic, so the parametric
+family we chose was very nearly correct.
+
+When it is not, the two part ways, and the trade-off is the one familiar from econometrics:
+
+* the **parametric** scheme learns fast and extrapolates gracefully, but is only as good as
+ its family; a poorly chosen family delivers an approximate equilibrium and no warning;
+* the **non-parametric** scheme cannot be led astray by a bad functional form, but pays for its
+ flexibility with slower learning and with bias where the data is thin, at the edges of the
+ deficit's range, where the kernel has few neighbours to average.
+
+Both are recursions of the same shape.
+
+Which one an adaptive agent should use is, once again, one of the modelling choices that the
+bounded-rationality program forces into the open and that rational expectations quietly made
+for us.
+
+## Concluding remarks
+
+The overlapping generations model gave us a clean laboratory for a general lesson.
+
+An adaptive agent in an environment with a large or continuous state cannot learn a separate
+response for every contingency.
+
+It must **generalize**, imposing some structure that lets a finite amount of experience inform
+behavior across the whole state space.
+
+Whether that structure is a low-order polynomial, a kernel bandwidth, or something else is a
+modelling choice with real consequences: it fixes whether the achievable limit is a rational
+expectations equilibrium or merely an approximation to one.
+
+That same act of generalization is what makes learning and equilibrium computation two faces of
+one object.
+
+Marcet's algorithm computes an equilibrium by parameterizing an expectation and regressing; an
+adaptive agent learns by parameterizing an expectation and regressing; the recursions coincide.
+
+The next lecture, {doc}`marimon_mcgrattan_sargent`, carries the generalization problem to its
+sharpest form.
+
+There the state and action spaces are large enough that enumerating rules is out of the
+question, and agents must **discover** a compact set of good rules from scratch, using John
+Holland's classifier systems and genetic algorithm in place of the least squares and kernel
+machinery of this lecture.
+
+## Exercises
+
+```{exercise-start}
+:label: lae_ex1
+```
+
+The note above claimed that pushing the polynomial degree too high can *hurt* the sup-norm fit,
+because high-order terms chase sampling noise in the sparsely-visited tails of the deficit
+distribution.
+
+Verify it.
+
+Run the batch parameterized-expectations algorithm for degrees $0$ through $5$ and report two
+error measures against the benchmark: the sup norm over the whole grid, and the RMS error
+weighted by the ergodic distribution of the deficit (i.e. weighted by where agents actually
+operate).
+
+Which measure is monotone in the degree, and which is not?
+
+Interpret.
+
+```{exercise-end}
+```
+
+```{solution-start} lae_ex1
+:class: dropdown
+```
+
+```{code-cell} ipython3
+rows = []
+for deg in range(6):
+ s_hat = parameterized_expectations(deg, T=150_000)
+ sup = np.max(np.abs(s_hat - s_benchmark))
+ rms = np.sqrt(ergodic @ (s_hat - s_benchmark)**2)
+ rows.append([deg, sup, rms])
+
+pd.DataFrame(rows, columns=["degree", "sup error (whole grid)",
+ "ergodic RMS (where agents live)"]).set_index("degree").round(4)
+```
+
+The ergodic-weighted error falls monotonically and then flattens: once the family is rich
+enough to capture the curvature of $f$ where the deficit actually spends its time, extra terms
+add nothing.
+
+The sup-norm error falls at first but then rises again.
+
+The high-order polynomials fit the center well but oscillate at the ends of the grid, where the
+ergodic distribution puts almost no mass and so the regression has almost no data to discipline
+them.
+
+This is the approximation-side image of the rare-state problem from {doc}`olg_adaptive_money`.
+
+A learning scheme allocates its accuracy to where the observations are.
+
+Misfitting a rarely-visited region shows up starkly in the sup norm but costs almost nothing in
+expected utility, which is why the ergodic-weighted measure is the economically relevant one.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: lae_ex2
+```
+
+The constant family (degree 0) delivers an *approximate* equilibrium: a single saving rate,
+independent of the deficit, that satisfies the first-order condition only on average.
+
+There is a natural benchmark for it.
+
+Compute the single saving rate $\bar s$ that a degree-0 parameterized-expectations agent
+converges to, and compare it with the saving rate of the economy in which the deficit is fixed
+forever at its mean $\bar G$.
+
+Are they the same?
+
+Should they be?
+
+```{exercise-end}
+```
+
+```{solution-start} lae_ex2
+:class: dropdown
+```
+
+```{code-cell} ipython3
+# the degree-0 approximate equilibrium
+s_const = parameterized_expectations(0)[0] # constant across states
+
+# the deterministic economy with G fixed at its mean. Like the constant-deficit
+# model of the previous lecture it has two steady states; we take the high-saving
+# (low-inflation) one, found by iterating the saving map to its stable fixed point.
+G_mean = ergodic @ G
+
+def deterministic_saving(G_fixed, damp=0.5):
+ s = (w1 - w2) / 2
+ for _ in range(1000):
+ ψ = (s - G_fixed) / (w2 + (s - G_fixed))
+ s = damp * s + (1 - damp) * s_from_psi(ψ)
+ return s
+
+s_fixed = deterministic_saving(G_mean)
+
+print(f"degree-0 approximate-equilibrium saving rate : {s_const:.4f}")
+print(f"mean deficit E[G] : {G_mean:.4f}")
+print(f"saving in the economy with G fixed at E[G] : {s_fixed:.4f}")
+```
+
+They are close but not equal.
+
+The degree-0 agent picks the single saving rate that best fits the first-order condition
+*across the stochastic economy*, an average over the realized distribution of returns.
+
+The fixed-deficit economy replaces that whole distribution with a single degenerate one at the
+mean deficit.
+
+By a Jensen-type argument these need not coincide: the first-order condition is nonlinear in
+the return, so its expectation over a spread-out distribution of deficits differs from its
+value at the mean deficit.
+
+They would agree only if the deficit were deterministic to begin with.
+
+The lesson is that an approximate equilibrium is not a naive "certainty-equivalent" object.
+
+Even the crudest one-parameter learner is responding to the *whole* distribution of outcomes it
+experiences, not to a point forecast.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: lae_ex3
+```
+
+The non-parametric kernel learner has one free tuning constant: the bandwidth $h$.
+
+Explore its effect.
+
+Run the kernel learner for a range of bandwidths and report the sup and ergodic-weighted errors
+against the benchmark.
+
+Explain the shape of the trade-off, and relate the two failure modes to the bias–variance
+decomposition.
+
+```{exercise-end}
+```
+
+```{solution-start} lae_ex3
+:class: dropdown
+```
+
+```{code-cell} ipython3
+rows = []
+for h in (0.03, 0.05, 0.08, 0.12, 0.20):
+ s_h = kernel_smoother(h=h)
+ sup = np.max(np.abs(s_h - s_benchmark))
+ rms = np.sqrt(ergodic @ (s_h - s_benchmark)**2)
+ rows.append([h, sup, rms])
+
+table = pd.DataFrame(rows, columns=["bandwidth $h$", "sup error",
+ "ergodic RMS"]).set_index("bandwidth $h$")
+table.round(4)
+```
+
+```{code-cell} ipython3
+fig, ax = plt.subplots(figsize=(7, 4.2))
+for h, colour in [(0.03, 'C0'), (0.08, 'C1'), (0.20, 'C2')]:
+ ax.plot(G, kernel_smoother(h=h), 'o-', ms=3, lw=0.8, color=colour, label=f"$h = {h}$")
+ax.plot(G, s_benchmark, 'k-', lw=2, label="benchmark")
+ax.set_xlabel("deficit $G$")
+ax.set_ylabel("saving $s$")
+ax.legend(frameon=False)
+plt.show()
+```
+
+There is an interior optimum.
+
+A **small** bandwidth averages over few neighbours, so the estimate is noisy: it wiggles around
+the benchmark, and the wiggle is worst at the edges where neighbours are scarce.
+
+This is the **variance** end of the trade-off.
+
+A **large** bandwidth averages over a wide window, smoothing the estimate but flattening the
+genuine curvature of $f(G)$, so the rule is pulled toward a straight line.
+
+This is the **bias** end.
+
+The bandwidth is the non-parametric counterpart of the polynomial degree in {ref}`lae_ex1`:
+both control how much structure the learner imposes, and both have a sweet spot that trades
+misfit-from-too-little-flexibility against noise-from-too-much.
+
+The bounded-rationality program does not tell us where that spot is; it is one more choice that
+replaces the single discipline of rational expectations with a menu of plausible alternatives.
+
+```{solution-end}
+```
diff --git a/lectures/marimon_mcgrattan_sargent.md b/lectures/marimon_mcgrattan_sargent.md
new file mode 100644
index 000000000..c09115200
--- /dev/null
+++ b/lectures/marimon_mcgrattan_sargent.md
@@ -0,0 +1,2033 @@
+---
+jupytext:
+ text_representation:
+ extension: .md
+ format_name: myst
+ format_version: 0.13
+ jupytext_version: 1.17.1
+kernelspec:
+ display_name: Python 3 (ipykernel)
+ language: python
+ name: python3
+---
+
+(marimon_mcgrattan_sargent)=
+```{raw} jupyter
+
+```
+
+# Money as a Medium of Exchange among Artificially Intelligent Agents
+
+```{index} single: Bounded Rationality; Money as a Medium of Exchange
+```
+
+```{contents} Contents
+:depth: 2
+```
+
+## Overview
+
+Kiyotaki and Wright {cite}`KiyotakiWright1989` studied an economy in which there is no double
+coincidence of wants.
+
+Trade can occur only if some good is accepted not because it is wanted, but because it can
+be passed along later.
+
+A good that plays that role is a **medium of exchange**.
+
+Kiyotaki and Wright characterized *stationary Nash equilibria* of such an economy under the
+assumption that agents are fully rational: agents know the distribution of goods across
+their trading partners and best-respond to it.
+
+Marimon, McGrattan and Sargent {cite}`MarimonMcGrattanSargent1990` asked a different question.
+
+Suppose we drop rationality altogether and put in its place a collection of **artificially
+intelligent agents** who begin with arbitrary, even random, rules of thumb, and who adapt
+those rules only by keeping score of what has paid off in the past.
+
+Will such agents *learn* to use a medium of exchange?
+
+And when the rational-expectations model has more than one equilibrium, which one, if any,
+will emerge?
+
+The learning device that Marimon, McGrattan and Sargent used is John Holland's
+**classifier system** {cite}`Holland1975,HollandHolyoakNisbettThagard1986`: a population of if-then rules, an auction that
+decides which rule acts, and an accounting system that credits rules that lead to good
+outcomes and debits rules that lead to bad ones.
+
+Optionally, a **genetic algorithm** breeds new rules and retires old ones.
+
+This lecture rebuilds their computational experiments.
+
+The main findings that we shall reproduce are:
+
+1. Starting from a complete enumeration of rules with equal strengths, or even from
+ randomly generated rules, holdings and trading patterns converge to a stationary Nash
+ equilibrium of the Kiyotaki-Wright model.
+1. When the Kiyotaki-Wright model has both a *fundamental* and a *speculative* equilibrium,
+ the artificially intelligent agents select the fundamental one -- the one in which the
+ good with the lowest storage cost circulates as money.
+1. An intrinsically worthless, costlessly stored object -- **fiat money** -- is accepted in
+ trade by agents who have to discover its usefulness for themselves.
+1. The same machinery works in an economy with five goods and five types for which the
+ authors had no analytical characterization of equilibrium: the algorithm is used as an
+ *equilibrium-discovery device*.
+
+Let's start with some imports.
+
+```{code-cell} ipython3
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+from dataclasses import dataclass
+```
+
+## The Kiyotaki-Wright environment
+
+There are three types of agents, indexed by $i = 1, 2, 3$, and three goods, indexed by
+$k = 1, 2, 3$.
+
+A type $i$ agent derives utility only from consuming good $i$.
+
+He has a technology for producing good $i^*$, where $i^* \neq i$.
+
+In Kiyotaki and Wright's *model A*, the production pattern is the "Wicksell triangle"
+
+| type $i$ | produces $i^*$ | consumes |
+|---|---|---|
+| 1 | 2 | 1 |
+| 2 | 3 | 2 |
+| 3 | 1 | 3 |
+
+so there is no double coincidence of wants: the agent who has what you want never wants
+what you have.
+
+All goods are indivisible and each agent can store exactly one unit of exactly one good
+from one period to the next.
+
+Storing good $k$ for one period costs $s_k$, with
+
+$$
+s_3 > s_2 > s_1 > 0 .
+$$
+
+There are equal numbers $A_i$ of agents of each type, so $A = 3 A_i$ in total.
+
+Each period every agent is randomly matched with exactly one other agent, without regard to
+type.
+
+Write $x_{at}$ for the good that agent $a$ carries into date $t$ and $\rho_t(a)$ for the
+agent with whom $a$ is matched.
+
+The **pre-trade state** of agent $a$ is the pair
+
+$$
+z_{at} = \bigl(x_{at},\; x_{\rho_t(a)t}\bigr) .
+$$
+
+Each period each agent makes two decisions in sequence.
+
+**First**, having seen $z_{at}$, he decides whether to propose a trade,
+
+$$
+\lambda_{at} = \begin{cases}
+1 & \text{propose to trade } x_{at} \text{ for } x_{\rho_t(a)t} \\
+0 & \text{refuse}
+\end{cases}
+$$
+
+Trade occurs if and only if $\lambda_{at} \lambda_{\rho_t(a)t} = 1$, so post-trade holdings
+are
+
+```{math}
+:label: mms_posttrade
+
+x^+_{at} = (1 - \lambda_{at}\lambda_{\rho_t(a)t}) x_{at}
+ + \lambda_{at}\lambda_{\rho_t(a)t} x_{\rho_t(a)t} .
+```
+
+**Second**, he decides whether to consume what he is left holding,
+
+$$
+\gamma_{at} = \begin{cases}
+1 & \text{consume } x^+_{at} \\
+0 & \text{carry } x^+_{at} \text{ into } t+1
+\end{cases}
+$$
+
+If he consumes he immediately produces good $f(a) = i^*$ and carries that into $t+1$.
+
+Hence
+
+```{math}
+:label: mms_lom
+
+x_{a,t+1} = \gamma_{at} f(a) + (1 - \gamma_{at}) x^+_{at} .
+```
+
+The one-period net payoff is
+
+```{math}
+:label: mms_payoff
+
+U_a(\gamma_{at}) =
+\gamma_{at}\bigl[u_i(x^+_{at}) - s(f(a))\bigr]
+- (1 - \gamma_{at})\, s(x^+_{at}) ,
+```
+
+where $u_i(k) = u_i > 0$ if $k = i$ and $u_i(k) = 0$ otherwise.
+
+Note that an agent is *permitted* to consume a good he does not want; he simply gets zero
+utility from it while still producing and paying to store $f(a)$.
+
+Not consuming is not free either -- it costs $s(x^+_{at})$.
+
+Learning which of these to do is part of the problem.
+
+```{note}
+Kiyotaki and Wright ranked payoff streams by expected discounted utility.
+
+Marimon, McGrattan and Sargent instead assume that each agent cares about his **long-run
+average** utility.
+
+This matters: as we shall see, it is the reason the accounting system inside a classifier
+system is built out of running averages.
+```
+
+### Two equilibria
+
+Since agents' payoffs depend on what other agents do, the model can have more than one
+stationary equilibrium.
+
+Kiyotaki and Wright characterize equilibria by a set of probabilities, of which the most
+useful for us is
+
+$$
+\pi^h_{it}(k) = \text{probability that a type } i \text{ agent holds good } k \text{ at } t .
+$$
+
+In the **fundamental equilibrium**, good 1 -- the cheapest to store -- is the general medium
+of exchange:
+
+* type 1 agents always hold good 2 (which they produce) and trade it for good 1;
+* type 3 agents always hold good 1 and trade it for good 3;
+* type 2 agents hold good 1 half the time and good 3 half the time.
+
+So the equilibrium holding probabilities are
+
+| | $k=1$ | $k=2$ | $k=3$ |
+|---|---|---|---|
+| $i=1$ | 0 | 1 | 0 |
+| $i=2$ | 0.5 | 0 | 0.5 |
+| $i=3$ | 1 | 0 | 0 |
+
+Type 2 agents accept good 1 even though they never consume it: they use it as money.
+
+In the **speculative equilibrium**, type 1 agents additionally accept good 3 -- the *most*
+expensive good to store -- because they expect to trade it away quickly for good 1.
+
+Kiyotaki and Wright show that, in the limit as the discount rate goes to zero, the
+fundamental equilibrium is the unique stationary equilibrium if
+
+```{math}
+:label: mms_fundcond
+
+s_3 - s_2 > \bigl(\pi^h_1(3) - \pi^h_1(2)\bigr)\tfrac{1}{3} u_1 ,
+```
+
+and the speculative equilibrium is the unique one when the inequality is reversed.
+
+Raising $u_1$ enough therefore flips the model's prediction from fundamental to
+speculative.
+
+Economy A2 below does exactly that, and it is a good test of whether adaptive agents track
+the rational-expectations prediction.
+
+## Classifier systems
+
+An agent is not endowed with a strategy.
+
+He is endowed with a **population of candidate rules** and a way of keeping score.
+
+A *classifier* is a string in the trinary alphabet $\{0, 1, \#\}$ split into a
+**condition** and an **action**, where $\#$ means "don't care".
+
+{cite}`Goldberg1989` is a book-length treatment of classifier systems and genetic
+algorithms.
+
+Goods are encoded in binary and conditions in trinary, so with three goods two positions
+suffice:
+
+| code | meaning |
+|---|---|
+| `1 0` | good 1 |
+| `0 1` | good 2 |
+| `0 0` | good 3 |
+| `0 #` | not good 1 |
+| `# 0` | not good 2 |
+| `# #` | any good |
+
+An **exchange classifier** is a string of length 7: two positions for own holding, two for
+the partner's holding, and a final binary digit for the action ($1$ = propose trade, $0$ =
+refuse).
+
+For example
+
+```
+1 0 0 0 1 -> 1 "if I hold good 1 and my partner holds good 3, propose to trade"
+1 0 # # -> 0 "if I hold good 1, refuse to trade with anyone"
+```
+
+There are $6 \times 6 \times 2 = 72$ distinct exchange classifiers, which is a *complete
+enumeration* of all rules definable on the state $z_{at}$.
+
+A **consumption classifier** is a string of length 4: two positions for the post-trade holding
+$x^+_{at}$ and one action digit ($1$ = consume).
+
+There are $6 \times 2 = 12$ of these.
+
+### The auction
+
+Attached to classifier $e$ at date $t$ is a **strength** $S^a_e(t)$.
+
+Given the state $z_{at}$, let
+
+$$
+M_e(z_{at}) = \{e : z_{at} \text{ matches the condition part of } e\}
+$$
+
+be the set of classifiers whose conditions are satisfied.
+
+The classifier that acts is the strongest matched one,
+
+```{math}
+:label: mms_auction
+
+e_t(z_{at}) = \arg\max\,\{S^a_e(t) : e \in M_e(z_{at})\} ,
+```
+
+and the consumption classifier is chosen the same way from $M_c(z_{at})$.
+
+### The bucket brigade
+
+Strengths are updated by a system of internal payments that Holland calls a *bucket
+brigade*.
+
+Only the classifier that wins an auction changes its strength, so we attach to each
+classifier a counter $\tau^a_e(t)$ recording the number of auctions it has won up to $t$,
+initialized at 1.
+
+A matched classifier $e$ bids the fraction
+
+$$
+b_1(e) = b_{11} + b_{12}\sigma_e ,
+\qquad
+\sigma_e = \frac{1}{1 + \text{number of } \#\text{'s in } e}
+$$
+
+of its strength, and similarly $b_2(c) = b_{21} + b_{22}\sigma_c$ for consumption
+classifiers.
+
+Because $\sigma_e$ rises with specificity, specific rules outbid general ones of equal
+strength.
+
+Payments flow as follows.
+
+* The external payoff $U_a(\gamma_t)$ goes to the winning **consumption** classifier at $t$.
+* The winning consumption classifier at $t$ pays its bid to the winning **exchange**
+ classifier at $t$, which created the state that gave it a chance to act.
+* The winning exchange classifier at $t$ pays its bid to the winning **consumption**
+ classifier at $t-1$, which set up the state $z_{at}$.
+
+This chain is what transmits the reward for consuming backwards to the trades that made
+consumption possible.
+
+The resulting laws of motion are
+
+```{math}
+:label: mms_strengthc
+
+S^a_{c,\tau_c(t)} = S^a_{c,\tau_c(t)-1}
+ - \frac{1}{\tau_c(t)-1}\Bigl[(1 + b_2(c))S^a_{c,\tau_c(t)-1}
+ - \sum_e I^a_e(t) b_1(e) S^a_{e,\tau_e(t)} - U_a(\gamma_{ct})\Bigr]
+```
+
+```{math}
+:label: mms_strengthe
+
+S^a_{e,\tau_e(t)+1} = S^a_{e,\tau_e(t)}
+ - \frac{1}{\tau_e(t)}\Bigl[(1 + b_1(e))S^a_{e,\tau_e(t)}
+ - \sum_c I^a_c(t) b_2(c) S^a_{c,\tau_c(t)}\Bigr]
+```
+
+where $I^a_e(t)$ and $I^a_c(t)$ are indicators for winning the auction at $t$.
+
+The timing in {eq}`mms_strengthc` is worth reading carefully, because it is what carries reward
+across periods.
+
+The consumption classifier updated at date $t$ is the one that won at $t-1$: it collects the
+external payoff its own decision earned, and it collects the bid $b_1(e)S_e$ from the exchange
+classifier winning *now*, because that is the classifier whose opportunity to act it created.
+
+The exchange classifier, in turn, is paid within the period by the consumption classifier that
+follows it.
+
+So each bid travels one step back along the chain
+
+$$
+\cdots \;\to\; c_{t-1} \;\to\; e_t \;\to\; c_t \;\to\; e_{t+1} \;\to\; \cdots
+$$
+
+and a payoff collected at consumption seeps backwards, one link per payment, to the trades that
+made it possible.
+
+An exchange classifier that proposes a trade which is *not* reciprocated is not charged and
+its counter does not advance: the agent learns nothing from an offer that was refused.
+
+```{note}
+Equations {eq}`mms_strengthc`-{eq}`mms_strengthe` make strength a **cumulative average** of
+past net receipts rather than a cumulative total, which is what Holland's original
+specification used.
+
+This is the innovation that makes strengths converge.
+
+With $1/\tau$ gains these are stochastic approximation recursions, so any limit point must
+satisfy
+
+$$
+\mathbb{E}\Bigl[(1 + b_2(c))S_c - \sum_e I_e b_1(e) S_e - U(\gamma_c)\Bigr] = 0,
+\qquad
+\mathbb{E}\Bigl[(1 + b_1(e))S_e - \sum_c I_c b_2(c) S_c\Bigr] = 0 .
+$$
+
+Marimon, McGrattan and Sargent define a set of strengths solving these equations to be
+*stationary*, and a stationary Nash equilibrium to be supported when the rules that win
+auctions at stationary strengths are exactly the rules that support equilibrium behavior.
+```
+
+All agents of a given type share one classifier system, as in the paper; this economizes on
+computation at the cost of making all type $i$ agents experiment simultaneously.
+
+## Implementation
+
+We store a population of classifiers as a set of parallel NumPy arrays rather than as a
+list of objects, which lets us find all matched rules with a single vectorized comparison.
+
+The wildcard $\#$ is represented by $-1$.
+
+```{code-cell} ipython3
+WILD = -1
+
+# binary codes for goods, one row per good
+CODES = {
+ 3: np.array([[1, 0], [0, 1], [0, 0]]),
+ 4: np.array([[1, 0], [0, 1], [0, 0], [1, 1]]), # good 4 = fiat money
+ 5: np.array([[1, 0, 0], [0, 1, 0], [0, 0, 1],
+ [1, 1, 0], [1, 0, 1]]),
+}
+
+# the six conditions expressible with two trits (see the table above)
+CONDS_2 = np.array([[1, 0], [0, 1], [0, 0],
+ [0, WILD], [WILD, 0], [WILD, WILD]])
+
+
+def rule_string(cond, action):
+ "Print a classifier the way the paper does."
+ body = ''.join('#' if b == WILD else str(int(b)) for b in cond)
+ return f"{body} -> {int(action)}"
+```
+
+```{code-cell} ipython3
+class Rules:
+ """
+ A population of classifiers held as parallel arrays.
+
+ cond[i] condition part of rule i, entries in {0, 1, WILD}
+ action[i] action part of rule i, in {0, 1}
+ strength[i] S_i, a running average of net receipts
+ used[i] the counter tau_i, initialized at 1
+ traded[i] number of times rule i actually executed a trade
+
+ """
+
+ def __init__(self, cond, action, strength=None):
+ self.cond = np.asarray(cond, dtype=np.int64)
+ self.action = np.asarray(action, dtype=np.int64)
+ n = len(self.action)
+ self.strength = (np.zeros(n) if strength is None
+ else np.asarray(strength, float).copy())
+ self.used = np.ones(n, dtype=np.int64)
+ self.traded = np.zeros(n, dtype=np.int64)
+
+ @property
+ def n(self):
+ return len(self.action)
+
+ @property
+ def length(self):
+ return self.cond.shape[1]
+
+ def matched(self, state):
+ "Boolean mask of rules whose condition is satisfied by state."
+ return np.all((self.cond == WILD) | (self.cond == state), axis=1)
+
+ def specificity(self, i):
+ "sigma_i = 1 / (1 + number of wildcards)."
+ return 1.0 / (1.0 + np.count_nonzero(self.cond[i] == WILD))
+
+ def replace(self, i, cond, action, strength, used=1, traded=0):
+ "Overwrite rule i."
+ self.cond[i] = cond
+ self.action[i] = action
+ self.strength[i] = strength
+ self.used[i] = used
+ self.traded[i] = traded
+```
+
+A complete enumeration pairs every condition with both actions.
+
+A random population draws conditions and actions uniformly, which is how the
+incomplete-enumeration economies start.
+
+```{code-cell} ipython3
+def complete_rules(pair):
+ "All two-trit rules; pair=True for exchange rules, False for consumption rules."
+ if pair:
+ conds = np.array([np.concatenate([a, b])
+ for a in CONDS_2 for b in CONDS_2])
+ else:
+ conds = CONDS_2.copy()
+ return Rules(np.repeat(conds, 2, axis=0), np.tile([0, 1], len(conds)))
+
+
+def random_rules(n, length, rng):
+ return Rules(rng.integers(-1, 2, size=(n, length)),
+ rng.integers(0, 2, size=n),
+ rng.random(n) * 0.1)
+```
+
+The auction {eq}`mms_auction` picks the strongest matched rule, with ties broken by
+position.
+
+```{code-cell} ipython3
+def auction(rules, state):
+ "Index of the strongest rule matching state, or -1 if none matches."
+ idx = np.flatnonzero(rules.matched(state))
+ if idx.size == 0:
+ return -1, idx
+ return idx[np.argmax(rules.strength[idx])], idx
+```
+
+Two operators are needed even without a genetic algorithm.
+
+**Creation** fires when the current state matches no rule at all: a redundant or weak rule
+is overwritten by one whose condition is exactly the state just observed, with a randomly
+drawn action.
+
+**Diversification** fires when every matched rule calls for the same action: a rule with the
+opposite action is planted so that the alternative can be tried and scored.
+
+Both keep the population size constant.
+
+```{code-cell} ipython3
+def create(rules, state, rng):
+ "No rule matches state, so overwrite the weakest of the most redundant rules."
+ groups = {}
+ for i in range(rules.n):
+ groups.setdefault(tuple(rules.cond[i]), []).append(i)
+ biggest = max(groups.values(), key=len)
+ group = biggest if len(biggest) > 1 else range(rules.n)
+ j = min(group, key=lambda i: rules.strength[i])
+ rules.replace(j, state, rng.integers(0, 2), rules.strength.mean())
+ return j
+
+
+def diversify_simple(rules, matches):
+ "If every matched rule takes the same action, plant the opposite action."
+ if len(set(rules.action[matches])) > 1:
+ return
+ weak = matches[np.argmin(rules.strength[matches])]
+ rules.replace(weak, rules.cond[weak], 1 - rules.action[matches[0]],
+ rules.strength[matches].mean())
+```
+
+### Describing an economy
+
+An `Economy` collects the parameters of the physical environment together with the settings
+of the learning algorithm.
+
+The field `method` selects one of the three programs used in the paper:
+
+* `'enumerate'` -- start from a complete enumeration of rules with zero strengths, and use
+ no genetic algorithm;
+* `'ga3'` -- start from random rules and evolve them with single-point crossover and
+ mutation;
+* `'ga4'` -- start from random rules and evolve them with a *generalizing* crossover, used
+ for the largest economy.
+
+Fiat money, when present, is the last good: it has zero storage cost, gives no utility, and
+cannot be consumed.
+
+```{code-cell} ipython3
+@dataclass
+class Economy:
+ name: str
+ produces: np.ndarray # produces[i] = good produced by type i
+ storage_costs: np.ndarray # one entry per good
+ u: float = 100.0 # utility from own consumption good
+ n_agents_per_type: int = 50
+ method: str = 'enumerate' # 'enumerate' | 'ga3' | 'ga4'
+ n_trade_rules: int = 72
+ n_consume_rules: int = 12
+ b_trade: tuple = (0.025, 0.025) # (b11, b12)
+ b_consume: tuple = (0.25, 0.25) # (b21, b22)
+ n_fiat: int = 0 # units of fiat money injected at t = 0
+ start: str = 'random' # initial holdings: 'random' | 'production'
+ pcross: float = 0.6
+ pmutation: float = 0.01
+
+ @property
+ def n_types(self):
+ return len(self.produces)
+
+ @property
+ def n_goods(self):
+ return len(self.storage_costs)
+
+ @property
+ def n_agents(self):
+ return self.n_types * self.n_agents_per_type
+
+ @property
+ def fiat(self):
+ return self.n_goods > self.n_types
+
+ @property
+ def n_bits(self):
+ return CODES[self.n_goods].shape[1]
+
+ def code(self, good):
+ return CODES[self.n_goods][good]
+```
+
+### The agent
+
+An `Agent` represents one *type*: it owns an exchange population and a consumption
+population, and remembers which consumption rule won last period so that it can be paid by
+this period's exchange rule.
+
+```{code-cell} ipython3
+class Agent:
+
+ def __init__(self, i, econ, rng):
+ self.i, self.econ, self.rng = i, econ, rng
+ if econ.method == 'enumerate':
+ self.trade = complete_rules(pair=True)
+ self.consume = complete_rules(pair=False)
+ else:
+ self.trade = random_rules(econ.n_trade_rules, 2 * econ.n_bits, rng)
+ self.consume = random_rules(econ.n_consume_rules, econ.n_bits, rng)
+ self.pending = None # last period's consumption rule, awaiting settlement
+
+ def decide(self, rules, state, specialize=False):
+ "Run the auction on `state`, applying the operators the method calls for."
+ win, matches = auction(rules, state)
+ if win < 0: # creation
+ j = create(rules, state, self.rng)
+ return rules.action[j], j
+ if self.econ.method == 'enumerate':
+ diversify_simple(rules, matches)
+ elif self.econ.method == 'ga4':
+ diversify_clone(rules, matches, win)
+ if specialize:
+ specialize_winner(rules, win, self.econ.pmutation, self.rng)
+ if self.econ.method != 'ga3':
+ win, matches = auction(rules, state) # the population may have changed
+ return rules.action[win], win
+
+ def trade_decision(self, own, partner, specialize=False):
+ state = np.concatenate([self.econ.code(own), self.econ.code(partner)])
+ return self.decide(self.trade, state, specialize)
+
+ def consume_decision(self, good, specialize=False):
+ return self.decide(self.consume, self.econ.code(good), specialize)
+
+ def update(self, e, c, payoff, active):
+ """
+ The bucket brigade laws of motion for strengths.
+
+ `e` and `c` index the winning exchange and consumption rules and `active`
+ records whether the exchange rule's action was actually carried out.
+
+ """
+ b11, b12 = self.econ.b_trade
+ b21, b22 = self.econ.b_consume
+ T, C = self.trade, self.consume
+ b1 = b11 + b12 * T.specificity(e)
+ b2 = b21 + b22 * C.specificity(c)
+
+ if active: # exchange rule: pays b1, receives b2 * S_c
+ τ = T.used[e]
+ T.used[e] += 1
+ T.traded[e] += int(T.action[e] == 1)
+ T.strength[e] -= ((1 + b1) * T.strength[e] - b2 * C.strength[c]) / τ
+
+ # Now settle last period's consumption rule. Its update waits a period
+ # because only now is the second of its two receipts known: it collects
+ # the payoff its own decision earned *and* the bid of the exchange rule
+ # winning today, whose chance to act it created.
+ if self.pending is not None:
+ p, u_prev = self.pending
+ inflow = b1 * T.strength[e] if active else 0.0
+ b2p = b21 + b22 * C.specificity(p)
+ tau_p = C.used[p]
+ C.used[p] += 1
+ C.strength[p] -= ((1 + b2p) * C.strength[p] - inflow - u_prev) / tau_p
+
+ self.pending = (c, payoff)
+```
+
+### The simulation
+
+Each period, all $A$ agents are shuffled into $A/2$ pairs; each pair trades, consumes, and
+updates strengths; then, in the incomplete-enumeration economies, the genetic operators run.
+
+A small proportional tax on exchange strengths, present in the authors' code, keeps
+strengths from locking in permanently.
+
+```{code-cell} ipython3
+class Simulation:
+
+ def __init__(self, econ, seed=0):
+ self.econ = econ
+ self.rng = np.random.default_rng(seed)
+ self.agents = [Agent(i, econ, self.rng) for i in range(econ.n_types)]
+ self.types = np.repeat(np.arange(econ.n_types), econ.n_agents_per_type)
+
+ if econ.start == 'random':
+ self.holdings = self.rng.integers(0, econ.n_types, size=econ.n_agents)
+ else:
+ self.holdings = econ.produces[self.types].copy()
+ if econ.n_fiat:
+ who = self.rng.choice(econ.n_agents, size=econ.n_fiat, replace=False)
+ self.holdings[who] = econ.n_goods - 1
+
+ self.hold_hist, self.exch_hist, self.cons_hist = [], [], []
+ self.trades, self.eaten = [], []
+
+ def run(self, T, verbose=False):
+ econ, rng = self.econ, self.rng
+ evolving = econ.method != 'enumerate'
+
+ if evolving:
+ # the genetic algorithm fires on even dates with probability 1/sqrt(t/2)
+ p = 1.0 / np.sqrt(np.arange(1, T // 2 + 1))
+ even = np.arange(1, T, 2)
+ ga_dates = np.zeros(T, dtype=bool)
+ ga_dates[even] = p[:len(even)] > rng.random(len(even))
+ spec_dates = np.zeros(T, dtype=bool)
+ spec_dates[even] = p[:len(even)] > rng.random(len(even))
+
+ for t in range(1, T + 1):
+ spec = evolving and econ.method == 'ga4' and spec_dates[t - 1]
+ n_trades = n_eaten = 0
+ exch = np.zeros((econ.n_types, econ.n_goods, econ.n_goods))
+ cons = np.zeros((econ.n_types, econ.n_goods, 2))
+
+ order = rng.permutation(econ.n_agents)
+ for k in range(econ.n_agents // 2):
+ a, b = order[2 * k], order[2 * k + 1]
+ ia, ib = self.types[a], self.types[b]
+ ga, gb = self.holdings[a], self.holdings[b]
+ A, B = self.agents[ia], self.agents[ib]
+
+ # --- exchange ---
+ la, ea = A.trade_decision(ga, gb, spec)
+ lb, eb = B.trade_decision(gb, ga, spec)
+ swap = (la == 1) and (lb == 1)
+ if swap:
+ self.holdings[a], self.holdings[b] = gb, ga
+ n_trades += 1
+ exch[ia, ga, gb] += 1
+ exch[ib, gb, ga] += 1
+
+ # --- consumption ---
+ pa, pb = self.holdings[a], self.holdings[b]
+ ca, wa = A.consume_decision(pa, spec)
+ cb, wb = B.consume_decision(pb, spec)
+ cons[ia, pa, 0] += 1
+ cons[ib, pb, 0] += 1
+
+ ua = self.consume_and_produce(a, ia, pa, ca)
+ if ca == 1:
+ cons[ia, pa, 1] += 1
+ n_eaten += pa == ia
+ ub = self.consume_and_produce(b, ib, pb, cb)
+ if cb == 1:
+ cons[ib, pb, 1] += 1
+ n_eaten += pb == ib
+
+ # --- accounting ---
+ A.update(ea, wa, ua, swap or la == 0)
+ B.update(eb, wb, ub, swap or lb == 0)
+
+ if evolving:
+ if ga_dates[t - 1]:
+ self.evolve()
+ if econ.method == 'ga3':
+ for A in self.agents:
+ specialize_all(A.trade, t, rng)
+ specialize_all(A.consume, t, rng)
+ self.tax()
+
+ self.hold_hist.append(self.distribution())
+ self.exch_hist.append(exch)
+ self.cons_hist.append(cons)
+ self.trades.append(n_trades)
+ self.eaten.append(n_eaten)
+ if verbose and t % max(1, T // 10) == 0:
+ print(f" period {t:5d}: trades = {n_trades:3d},"
+ f" consumptions = {n_eaten:3d}")
+
+ def consume_and_produce(self, a, i, good, action):
+ """
+ Carry out the consumption decision and return the external payoff.
+
+ Consuming fiat money is not allowed. If the agent does consume, he
+ immediately produces his own good, so this updates his holding.
+
+ """
+ econ = self.econ
+ if action == 1 and not (econ.fiat and good == econ.n_goods - 1):
+ new = econ.produces[i]
+ self.holdings[a] = new
+ u = econ.u if good == i else 0.0
+ return u - econ.storage_costs[new]
+ return -econ.storage_costs[good]
+
+ def distribution(self):
+ "The matrix of holding frequencies pi^h_it(k)."
+ econ = self.econ
+ d = np.zeros((econ.n_types, econ.n_goods))
+ for i in range(econ.n_types):
+ h = self.holdings[self.types == i]
+ for k in range(econ.n_goods):
+ d[i, k] = np.mean(h == k)
+ return d
+
+ def tax(self):
+ if self.econ.method == 'ga4':
+ for A in self.agents:
+ T, C = A.trade, A.consume
+ P = np.where(T.action == 1, T.traded, T.used) + 1
+ T.strength -= (T.strength + 1.0) / P
+ C.strength -= (C.strength + 1.0) / (C.used + 1)
+ else:
+ for A in self.agents:
+ A.trade.strength -= 1e-4 * np.abs(A.trade.strength)
+
+ def pick_types(self):
+ "Send one type to the genetic algorithm, a second and a third each w.p. 0.33."
+ rng, n = self.rng, self.econ.n_types
+ chosen = [int(rng.integers(n))]
+ rest = [i for i in range(n) if i not in chosen]
+ while rest and len(chosen) < 3 and rng.random() < 0.33:
+ pick = rest[int(rng.integers(len(rest)))]
+ chosen.append(pick)
+ rest.remove(pick)
+ return chosen
+
+ def evolve(self):
+ econ = self.econ
+ gen = econ.method == 'ga4'
+ for i in self.pick_types():
+ genetic_algorithm(self.agents[i].trade, self.rng, generalize=gen,
+ pcross=econ.pcross, pmutation=econ.pmutation,
+ crowd_factor=8)
+ for i in self.pick_types():
+ genetic_algorithm(self.agents[i].consume, self.rng, generalize=gen,
+ pcross=econ.pcross, pmutation=econ.pmutation,
+ crowd_factor=4)
+```
+
+```{note}
+`Agent` and `Simulation` refer to four functions -- `genetic_algorithm`, `specialize_all`,
+`diversify_clone` and `specialize_winner` -- that belong to the genetic algorithm and are
+therefore deferred to {ref}`mms_ga` below, where they can be motivated by the problem they
+solve.
+
+Deferring them costs nothing.
+
+Python looks names up when a function runs rather than when it is defined, and the
+complete-enumeration economies we study first never call any of the four.
+
+Readers who prefer to see the machinery before it is used can run that section's cells first.
+```
+
+### Reporting
+
+The following helpers put simulation output into the format of the paper's tables.
+
+We follow the paper in reporting ten-period moving averages.
+
+```{code-cell} ipython3
+def good_names(econ):
+ names = [f"good {k+1}" for k in range(econ.n_types)]
+ return names + ["fiat"] if econ.fiat else names
+
+
+def type_names(econ):
+ return [f"type {i+1}" for i in range(econ.n_types)]
+
+
+def holdings(sim, t=None, window=10):
+ r"Table of $\pi^h_{it}(j)$, averaged over the `window` periods ending at `t`."
+ h = np.array(sim.hold_hist)
+ t = len(h) if t is None else t
+ d = h[max(0, t - window):t].mean(axis=0)
+ return pd.DataFrame(d, index=type_names(sim.econ),
+ columns=good_names(sim.econ)).round(3)
+
+
+def exchanges(sim, t=None, window=10):
+ r"""
+ Table of $\pi^e_{it}(jk)$: the frequency with which a type $i$ agent holds
+ good $j$, meets an agent holding good $k$, and trades. Row $j$ of column
+ $i$ holds the triple over $k$.
+ """
+ econ = sim.econ
+ e = np.array(sim.exch_hist)
+ t = len(e) if t is None else t
+ f = e[max(0, t - window):t].mean(axis=0) / econ.n_agents_per_type
+ cols = {type_names(econ)[i]:
+ ["(" + ", ".join(f"{f[i, j, k]:.2f}" for k in range(econ.n_goods)) + ")"
+ for j in range(econ.n_goods)]
+ for i in range(econ.n_types)}
+ return pd.DataFrame(cols, index=good_names(econ)).T
+
+
+def winning_actions(sim):
+ r"""
+ Table of $\tilde\pi^e_{it}(jk|j)$: the action chosen by the winning exchange
+ rule in each state, whether or not that state is ever visited.
+ """
+ econ = sim.econ
+ cols = {}
+ for i, A in enumerate(sim.agents):
+ col = []
+ for j in range(econ.n_goods):
+ acts = []
+ for k in range(econ.n_goods):
+ w, _ = auction(A.trade, np.concatenate([econ.code(j), econ.code(k)]))
+ acts.append('-' if w < 0 else str(int(A.trade.action[w])))
+ col.append("(" + ",".join(acts) + ")")
+ cols[type_names(econ)[i]] = col
+ return pd.DataFrame(cols, index=good_names(econ)).T
+
+
+def consume_actions(sim):
+ r"Table of the winning consumption action for each post-trade holding."
+ econ = sim.econ
+ cols = {}
+ for i, A in enumerate(sim.agents):
+ col = []
+ for j in range(econ.n_goods):
+ w, _ = auction(A.consume, econ.code(j))
+ col.append('-' if w < 0 else int(A.consume.action[w]))
+ cols[type_names(econ)[i]] = col
+ return pd.DataFrame(cols, index=good_names(econ)).T
+
+
+def strongest(rules, n=5):
+ "The n highest-strength classifiers in a population."
+ order = np.argsort(-rules.strength)[:n]
+ return pd.DataFrame({
+ 'classifier': [rule_string(rules.cond[i], rules.action[i]) for i in order],
+ 'strength': rules.strength[order].round(2),
+ 'times used': rules.used[order]})
+```
+
+Two figures: the time path of holdings, which corresponds to the paper's figures 6-9, and a
+diagram of the exchange pattern that the system discovers, which corresponds to its
+figures 2, 4, 9 and 11.
+
+```{code-cell} ipython3
+def plot_holdings(sim, title=None):
+ econ = sim.econ
+ h = np.array(sim.hold_hist)
+ names = good_names(econ)
+ fig, axes = plt.subplots(1, econ.n_types,
+ figsize=(3.2 * econ.n_types, 3.2), sharey=True)
+ axes = np.atleast_1d(axes)
+ for i, ax in enumerate(axes):
+ for k in range(econ.n_goods):
+ ax.plot(h[:, i, k], lw=1.0, label=names[k])
+ ax.set_title(f"type {i+1}")
+ ax.set_xlabel("$t$")
+ ax.set_ylim(-0.03, 1.03)
+ axes[0].set_ylabel(r"$\pi^h_{it}(j)$")
+ axes[-1].legend(frameon=False, fontsize=8, loc='center right')
+ if title:
+ fig.suptitle(title)
+ plt.tight_layout()
+ plt.show()
+
+
+def plot_flows(sim, window=100, cutoff=0.02, title=None):
+ """
+ One panel per type. An arrow from good j to good k means that agents of that
+ type give up j and receive k in trade; its width is the frequency with which
+ that exchange occurs over the last `window` periods.
+ """
+ econ = sim.econ
+ f = np.array(sim.exch_hist)[-window:].mean(axis=0) / econ.n_agents_per_type
+ names, n = good_names(econ), econ.n_goods
+ ang = np.pi / 2 + 2 * np.pi * np.arange(n) / n
+ xy = np.column_stack([np.cos(ang), np.sin(ang)])
+
+ fig, axes = plt.subplots(1, econ.n_types, figsize=(2.9 * econ.n_types, 3.1))
+ axes = np.atleast_1d(axes)
+ for i, ax in enumerate(axes):
+ for k in range(n):
+ ax.plot(*xy[k], 'o', ms=22, mfc='white', mec='black', zorder=2)
+ ax.annotate(names[k].replace(' ', '\n'), xy[k], ha='center',
+ va='center', fontsize=6.5, zorder=3)
+ for j in range(n):
+ for k in range(n):
+ if j == k or f[i, j, k] < cutoff:
+ continue
+ a, b = xy[j], xy[k]
+ d = b - a
+ ax.annotate("", xy=b - 0.24 * d, xytext=a + 0.24 * d, zorder=1,
+ arrowprops=dict(arrowstyle="-|>", color="C0",
+ lw=1 + 8 * f[i, j, k], alpha=0.7,
+ connectionstyle="arc3,rad=0.15"))
+ ax.set_title(f"type {i+1}", fontsize=10)
+ ax.set_xlim(-1.45, 1.45)
+ ax.set_ylim(-1.45, 1.45)
+ ax.set_aspect('equal')
+ ax.axis('off')
+ if title:
+ fig.suptitle(title)
+ plt.tight_layout()
+ plt.show()
+```
+
+## Economy A1.1: does a medium of exchange emerge?
+
+Our first economy is the Wicksell triangle with
+
+$$
+s_1 = 0.1, \quad s_2 = 1, \quad s_3 = 20, \quad u_i = 100 ,
+$$
+
+fifty agents of each type, and a complete enumeration of the 72 exchange and 12 consumption
+classifiers, all with strength zero.
+
+Since all strengths start equal, the initial auction winners are effectively arbitrary:
+agents begin with no idea what to do.
+
+Condition {eq}`mms_fundcond` holds comfortably here, so the fundamental equilibrium is the
+Kiyotaki-Wright prediction.
+
+```{code-cell} ipython3
+economy_a11 = Economy(
+ name='A1.1',
+ produces=np.array([1, 2, 0]), # type 1 -> good 2, etc. (0-indexed)
+ storage_costs=np.array([0.1, 1.0, 20.0]),
+ u=100.0,
+ method='enumerate',
+)
+
+sim_a11 = Simulation(economy_a11, seed=42)
+sim_a11.run(1000, verbose=True)
+```
+
+Here are holdings at $t = 500$ and $t = 1000$.
+
+```{code-cell} ipython3
+holdings(sim_a11, t=500)
+```
+
+```{code-cell} ipython3
+holdings(sim_a11)
+```
+
+Compare these with the fundamental equilibrium tabulated above: type 1 holds good 2 with
+probability one, type 3 holds good 1 with probability one, and type 2 splits evenly between
+goods 1 and 3.
+
+Rather than assert the comparison, let us compute it.
+
+```{code-cell} ipython3
+fundamental = np.array([[0.0, 1.0, 0.0],
+ [0.5, 0.0, 0.5],
+ [1.0, 0.0, 0.0]])
+paper_a11 = np.array([[0.0, 1.0, 0.0], # the paper's table at t = 1000
+ [0.506, 0.0, 0.494],
+ [1.0, 0.0, 0.0]])
+
+simulated = holdings(sim_a11).to_numpy()
+print(f"max |simulated - fundamental equilibrium| = "
+ f"{np.abs(simulated - fundamental).max():.3f}")
+print(f"max |simulated - paper's table| = "
+ f"{np.abs(simulated - paper_a11).max():.3f}")
+```
+
+The convergence is essentially immediate, as the time paths show.
+
+```{code-cell} ipython3
+plot_holdings(sim_a11, title="Economy A1.1")
+```
+
+Only type 2 agents carry any randomness, and the reason is instructive: they are the agents
+who use good 1 as money, so which good they are holding depends on where they are in the
+cycle "acquire money, spend money".
+
+Now let us look at the trades themselves.
+
+The entry in row $i$, column $j$ is the triple over $k$ of the frequency with which a type
+$i$ agent holds $j$, meets someone holding $k$, and trades.
+
+```{code-cell} ipython3
+exchanges(sim_a11)
+```
+
+The exchange pattern is easier to read as a picture.
+
+```{code-cell} ipython3
+plot_flows(sim_a11, title="Economy A1.1: discovered exchange pattern")
+```
+
+This is exactly the fundamental equilibrium triangle: type 1 gives up good 2 for good 1,
+type 3 gives up good 1 for good 3, and type 2 runs both legs, giving up good 3 for good 1
+and later good 1 for good 2.
+
+Good 1 circulates as money.
+
+We can also ask what the winning rules would do in states that are *never* visited, which is
+what Kiyotaki and Wright's strategies specify and what the paper reports.
+
+```{code-cell} ipython3
+winning_actions(sim_a11)
+```
+
+Reading row "type 1", the entry in column "good 2" is the triple
+$(\tilde\pi^e_1(21|2), \tilde\pi^e_1(22|2), \tilde\pi^e_1(23|2))$.
+
+A type 1 agent holding good 2 accepts good 1 and refuses good 3: he does not speculate.
+
+Finally, we can look inside a classifier system and read off the rules that won the
+competition.
+
+```{code-cell} ipython3
+strongest(sim_a11.agents[0].trade)
+```
+
+```{code-cell} ipython3
+strongest(sim_a11.agents[0].consume)
+```
+
+The strongest exchange rule for a type 1 agent is `0110 -> 1`, that is, "if I am holding
+good 2 (`01`) and my partner holds good 1 (`10`), trade" -- and it is used thousands of
+times, while the runners-up are essentially never used.
+
+The strongest consumption rule is `10 -> 1`: "if I am holding good 1, eat it".
+
+These are exactly the classifiers $e^1_{2,1,1}$ and $c^1_{1,1}$ that the paper shows support
+the fundamental equilibrium.
+
+## Economy A2.1: when theory predicts speculation
+
+Economy A2 differs from A1 only in that $u_i = 500$ instead of $100$.
+
+That is enough to violate the Kiyotaki-Wright inequality {eq}`mms_fundcond`, so for patient
+agents the unique stationary rational-expectations equilibrium is now the **speculative**
+one, in which type 1 agents accept good 3 in the expectation of trading it for good 1.
+
+Do our adaptive agents find it?
+
+```{code-cell} ipython3
+economy_a21 = Economy(
+ name='A2.1',
+ produces=np.array([1, 2, 0]),
+ storage_costs=np.array([0.1, 1.0, 20.0]),
+ u=500.0,
+ method='enumerate',
+)
+
+sim_a21 = Simulation(economy_a21, seed=42)
+sim_a21.run(1000)
+
+holdings(sim_a21)
+```
+
+Here is the speculative equilibrium that Kiyotaki and Wright's theory predicts for these
+parameters, for comparison.
+
+```{code-cell} ipython3
+speculative = pd.DataFrame([[0, 0.707, 0.293],
+ [0.586, 0, 0.414],
+ [1, 0, 0]],
+ index=type_names(economy_a21),
+ columns=good_names(economy_a21))
+speculative
+```
+
+The answer is no.
+
+The simulated holdings are those of the *fundamental* equilibrium: type 1 agents hold good 2
+essentially always and never accumulate good 3, whereas speculation requires them to hold
+good 3 nearly thirty percent of the time.
+
+It is worth looking at why.
+
+```{code-cell} ipython3
+winning_actions(sim_a21)
+```
+
+Reading row "type 1", column "good 2", the winning exchange rule refuses good 3: type 1
+agents will not make the speculative swap.
+
+Now look at what they would do with good 3 if they had it.
+
+```{code-cell} ipython3
+consume_actions(sim_a21)
+```
+
+```{code-cell} ipython3
+strongest(sim_a21.agents[0].consume, n=4)
+```
+
+This is the paper's diagnosis, visible in the output.
+
+One consumption rule, `10 -> 1`, is specific and enormously strong: "if I hold good 1, eat
+it".
+
+The rule that decides everything else is `## -> 1` -- *eat whatever you are holding* -- a
+maximally general rule with two wildcards and a strength barely above zero, yet it is the
+strongest rule matching any state other than "holding good 1".
+
+So a type 1 agent who acquired good 3 would immediately consume it for zero utility rather
+than carry it and trade it for good 1.
+
+As the paper puts it, the winning consumption classifiers of type 1 agents are too general
+-- they have too many $\#$'s to distinguish among stored goods -- so type 1 agents
+overconsume good 3, and the information that would make speculation pay never reaches the
+exchange classifiers.
+
+The authors' diagnosis is worth quoting:
+
+> *Patience requires experience.* The transfer system inside the classifier system is
+> designed to converge to a set of long-run average strengths. In the limit, the artificially
+> intelligent agents should behave as long-run average payoff maximizers... It takes time,
+> however, for optimal rules to achieve the desired strengths. The behavior of our
+> artificially intelligent agents can be very myopic at the beginning... The present
+> algorithm seems defective in that it has too little experimentation to support the
+> speculative equilibrium even in the long simulations we have run.
+
+Early on, before any rule has accumulated a meaningful average, agents behave myopically,
+and a myopic agent never accepts an expensive good.
+
+By the time strengths have settled, everyone else has already stopped speculating, so
+speculation no longer pays.
+
+The equilibrium selected is a consequence of the *learning dynamics*, not of the payoffs
+alone.
+
+## Economy B.1: a different production pattern
+
+Economy B changes the production technology -- type 1 produces good 3, type 2 produces good
+1, type 3 produces good 2 -- and compresses storage costs to $s = (1, 4, 9)$ with $u_i = 100$.
+
+Both a fundamental and a speculative equilibrium exist.
+
+The paper reports something striking: at $t = 500$ the economy looks speculative, but by
+$t = 1000$ it has moved to the fundamental equilibrium.
+
+```{code-cell} ipython3
+economy_b1 = Economy(
+ name='B.1',
+ produces=np.array([2, 0, 1]), # type 1 -> good 3, type 2 -> good 1, ...
+ storage_costs=np.array([1.0, 4.0, 9.0]),
+ u=100.0,
+ method='enumerate',
+ b_trade=(0.25, 0.25),
+)
+
+sim_b1 = Simulation(economy_b1, seed=42)
+sim_b1.run(1000)
+```
+
+```{code-cell} ipython3
+holdings(sim_b1, t=500)
+```
+
+```{code-cell} ipython3
+holdings(sim_b1)
+```
+
+```{code-cell} ipython3
+plot_holdings(sim_b1, title="Economy B.1")
+```
+
+```{code-cell} ipython3
+paper_b1 = np.array([[0.0, 0.28, 0.72], # the paper's table at t = 1000
+ [0.994, 0.0, 0.006],
+ [0.526, 0.474, 0.0]])
+
+print(f"max |simulated - paper's table| = "
+ f"{np.abs(holdings(sim_b1).to_numpy() - paper_b1).max():.3f}")
+```
+
+The end state reproduces the paper's qualitative pattern, with the largest single discrepancy
+about a tenth.
+
+Type 2 agents hold good 1 always, and type 3 agents split their holdings between good 1 and
+their own production good 2.
+
+The paper reports one further feature that our run does not reproduce.
+
+In the authors' simulation, type 3 agents held good 2 with probability one at $t = 500$ --
+refusing the cheaper good 1, which is the speculative pattern -- and only later shifted into
+good 1:
+
+> Economy B.1 displays an interesting pattern of evolution. At iteration 500 the
+> distribution of holdings and, especially, the trading patterns correspond to the
+> speculative equilibrium. However, the economy moves away from this state and by iteration
+> 1000 has practically converged to the fundamental equilibrium.
+
+In our implementation the shift happens within the first fifty periods, so the speculative
+phase is over before $t = 500$.
+
+The transient is evidently sensitive to details of the implementation in a way that the
+limit is not.
+
+The economics of the authors' observation is nevertheless worth stating, because it bears on
+how we should read Economy A2.
+
+There, one might suspect that the fundamental equilibrium was selected merely because myopic
+agents refuse expensive goods from the very beginning.
+
+In Economy B the system *starts* in the speculative pattern and *moves away from it* as
+strengths accumulate, which suggests that the selection of the fundamental equilibrium is
+something the learning dynamics do, not just an artifact of where they start.
+
+```{code-cell} ipython3
+plot_flows(sim_b1, title="Economy B.1: discovered exchange pattern")
+```
+
+(mms_ga)=
+## The genetic algorithm
+
+Complete enumeration is only feasible because the Kiyotaki-Wright state space is tiny.
+
+In any larger problem the list of all conceivable rules is far too long to carry around, and
+an agent must instead work with a limited population of rules that is continually revised.
+
+Marimon, McGrattan and Sargent add four operations for this case.
+
+**Creation** and **diversification** we have already implemented; they handle unforeseen
+states and guarantee that both actions get tried.
+
+For the five-good economy we shall use a variant of diversification that clones the winning
+rule with the opposite action, thereby preserving the winner's level of generality rather
+than manufacturing a fully specific rule.
+
+```{code-cell} ipython3
+def diversify_clone(rules, matches, winner, ufitness=0.5):
+ "Copy the winner with the opposite action over a rarely used rule."
+ if len(set(rules.action[matches])) > 1:
+ return
+ losers = np.flatnonzero(rules.used / (rules.used.max() + 1) < ufitness)
+ if losers.size == 0:
+ return
+ j = losers[np.argmin(rules.strength[losers])]
+ rules.replace(j, rules.cond[winner].copy(), 1 - rules.action[winner],
+ rules.strength[winner], used=rules.used[winner])
+```
+
+**Specialization** turns wildcards into specific bits, so that a rule which has been serving
+several states can split off a sharper version of itself.
+
+It is called with a probability $f_s(t) = 1/(2\sqrt{t})$ that declines over time:
+experimentation is cheap early and expensive late.
+
+```{code-cell} ipython3
+def specialize_all(rules, t, rng):
+ "Replace each wildcard by a bit with probability 1 / (2 sqrt(t))."
+ hit = (rules.cond == WILD) & (rng.random(rules.cond.shape)
+ < 1.0 / (2.0 * np.sqrt(t)))
+ if hit.any():
+ rules.cond[hit] = rng.integers(0, 2, size=hit.sum())
+
+
+def specialize_winner(rules, winner, pmutation, rng, ufitness=0.5):
+ "Plant a sharpened copy of a heavily used winner over a rarely used rule."
+ cond = rules.cond[winner]
+ if not np.any(cond == WILD):
+ return
+ if rules.used[winner] / (rules.used.max() + 1) <= ufitness:
+ return
+ pick = (cond == WILD) & (rng.random(cond.shape) < pmutation)
+ losers = np.flatnonzero(rules.used / (rules.used.max() + 1) < ufitness)
+ if not pick.any() or losers.size == 0:
+ return
+ j = losers[np.argmin(rules.strength[losers])]
+ new = cond.copy()
+ new[pick] = rng.integers(0, 2, size=pick.sum())
+ rules.replace(j, new, rules.action[winner], rules.strength[winner],
+ used=rules.used[winner])
+```
+
+**Generalization** is the genetic algorithm proper.
+
+Rules that are weak or rarely used become candidates for replacement.
+
+Parents are drawn in two stages -- first a subset weighted by how often rules have been
+used, then a roulette wheel on strength within that subset -- so that a rule must be both
+successful and *relevant* to reproduce.
+
+Two children are formed by crossover and, in the `'ga3'` variant, mutation.
+
+Each child then displaces the rule it most closely resembles among the replaceable ones,
+a device known as *crowding* that preserves diversity by making children compete against
+their own kind.
+
+The `'ga4'` variant replaces single-point crossover with a *generalizing* crossover: inside
+a randomly drawn interval, positions where the parents disagree become wildcards.
+
+This is the operator described in section 6 of the paper and illustrated in its figure 5,
+and it manufactures general rules rather than recombining specific ones.
+
+```{code-cell} ipython3
+def roulette(weights, rng):
+ total = weights.sum()
+ if total <= 0:
+ return int(rng.integers(0, len(weights)))
+ return int(np.searchsorted(np.cumsum(weights), rng.random() * total))
+
+
+def crowding_victim(rules, cond, action, cankill, rng, crowd_factor, crowd_subpop):
+ "De Jong crowding: the child displaces the replaceable rule it resembles most."
+ size = max(1, int(crowd_subpop * len(cankill)))
+ best, best_sim = cankill[0], -1
+ for _ in range(crowd_factor):
+ pool = (cankill if size >= len(cankill)
+ else list(rng.choice(cankill, size=size, replace=False)))
+ cand = min(pool, key=lambda i: rules.strength[i])
+ sim = (np.count_nonzero(cond == rules.cond[cand])
+ + (action != rules.action[cand]))
+ if sim > best_sim:
+ best, best_sim = cand, sim
+ return best
+
+
+def genetic_algorithm(rules, rng, generalize=False, pcross=0.6, pmutation=0.01,
+ propselect=0.2, propused=0.7, crowd_factor=8,
+ crowd_subpop=0.5, uratio=(0.0, 0.2)):
+ n, L = rules.n, rules.length
+ if n < 4:
+ return
+
+ # rules that are weak or seldom used may be replaced
+ max_used = max(rules.used.max() + (1 if generalize else 0), 1)
+ cankill = list(np.flatnonzero((rules.strength < uratio[0]) |
+ (rules.used / max_used < uratio[1])))
+ if not cankill:
+ return
+
+ fitness = rules.strength - min(rules.strength.min(), 0.0) + 1e-6
+ n_pairs = min(max(1, round(propselect * n * 0.5)), (len(cankill) + 1) // 2)
+ n_called = int(propused * n)
+
+ for _ in range(n_pairs):
+ if not cankill:
+ break
+
+ # stage 1: pre-select a pool with probability proportional to usage
+ if n_called < n:
+ avail = rules.used.astype(float) + 1.0
+ pool = []
+ for _ in range(n_called):
+ if avail.sum() <= 0:
+ break
+ k = roulette(avail, rng)
+ pool.append(k)
+ avail[k] = 0.0
+ if len(pool) < 2:
+ pool = list(range(n))
+ else:
+ pool = list(range(n))
+ pool = np.array(pool)
+
+ # stage 2: roulette wheel on fitness within the pool
+ mum, dad = pool[roulette(fitness[pool], rng)], pool[roulette(fitness[pool], rng)]
+ kids = [rules.cond[mum].copy(), rules.cond[dad].copy()]
+ acts = [rules.action[mum], rules.action[dad]]
+ avg = 0.5 * (rules.strength[mum] + rules.strength[dad])
+
+ if generalize:
+ # two-point crossover in which disagreements become wildcards
+ lo, hi = np.sort(rng.integers(0, L + 1, size=2))
+ inside = rng.random() > 0.5
+ region = (np.arange(lo, hi) if inside else
+ np.concatenate([np.arange(0, lo), np.arange(hi, L)]))
+ for j in region:
+ a, b = kids[0][j], kids[1][j]
+ if a >= 0 and b >= 0 and a != b:
+ kids[0][j] = kids[1][j] = WILD
+ else:
+ # single-point crossover with ternary mutation
+ jc = 1 + int((L - 1) * rng.random()) if rng.random() < pcross else L
+ kids = [np.concatenate([rules.cond[mum][:jc], rules.cond[dad][jc:]]),
+ np.concatenate([rules.cond[dad][:jc], rules.cond[mum][jc:]])]
+ for k in range(2):
+ flip = rng.random(L) < pmutation
+ if flip.any():
+ shift = rng.integers(1, 3, size=flip.sum())
+ kids[k][flip] = ((kids[k][flip] + 1 + shift) % 3) - 1
+ if rng.random() < pmutation:
+ acts[k] = 1 - acts[k]
+
+ for k, parent in zip(range(2), (mum, dad)):
+ if not cankill:
+ break
+ j = crowding_victim(rules, kids[k], acts[k], cankill, rng,
+ crowd_factor, crowd_subpop)
+ rules.replace(j, kids[k], acts[k], avg,
+ used=rules.used[parent], traded=rules.traded[parent])
+ cankill.remove(j)
+```
+
+## Economy A1.2: learning from random rules
+
+Economy A1.2 has the same parameters as A1.1, but agents now begin with 72 exchange rules
+and 12 consumption rules drawn *at random*.
+
+Most of these rules are nonsense, and there is no guarantee that the population even
+contains the rules needed to support an equilibrium.
+
+The genetic operators must manufacture them.
+
+```{code-cell} ipython3
+economy_a12 = Economy(
+ name='A1.2',
+ produces=np.array([1, 2, 0]),
+ storage_costs=np.array([0.1, 1.0, 20.0]),
+ u=100.0,
+ method='ga3',
+)
+
+sim_a12 = Simulation(economy_a12, seed=2)
+sim_a12.run(2000, verbose=True)
+```
+
+```{code-cell} ipython3
+holdings(sim_a12, t=1000)
+```
+
+```{code-cell} ipython3
+holdings(sim_a12)
+```
+
+```{code-cell} ipython3
+plot_holdings(sim_a12, title="Economy A1.2")
+```
+
+The fundamental equilibrium is reached again, though it takes noticeably longer and the
+early transition is much less orderly than under complete enumeration.
+
+Let us see which rules survived.
+
+```{code-cell} ipython3
+strongest(sim_a12.agents[0].trade, n=6)
+```
+
+The population has converged onto near-copies of a single rule.
+
+`0110 -> 1` -- "holding good 2, meeting good 1, trade" -- is exactly the rule that complete
+enumeration selected in Economy A1.1, but here the genetic algorithm had to build it, and
+several copies of it now occupy the population.
+
+The other entries are its offspring: `0111 -> 1` has the same own-holding condition but a
+partner condition of `11`, which is not the code of any of the three goods, so that rule can
+never fire.
+
+It carries a high strength and a large usage count only because children inherit both from
+their parents.
+
+```{code-cell} ipython3
+plot_flows(sim_a12, title="Economy A1.2: discovered exchange pattern")
+```
+
+```{warning}
+Convergence is not guaranteed run by run.
+
+With random initial rules the population sometimes fails to manufacture the rules needed for
+the fundamental equilibrium within 2000 periods, and the economy settles into a pattern with
+little trade.
+
+Try changing the seed above to see this.
+
+The paper is explicit that the algorithm has "too little experimentation" and that improving it
+is unfinished business.
+```
+
+## Economy C: fiat money
+
+Now add a fourth object, good 0, which
+
+* costs nothing to store, $s_0 = 0$;
+* yields no utility to anybody; and
+* cannot be consumed.
+
+It is intrinsically worthless.
+
+It is introduced by handing 48 units of it to 48 randomly chosen agents at $t = 0$, and
+commodity storage costs are raised to $s = (9, 14, 29)$ so that no commodity is nearly as
+cheap to store as money.
+
+If a good like this circulates, it can only be because agents have discovered that other
+agents will take it.
+
+```{code-cell} ipython3
+economy_c = Economy(
+ name='C',
+ produces=np.array([1, 2, 0]),
+ storage_costs=np.array([9.0, 14.0, 29.0, 0.0]), # last good is fiat money
+ u=100.0,
+ method='ga3',
+ n_trade_rules=150,
+ n_consume_rules=20,
+ b_consume=(0.025, 0.25),
+ n_fiat=48,
+ start='production',
+)
+
+sim_c = Simulation(economy_c, seed=2)
+sim_c.run(2000, verbose=True)
+```
+
+```{code-cell} ipython3
+holdings(sim_c, t=750)
+```
+
+```{code-cell} ipython3
+holdings(sim_c, t=1250)
+```
+
+Compare with the paper's table at $t = 1250$, where the columns are (good 1, good 2, good 3,
+fiat): $(0, 0.54, 0, 0.46)$ for type 1, $(0.18, 0, 0.53, 0.28)$ for type 2, and
+$(0.77, 0, 0, 0.21)$ for type 3.
+
+Every type holds fiat money a substantial fraction of the time.
+
+```{code-cell} ipython3
+plot_holdings(sim_c, title="Economy C: fiat money")
+```
+
+```{code-cell} ipython3
+plot_flows(sim_c, title="Economy C: discovered exchange pattern")
+```
+
+```{code-cell} ipython3
+winning_actions(sim_c)
+```
+
+The arrows into and out of the fiat node show money changing hands in both directions for
+every type: agents give up commodities to acquire it and give it up to acquire the commodity
+they consume.
+
+Nothing in the environment told them to do this.
+
+Each agent discovered only that rules which end up holding the costless object earn more
+than rules that do not -- and, because everyone was discovering this at once, the belief
+became self-confirming.
+
+This is a genuinely social arrangement built by agents who individually understand nothing
+about it.
+
+## Economy D: five goods, five types
+
+The last economy has five types and five goods, with production
+
+| type $i$ | produces | consumes |
+|---|---|---|
+| 1 | good 3 | good 1 |
+| 2 | good 4 | good 2 |
+| 3 | good 5 | good 3 |
+| 4 | good 1 | good 4 |
+| 5 | good 2 | good 5 |
+
+storage costs $s = (1, 4, 9, 16, 30)$ and $u_i = 200$.
+
+Goods are now encoded in three bits, and each agent carries 180 exchange rules and 20
+consumption rules -- a small fraction of the possible rules, so the genetic algorithm is
+essential.
+
+The authors emphasize that they had *no analytical characterization of equilibrium* for
+this economy before running it.
+
+The simulation is being used as a device to *discover* what an equilibrium might look like,
+which could then be verified analytically.
+
+```{code-cell} ipython3
+economy_d = Economy(
+ name='D',
+ produces=np.array([2, 3, 4, 0, 1]), # type 1 -> good 3, type 2 -> good 4, ...
+ storage_costs=np.array([1.0, 4.0, 9.0, 16.0, 30.0]),
+ u=200.0,
+ method='ga4',
+ n_trade_rules=180,
+ n_consume_rules=20,
+ start='production',
+)
+
+sim_d = Simulation(economy_d, seed=3)
+sim_d.run(2000, verbose=True)
+```
+
+```{code-cell} ipython3
+holdings(sim_d, t=500)
+```
+
+```{code-cell} ipython3
+holdings(sim_d)
+```
+
+```{code-cell} ipython3
+plot_holdings(sim_d, title="Economy D: five goods, five types")
+```
+
+Two features stand out.
+
+First, the diagonal is zero: no type ever ends a period holding the good it consumes,
+because it consumes it.
+
+Second, each type accumulates its own production good together with goods that are *cheaper*
+to store, never more expensive.
+
+Type 3, for instance, produces good 5, the most expensive good in the economy, and holds it
+only part of the time, having traded some of it away for the much cheaper good 2.
+
+Type 4 produces good 1, the cheapest good of all, and simply holds it.
+
+Let us classify the trades that actually take place.
+
+```{code-cell} ipython3
+def trade_composition(sim, window=200):
+ """
+ Classify realized trades by what the agent acquires: its own consumption
+ good, a good that is cheaper to store than the one given up, or neither.
+ """
+ econ = sim.econ
+ f = np.array(sim.exch_hist)[-window:].sum(axis=0)
+ s = econ.storage_costs
+ counts = {'own consumption good': 0.0, 'a cheaper good': 0.0, 'neither': 0.0}
+ for i in range(econ.n_types):
+ for j in range(econ.n_goods):
+ for k in range(econ.n_goods):
+ if j == k:
+ continue
+ key = ('own consumption good' if k == i else
+ 'a cheaper good' if s[k] < s[j] else 'neither')
+ counts[key] += f[i, j, k]
+ total = sum(counts.values())
+ return pd.Series({k: round(v / total, 3) for k, v in counts.items()},
+ name='share of realized trades')
+
+
+trade_composition(sim_d)
+```
+
+Roughly two thirds of realized trades acquire either the agent's own consumption good or a
+good that is cheaper to store, which is the pattern the paper describes.
+
+The remaining third deserves a comment, because it is not evidence against that pattern.
+
+Consider a type 1 agent, who produces good 3, swapping good 3 for the far more expensive
+good 5.
+
+He then consumes good 5 -- getting no utility from it -- and produces good 3 again, so he
+ends the period holding good 3 and paying $s_3$, exactly as he would have had he refused.
+
+Such trades are payoff-neutral, so nothing in the accounting system pushes the rules that
+generate them out of the population.
+
+```{code-cell} ipython3
+plot_flows(sim_d, window=200, cutoff=0.03,
+ title="Economy D: discovered exchange pattern")
+```
+
+The paper summarizes what it sees in these patterns:
+
+> From the simulation results, we can see that the trading patterns nearly seem to describe
+> a fundamental equilibrium in which agents are only willing to trade for commodities of
+> lower cost than the one currently in storage, except that they always accept the commodity
+> of their type. Some speculative moves can be detected.
+
+An example of such a speculative move is a type 2 agent accepting good 3 for good 1, not
+because good 3 is cheap but because type 3 agents will take it in exchange for good 2.
+
+## Concluding remarks
+
+Marimon, McGrattan and Sargent's agents know very little.
+
+They do not know their own utility functions, they do not know storage costs, they do not
+know the distribution of goods across the population, and they certainly do not solve
+dynamic programs.
+
+They recognize utility when they experience it and costs when they bear them, and they keep
+running averages.
+
+Out of this the following emerges:
+
+* **Nash-Markov behavior is learnable.** In most of the economies simulated, holdings and
+ trading patterns converge to a stationary Nash equilibrium of the Kiyotaki-Wright model.
+* **Learning selects among equilibria.** Where the rational-expectations model admits both a
+ fundamental and a speculative equilibrium, the classifier systems always found the
+ fundamental one.
+ - Economy A2 shows they found it even where theory says it should not be an equilibrium at
+ all for patient agents.
+ - Economy B shows this is not simply early myopia, since that economy moves *away* from a
+ speculative pattern as it learns.
+* **Institutions can be discovered.** Economy C's agents built a fiat monetary system out of
+ nothing but their own experience of storage costs.
+* **The method scales.** Economy D produced a credible description of equilibrium in a model
+ its authors had not solved.
+
+The paper is candid about what is missing.
+
+There are no convergence theorems, only a sketch of how stochastic approximation arguments
+might supply them; and the authors judge their genetic algorithm to provide "too little
+experimentation" -- which is precisely why the speculative equilibrium never appears.
+
+That diagnosis, that an adaptive system's selection among equilibria is governed by how much
+it explores and when, has proved durable.
+
+## Exercises
+
+```{exercise-start}
+:label: mms_ex1
+```
+
+Economy A2.2 is Economy A2 -- with $u_i = 500$ -- but started from randomly generated rules
+and evolved with the `'ga3'` genetic algorithm, exactly as A1.2 relates to A1.1.
+
+The paper's summary table lists its equilibrium type as speculative, and the authors report
+that after 1000 iterations the economy had not converged, with trading patterns closer to
+the fundamental equilibrium at $t = 1000$ than at $t = 500$.
+
+Simulate it for 2000 periods and report holdings at $t = 500$, $t = 1000$ and $t = 2000$.
+
+Does the extra experimentation supplied by the genetic algorithm produce speculation?
+
+```{exercise-end}
+```
+
+```{solution-start} mms_ex1
+:class: dropdown
+```
+
+```{code-cell} ipython3
+economy_a22 = Economy(
+ name='A2.2',
+ produces=np.array([1, 2, 0]),
+ storage_costs=np.array([0.1, 1.0, 20.0]),
+ u=500.0,
+ method='ga3',
+)
+
+sim_a22 = Simulation(economy_a22, seed=3)
+sim_a22.run(2000)
+
+for t in (500, 1000, 2000):
+ print(f"\nt = {t}")
+ print(holdings(sim_a22, t=t))
+```
+
+```{code-cell} ipython3
+plot_holdings(sim_a22, title="Economy A2.2")
+```
+
+The economy again settles on the fundamental pattern: type 1 holds good 2, type 3 holds
+good 1, and type 2 alternates between goods 1 and 3.
+
+Type 1 agents are not holding good 3, so they are not speculating.
+
+Randomizing the initial rules and letting the genetic algorithm run does not by itself
+generate enough experimentation to sustain speculation -- which is the paper's own
+conclusion.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: mms_ex2
+```
+
+Economy B.2 is Economy B started from random rules with the `'ga3'` genetic algorithm.
+
+The paper reports that it "had not converged after 2000 periods" but was "moving towards the
+fundamental equilibrium", with holdings at $t = 2000$ of $(0, 0.354, 0.646)$ for type 1,
+$(0.996, 0, 0.004)$ for type 2 and $(0.268, 0.732, 0)$ for type 3.
+
+Simulate it and compare with the complete-enumeration Economy B.1 that we ran above.
+
+```{exercise-end}
+```
+
+```{solution-start} mms_ex2
+:class: dropdown
+```
+
+```{code-cell} ipython3
+economy_b2 = Economy(
+ name='B.2',
+ produces=np.array([2, 0, 1]),
+ storage_costs=np.array([1.0, 4.0, 9.0]),
+ u=100.0,
+ method='ga3',
+)
+
+sim_b2 = Simulation(economy_b2, seed=1)
+sim_b2.run(2000)
+
+for t in (500, 1000, 2000):
+ print(f"B.2 at t = {t}")
+ print(holdings(sim_b2, t=t))
+ print()
+print("B.1 (complete enumeration) at t = 1000, for comparison")
+print(holdings(sim_b1))
+```
+
+```{code-cell} ipython3
+plot_holdings(sim_b2, title="Economy B.2")
+```
+
+Follow the two types that move.
+
+Type 2 agents are still split at $t = 500$ and have locked onto good 1 by $t = 1000$.
+
+Type 3 agents start out mostly holding their own production good 2 and shift a substantial
+part of their holdings into the cheaper good 1 over the same interval, which is the movement
+toward the fundamental equilibrium that the paper describes.
+
+After that the type 3 split stops moving in one direction and rattles around a half-and-half
+mix from one reading to the next, so this economy has not converged in the sense that B.1 has.
+
+That is the paper's own verdict on it.
+
+Compared with the complete-enumeration run, the genetic algorithm gets type 1 and type 2 to
+much the same place and leaves type 3 less far along.
+
+Building the needed rules from random material costs time that the enumerated economy never
+had to spend.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: mms_ex3
+```
+
+The bid functions $b_1(e) = b_{11} + b_{12}\sigma_e$ favor specific rules over general ones,
+since $\sigma_e$ falls with the number of wildcards.
+
+What happens if this tilt is removed?
+
+Rerun Economy A1.1 with $b_{12} = 0$, holding $b_{11} + b_{12}$ fixed at $0.05$, and compare
+the number of wildcards in the winning rules with those in the baseline.
+
+A single run will not settle the question, so average over several seeds.
+
+```{exercise-end}
+```
+
+```{solution-start} mms_ex3
+:class: dropdown
+```
+
+```{code-cell} ipython3
+economy_flat = Economy(
+ name='A1.1 with no specificity premium',
+ produces=np.array([1, 2, 0]),
+ storage_costs=np.array([0.1, 1.0, 20.0]),
+ u=100.0,
+ method='enumerate',
+ b_trade=(0.05, 0.0),
+)
+
+sim_flat = Simulation(economy_flat, seed=42)
+sim_flat.run(1000)
+
+print(holdings(sim_flat))
+```
+
+The holdings are unchanged: the fundamental equilibrium is reached either way.
+
+What changes is the *kind of rule* that gets there.
+
+```{code-cell} ipython3
+def wildcards_in_winners(sim):
+ "Average number of wildcards in the exchange rule that wins in each state."
+ econ = sim.econ
+ counts = []
+ for A in sim.agents:
+ for j in range(econ.n_goods):
+ for k in range(econ.n_goods):
+ w, _ = auction(A.trade, np.concatenate([econ.code(j), econ.code(k)]))
+ if w >= 0:
+ counts.append(np.count_nonzero(A.trade.cond[w] == WILD))
+ return np.mean(counts)
+
+
+def average_wildcards(b_trade, seeds=range(8)):
+ out = []
+ for seed in seeds:
+ econ = Economy(name='sweep', produces=np.array([1, 2, 0]),
+ storage_costs=np.array([0.1, 1.0, 20.0]), u=100.0,
+ method='enumerate', b_trade=b_trade)
+ sim = Simulation(econ, seed=seed)
+ sim.run(1000)
+ out.append(wildcards_in_winners(sim))
+ return np.array(out)
+
+
+base = average_wildcards((0.025, 0.025))
+flat = average_wildcards((0.05, 0.0))
+
+pd.DataFrame({"baseline $b_{12} = 0.025$": base.round(2),
+ "no premium $b_{12} = 0$": flat.round(2)},
+ index=pd.Index(range(8), name="seed"))
+```
+
+```{code-cell} ipython3
+print(f"mean wildcards, baseline : {base.mean():.3f}")
+print(f"mean wildcards, no specificity bid : {flat.mean():.3f}")
+print(f"higher without the premium in {np.sum(flat > base)} of {len(base)} seeds")
+```
+
+Removing the specificity premium raises the average generality of the winning rules, but the
+effect is modest and does not show up in every run.
+
+That is worth knowing rather than glossing over.
+
+With a complete enumeration the specific rules are all present from the start and accumulate
+strength on their own, so the bid premium is only one of the forces tilting the auction toward
+them; the counter is a genuine tendency visible in the average rather than a law obeyed run by
+run.
+
+The tilt matters more where it is harder to see.
+
+It is the mechanism the paper blames for the failure of Economy A2: general consumption
+classifiers cannot tell stored goods apart, so they cannot transmit to the exchange classifiers
+the information that would make speculation profitable.
+
+```{solution-end}
+```
diff --git a/lectures/olg_adaptive_money.md b/lectures/olg_adaptive_money.md
new file mode 100644
index 000000000..8331ba074
--- /dev/null
+++ b/lectures/olg_adaptive_money.md
@@ -0,0 +1,1460 @@
+---
+jupytext:
+ text_representation:
+ extension: .md
+ format_name: myst
+ format_version: 0.13
+ jupytext_version: 1.17.1
+kernelspec:
+ display_name: Python 3 (ipykernel)
+ language: python
+ name: python3
+---
+
+(olg_adaptive_money)=
+```{raw} jupyter
+
+```
+
+# Adaptive Agents in an Overlapping Generations Monetary Economy
+
+```{index} single: Bounded Rationality; Overlapping Generations
+```
+
+```{contents} Contents
+:depth: 2
+```
+
+## Overview
+
+{doc}`bounded_rationality` presented some models of monetary economies with multiple rational expectations equilibria.
+
+This lecture takes one of them — Samuelson's overlapping generations model of
+fiat money, used by Bryant, Wallace and others to study inflationary finance — and replaces Samuelson's agents who forecast according to the equilibrium law of motion for the price level with ''adaptive'' ones who do not.
+
+We do this twice, for two different purposes.
+
+**Part 1** puts a *stochastic* government deficit into the model and asks whether successive
+generations, groping their way by trial and error, can somehow as a society converge to a rational expectations
+equilibrium.
+
+They can.
+
+Each generation observes what happened to its predecessors and adjusts their saving decision in
+a utility-improving direction, using a **Robbins–Monro** recursion.
+
+Given enough
+time the economy converges to the stationary equilibrium we compute independently.
+
+This exercise also shows how sensitive learning is to the complexity of what must be learned:
+a deficit state that occurs rarely is learned about slowly, because the observations arrive
+slowly.
+
+**Part 2** removes the randomness and turns to a sharper question.
+
+With a *constant* deficit the model has **two** stationary equilibria, one with low inflation
+and one with high inflation.
+
+The low-inflation
+equilibrium Pareto-dominates the other.
+
+Under the rational expectations dynamics, the low-inflation equilibrium is **unstable** and
+the high-inflation one is stable: theory pushes the economy toward the bad outcome.
+
+Under least squares learning, the stability is **reversed**.
+
+And when {cite:t}`MarimonSunder1993` ran the economy as a laboratory experiment with paid
+human subjects, the subjects behaved like the adaptive model, not like the rational
+expectations model.
+
+That is the strongest evidence in {cite:t}`Sargent1993` that adaptive dynamics are doing real
+work as a selection device, tempered by a counterexample, due to Benjamin Bental, warning
+against concluding that adaptation reliably selects the *good* equilibrium.
+
+We then close with a **third application**, in which the adaptive agent is a government learning
+a Phillips curve.
+
+Now least squares learning selects the *bad*, high-inflation equilibrium, but a government
+that doubts its own model can **escape** toward the good one, a phenomenon that seeded Sargent's
+later work on *The Conquest of American Inflation*.
+
+Let's start with some imports.
+
+```{code-cell} ipython3
+import numpy as np
+import pandas as pd
+import quantecon as qe
+import matplotlib.pyplot as plt
+from scipy.optimize import fsolve
+```
+
+## The environment
+
+The economy consists of overlapping generations of two-period-lived agents.
+
+At each date $t \geq 1$, $N$ identical agents are born, endowed with $w_1$ units of a single
+consumption good when young and $w_2$ units when old.
+
+A young agent's preferences over lifetime consumption $(c_1, c_2)$ are ordered by the
+expected value of $u(c_1) + u(c_2)$.
+
+Throughout we use logarithmic utility, $u(c) = \ln c$.
+
+A government prints currency to finance expenditures $G_t$, subject to the budget constraint
+
+```{math}
+:label: govt_budget
+
+G_t = \frac{H_t - H_{t-1}}{p_t},
+```
+
+where $H_t$ is the stock of currency carried by the young at $t$ into $t+1$, and $p_t$ is the
+price level.
+
+Currency is the only store of value, so a young agent's saving $s_t$ is entirely in the form
+of real balances, and market clearing requires
+
+```{math}
+:label: market_clearing
+
+\frac{H_t}{p_t} = s_t N .
+```
+
+Write $R_t = p_t / p_{t+1}$ for the gross rate of return on currency between $t$ and $t+1$.
+
+Combining {eq}`govt_budget` and {eq}`market_clearing` gives a relation we will use constantly:
+the return on currency is determined entirely by saving behavior and the deficit,
+
+```{math}
+:label: return_from_saving
+
+R_t = \frac{N s_{t+1} - G_{t+1}}{N s_t} .
+```
+
+Currency holds its value only if the next generation is willing to absorb the outstanding
+stock plus the new issue.
+
+```{note}
+{eq}`return_from_saving` is worth pausing on, because it is what makes this a
+*self-referential* system, the property studied at length in {doc}`ls_learning`.
+
+The return that today's young earn on their saving depends on how much tomorrow's young choose
+to save, which in turn depends on what tomorrow's young expect to earn.
+
+Nobody's beliefs are about an exogenous process.
+```
+
+## Part 1: a stochastic deficit
+
+Now let government expenditures follow a Markov chain,
+
+$$
+\pi(i, j) = \operatorname{Prob}\{G_{t+1} = \bar G_j \mid G_t = \bar G_i\},
+$$
+
+over a finite list of levels $G = [\bar G_1, \ldots, \bar G_n]$.
+
+### Stationary rational expectations equilibrium
+
+In a stationary equilibrium, saving depends on the current deficit state, so it is described
+by an $n \times 1$ vector $s = [s_1, \ldots, s_n]$, and the return on currency by an
+$n \times n$ matrix $R(i,j)$.
+
+A young agent who observes $G_t = \bar G_i$ chooses $s_i$ to maximize
+
+$$
+V(s_i) = u(w_1 - s_i) + \sum_j u\bigl(w_2 + s_i R(i,j)\bigr) \pi(i,j),
+$$
+
+with first-order condition
+
+```{math}
+:label: olg_foc
+
+u'(w_1 - s_i) = \sum_{j=1}^n u'\bigl(w_2 + s_i R(i,j)\bigr) R(i,j) \pi(i,j).
+```
+
+Substituting {eq}`return_from_saving` — which in the stationary equilibrium reads
+$R(i,j) = (s_j N - \bar G_j)/(s_i N)$ — turns {eq}`olg_foc` into a system of $n$ nonlinear
+equations in the saving rates alone:
+
+```{math}
+:label: olg_equilibrium
+
+u'(w_1 - s_i)
+= \sum_{j=1}^n u'\left(w_2 + \frac{s_j N - \bar G_j}{N}\right)
+ \cdot \frac{s_j N - \bar G_j}{s_i N} \cdot \pi(i,j) .
+```
+
+A stationary equilibrium exists if and only if {eq}`olg_equilibrium` has a solution with
+$s_i \in (0, w_1)$ for every $i$.
+
+```{code-cell} ipython3
+w1, w2, N = 20.0, 10.0, 1.0
+
+def u_prime(c):
+ return 1 / c # log utility
+
+def equilibrium_residual(s, G, P):
+ "Left minus right side of the equilibrium condition, one entry per deficit state."
+ n = len(s)
+ out = np.empty(n)
+ for i in range(n):
+ rhs = sum(u_prime(w2 + (s[j]*N - G[j])/N)
+ * (s[j]*N - G[j])/(s[i]*N)
+ * P[i, j] for j in range(n))
+ out[i] = u_prime(w1 - s[i]) - rhs
+ return out
+
+def stationary_equilibrium(G, P, guess=4.0):
+ s = fsolve(equilibrium_residual, np.full(len(G), guess), args=(G, P))
+ R = np.array([[(s[j]*N - G[j])/(s[i]*N) for j in range(len(s))]
+ for i in range(len(s))])
+ return s, R
+```
+
+Following {cite:t}`Sargent1993`, we study two economies that differ only in the process for
+government expenditures, with $w_1 = 20$, $w_2 = 10$ and $N = 1$, so that $\bar G$ is measured
+per young person.
+
+In the first, the deficit is always zero and the chain is a fair coin.
+
+```{code-cell} ipython3
+G1, P1 = np.array([0.0, 0.0]), np.full((2, 2), 0.5)
+s1, R1 = stationary_equilibrium(G1, P1)
+print(f"saving rates {np.round(s1, 4)}")
+print(f"returns\n{np.round(R1, 4)}")
+```
+
+With no deficit to finance, currency simply holds its value: agents save $5$ in either state
+and $R \equiv 1$.
+
+In the second economy the deficit is $0.8$ in state 1 and zero in state 2, and the chain is
+persistent.
+
+```{code-cell} ipython3
+G2 = np.array([0.8, 0.0])
+P2 = np.array([[0.75, 0.25],
+ [0.50, 0.50]])
+s2, R2 = stationary_equilibrium(G2, P2)
+print(f"saving rates {np.round(s2, 4)} (Sargent reports 4.211, 4.364)")
+print(f"returns\n{np.round(R2, 4)}")
+print("Sargent reports [[0.81, 1.0362], [0.7817, 1.00]]")
+```
+
+Both match the values reported in the text.
+
+Notice the pattern in $R$: the return on currency is poor whenever the government is about to
+run a deficit (the first column), because new issue dilutes the outstanding stock.
+
+### Learning by successive generations
+
+Now withdraw the equilibrium from the agents.
+
+They are still assumed to know their own utility function and to remember the experience of
+earlier agents like themselves, but they are *not* told the distribution of returns, which is
+exactly the object that {eq}`olg_foc` requires.
+
+Instead, each generation adjusts its predecessors' saving decision in the direction that
+*realized* experience suggests would have been better.
+
+The realized lifetime utility of an agent who saved $s$ and earned $R$ is
+
+$$
+U(s) = u(w_1 - s) + u(w_2 + s R),
+$$
+
+with derivatives
+
+$$
+U'(s) = -u'(w_1 - s) + u'(w_2 + sR) R,
+\qquad
+U''(s) = u''(w_1 - s) + u''(w_2 + sR) R^2 .
+$$
+
+The connection to equilibrium is that in a rational expectations equilibrium
+$V'(s_i) = E_t U'(s_i)$ and $V''(s_i) = E_t U''(s_i)$, where $E_t$ conditions on
+$G_t = \bar G_i$.
+
+So an agent who could compute $E_t U'$ would just set it to zero and be done.
+
+Our agents cannot, and must estimate it from experience instead.
+
+They use a **Robbins–Monro** algorithm, applied state by state.
+
+Let $\tau_i$ count the number of times the deficit state $\bar G_i$ has been visited so far,
+and let $\gamma_\tau$ be a decreasing gain sequence.
+
+The rules are
+
+```{math}
+:label: robbins_monro
+
+\begin{aligned}
+M(i, \tau_i + 1) &= M(i, \tau_i) + \gamma_{\tau_i}\bigl(U''(s(i, \tau_i)) - M(i, \tau_i)\bigr), \\
+s(i, \tau_i + 1) &= s(i, \tau_i) - \gamma_{\tau_i} M(i, \tau_i + 1)^{-1} U'(s(i, \tau_i)) .
+\end{aligned}
+```
+
+$M$ accumulates a running estimate of $E_t U''$, and the saving rule takes a Newton step
+against it.
+
+If $\tau_i \to \infty$ then $M(i, \tau_i) \to E_i U''$ and $s(i, \tau_i)$ approaches a solution
+of $E_i U'(s_i) = 0$, which is exactly the equilibrium condition {eq}`olg_foc`.
+
+Two features of the setup deserve comment.
+
+**Two classes of agent.** To evaluate a saving decision we must wait until *two* periods of
+that agent's consumption are known, so the population is split into an "odd" and an "even"
+subsequence.
+
+Odd agents reset their rule in odd periods and learn only from earlier odd agents; even agents
+likewise.
+
+This is the same device used in the laboratory experiments discussed in Part 2.
+
+**A projection facility.** Nothing in {eq}`robbins_monro` prevents a Newton step from pushing
+saving below the deficit, at which point {eq}`market_clearing` has no positive price level and
+the economy ceases to exist.
+
+We therefore confine $s$ to a region where an equilibrium price level exists.
+
+Devices of this kind are needed for the convergence theorems too; {doc}`ls_learning` discusses
+the projection facility in detail.
+
+```{code-cell} ipython3
+def U_prime(s, R):
+ return -1/(w1 - s) + R/(w2 + s*R)
+
+def U_double(s, R):
+ return -1/(w1 - s)**2 - R**2/(w2 + s*R)**2
+
+
+def simulate(G, P, T=50_000, s0=None, seed=0, tau0=20, band=0.4):
+ """
+ Overlapping generations learning a state-contingent saving rule.
+
+ Returns the history of saving rules (shape (T, 2, n): time, class, state)
+ and the realized returns.
+ """
+ rng = np.random.default_rng(seed)
+ n = len(G)
+ s = np.array(s0, float)
+ M = np.array([[U_double(s[j, k], 1.0) for k in range(n)] for j in range(2)])
+ τ = np.zeros((2, n), int)
+ floor = G/N + band # projection facility: saving must cover the deficit
+
+ hist = np.empty((T, 2, n))
+ R_hist = np.full(T, np.nan)
+ i, prev = 0, None
+
+ for t in range(T):
+ j = t % 2 # which class is young this period
+ s_t = s[j, i]
+
+ if prev is not None: # last period's young now learn their return
+ j_p, i_p, s_p = prev
+ R = (N*s_t - G[i]) / (N*s_p) # the return on currency
+ R_hist[t-1] = R
+ τ[j_p, i_p] += 1
+ g = 1 / (τ[j_p, i_p] + tau0)
+ M[j_p, i_p] += g * (U_double(s_p, R) - M[j_p, i_p])
+ step = g * U_prime(s_p, R) / M[j_p, i_p]
+ s[j_p, i_p] = np.clip(s[j_p, i_p] - step, floor[i_p], w1 - 0.4)
+
+ prev = (j, i, s_t)
+ hist[t] = s
+ i = rng.choice(n, p=P[i])
+
+ return hist, R_hist
+```
+
+```{note}
+The simulation never computes a price level.
+
+Because {eq}`return_from_saving` expresses the return on currency purely in terms of saving
+rates and the deficit, the *real* side of the model is self-contained.
+
+This is more than a convenience: the nominal money stock in these economies grows without bound
+whenever the government runs a deficit, so a simulation that tracked $H_t$ and $p_t$ directly
+would overflow long before the learning converged.
+```
+
+### Do they find it?
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Saving rates converge in both economies"
+ name: fig-olg-saving
+---
+start = [[8.0, 2.0],
+ [2.0, 8.0]] # deliberately far from equilibrium, and asymmetric
+
+hist1, _ = simulate(G1, P1, s0=start)
+hist2, _ = simulate(G2, P2, s0=start)
+
+fig, axes = plt.subplots(1, 2, figsize=(11, 4), sharex=True)
+for ax, hist, s_star, title in [(axes[0], hist1, s1, "Economy 1: $\\bar G = (0, 0)$"),
+ (axes[1], hist2, s2, "Economy 2: $\\bar G = (0.8, 0)$")]:
+ for k, colour in enumerate(['C0', 'C1']):
+ ax.plot(hist[:, 0, k], color=colour, lw=1, label=f"state {k+1}, even")
+ ax.plot(hist[:, 1, k], color=colour, lw=1, ls=':', label=f"state {k+1}, odd")
+ ax.axhline(s_star[k], color='k', lw=0.8, ls='--')
+ ax.set_xscale('log')
+ ax.set_xlabel("$t$ (log scale)")
+ ax.set_title(title)
+axes[0].set_ylabel("saving rate")
+axes[0].legend(frameon=False, fontsize=8)
+plt.tight_layout()
+plt.show()
+```
+
+Both economies converge, and both classes of agent converge to the same rule even though
+neither learns from the other.
+
+The dashed black lines are the equilibrium saving rates computed earlier from
+{eq}`olg_equilibrium`, objects that no agent in the simulation has access to.
+
+```{code-cell} ipython3
+rows = []
+for T in (5_000, 50_000, 200_000):
+ h1, _ = simulate(G1, P1, T=T, s0=start)
+ h2, _ = simulate(G2, P2, T=T, s0=start)
+ rows.append([T, *h1[-1].mean(axis=0), *h2[-1].mean(axis=0)])
+
+pd.DataFrame(rows, columns=["T", "econ 1 state 1", "econ 1 state 2",
+ "econ 2 state 1", "econ 2 state 2"]).set_index("T").round(4)
+```
+
+```{code-cell} ipython3
+print(f"rational expectations: economy 1 = {np.round(s1, 4)}, economy 2 = {np.round(s2, 4)}")
+```
+
+Convergence is slow — that is what a $1/\tau$ gain buys you — but it is convergence, and to
+the right place.
+
+### How much do the agents have to be told?
+
+The specification above lets agents learn a *separate* saving rate for each level of the
+deficit.
+
+In effect they are learning the policy function $s = f(G)$ **non-parametrically**, state by
+state.
+
+That is feasible here because there are only two states.
+
+It has two drawbacks that Sargent emphasizes.
+
+First, states that are rarely visited are learned about slowly, because the observations arrive
+slowly.
+
+{ref}`olg_ex1` makes this concrete.
+
+Second, when the number of states is large, one parameter per state becomes unmanageable.
+
+An econometrician's response would be to impose a **parametric** form $s = f(G, \theta)$ with
+$\theta$ of low dimension and use every observation to estimate it, replacing
+{eq}`robbins_monro` with a recursion in $\theta$ driven by $\partial U/\partial \theta$.
+
+This buys speed at the cost of *approximation*: a parametric learning scheme can converge to a
+rational expectations equilibrium only if some member of the family $f(\cdot, \theta)$ supports
+one.
+
+Otherwise the best it can do is converge to an approximate equilibrium.
+
+```{note}
+This is the point at which models of learning and algorithms for *computing* equilibria become
+hard to tell apart.
+
+Marcet's method of parameterized expectations posits a parametric form for a conditional
+expectation, simulates, regresses realized outcomes on the parametric form to update the
+coefficients, and iterates.
+
+Written recursively, that algorithm is a nonlinear-least-squares recursion of the same shape as
+{eq}`robbins_monro`.
+
+Sargent's summary: "Learning algorithms and equilibrium computation algorithms look like each
+other."
+
+An equilibrium computation is a centralized learning algorithm run by the modeller; a learning
+economy is a decentralized equilibrium computation run by the agents.
+```
+
+## Part 2: a constant deficit and two steady states
+
+Now strip out the randomness.
+
+Let the government finance a *constant* deficit $G$ per young person, and let utility be
+$\ln c_1 + \ln c_2$ as before.
+
+With perfect foresight over the price level, the young save
+
+```{math}
+:label: saving_deterministic
+
+s_t = \frac{w_1 - w_2 \pi_t}{2},
+\qquad \pi_t \equiv \frac{p_{t+1}}{p_t},
+```
+
+where $\pi_t$ is the gross inflation rate, the reciprocal of the return on currency.
+
+The government's budget constraint in real terms is $h_t = h_{t-1}/\pi_{t-1} + G$, where
+$h_t = m_t / p_t$, and equilibrium requires $h_t = s_t$.
+
+Eliminating $h$ between these gives an autonomous difference equation in the inflation rate,
+
+```{math}
+:label: inflation_map
+
+\pi_{t+1} = g(\pi_t) \equiv A_1 - \frac{A_2}{\pi_t},
+\qquad
+A_1 = \frac{w_1 + w_2 - 2G}{w_2},
+\qquad
+A_2 = \frac{w_1}{w_2} .
+```
+
+Stationary equilibria are the fixed points of $g$, and since $g$ is a hyperbola there are
+generically two of them.
+
+We use the parameters of {cite:t}`MarimonSunder1993`'s "Economy 7C": $w_1 = 6$, $w_2 = 1$,
+$G = 1$.
+
+```{code-cell} ipython3
+w1_d, w2_d, G_d = 6.0, 1.0, 1.0
+A1 = (w1_d + w2_d - 2*G_d) / w2_d
+A2 = w1_d / w2_d
+
+def g_map(π):
+ return A1 - A2/π
+
+π_low, π_high = np.sort(np.roots([1, -A1, A2]))
+print(f"A1 = {A1}, A2 = {A2}")
+print(f"stationary gross inflation rates: {π_low} and {π_high}")
+print(f" i.e. net inflation of {100*(π_low-1):.0f}% and {100*(π_high-1):.0f}% per period")
+print(f"\nslope of g at the low rate: {A2/π_low**2:.4f} -> unstable under RE dynamics")
+print(f"slope of g at the high rate: {A2/π_high**2:.4f} -> stable under RE dynamics")
+```
+
+There it is: the *low*-inflation stationary equilibrium is **unstable** under the rational
+expectations dynamics, and the high-inflation one is stable.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Rational expectations inflation dynamics"
+ name: fig-olg-re-dynamics
+---
+π_grid = np.linspace(1.35, 4.2, 400)
+
+fig, ax = plt.subplots(figsize=(6.5, 5.5))
+ax.plot(π_grid, g_map(π_grid), lw=1.6, label=r"$g(\pi) = A_1 - A_2/\pi$")
+ax.plot(π_grid, π_grid, 'k--', lw=0.8, label="45 degree line")
+
+# cobweb the RE dynamics away from the low steady state
+π = π_low + 0.05
+for _ in range(7):
+ nxt = g_map(π)
+ ax.plot([π, π], [π, nxt], color='C3', lw=0.9)
+ ax.plot([π, nxt], [nxt, nxt], color='C3', lw=0.9)
+ π = nxt
+
+for p, name in [(π_low, "low"), (π_high, "high")]:
+ ax.plot(p, p, 'ko', ms=6)
+ ax.annotate(f"{name}\n$\\pi = {p:.0f}$", (p, p), textcoords="offset points",
+ xytext=(10, -22), fontsize=9)
+ax.set_xlabel(r"$\pi_t$")
+ax.set_ylabel(r"$\pi_{t+1}$")
+ax.legend(frameon=False, loc='upper left')
+plt.show()
+```
+
+The red staircase starts just above the low-inflation steady state and marches away from it,
+straight to the high-inflation one.
+
+```{code-cell} ipython3
+def re_path(π0, T=25):
+ path = [π0]
+ for _ in range(T):
+ nxt = g_map(path[-1])
+ path.append(nxt if nxt > 0 else np.nan)
+ return np.array(path)
+
+pd.DataFrame({f"$\\pi_0 = {p}$": re_path(p) for p in (1.99, 2.01, 2.5, 4.0)}
+ ).rename_axis("t").round(4).iloc[[0, 1, 2, 5, 10, 25]]
+```
+
+Started fractionally below the low steady state the economy collapses (the price level
+implied by {eq}`inflation_map` goes negative); started fractionally above it, the economy
+converges to $\pi = 3$.
+
+This matters because of how the two equilibria rank.
+
+The comparative statics of the low-inflation equilibrium are "classical": raising the deficit
+$G$ lowers $A_1$, which raises the low stationary inflation rate.
+
+The comparative statics of the high-inflation equilibrium are the reverse, because the economy
+is on the wrong side of an inflation-tax **Laffer curve**; see {ref}`olg_ex3`.
+
+And the low-inflation equilibrium Pareto-dominates every other equilibrium of this model,
+stationary or not.
+
+So the rational expectations dynamics repel the economy from the Pareto-optimal outcome, and
+most classical doctrine in monetary theory depends on selecting the equilibrium that those
+dynamics reject.
+
+### Least squares dynamics
+
+{cite:t}`MarcetSargent1989hyper` studied a version of this economy in which agents do not have
+perfect foresight but instead **run a regression**.
+
+Each period they regress the price level on its own lagged value and use the fitted slope as
+their forecast of gross inflation,
+
+$$
+\beta_t = \frac{\sum_{s < t} p_s p_{s-1}}{\sum_{s < t} p_{s-1}^2},
+\qquad
+\pi^e_t = \beta_t ,
+$$
+
+and then save $s_t = (w_1 - w_2 \beta_t)/2$ according to {eq}`saving_deterministic`.
+
+Since $\beta_t$ uses data through $t-1$ only, there is no simultaneity: beliefs determine
+saving, saving determines the price level, and the new price level enters next period's
+regression.
+
+Writing $\beta_t$ as
+
+$$
+\beta_t = \frac{\sum_{s 0$ by printing currency.
+
+Under the parameter restriction $\gamma(y - g) > g$ the model has a unique stationary
+equilibrium, with gross inflation
+
+$$
+\pi = \frac{(y-g)\gamma - \beta g}{(y-g)\gamma - g},
+$$
+
+which has the classical property of rising with the deficit.
+
+```{code-cell} ipython3
+γ_b, β_b, y_b = 1.0, 1/1.05, 11.0
+
+def brock_π(g):
+ return ((y_b - g)*γ_b - β_b*g) / ((y_b - g)*γ_b - g)
+
+pd.DataFrame({
+ "$g$": [0.5, 1.0, 1.5, 2.0],
+ "restriction $\\gamma(y-g) > g$": [γ_b*(y_b - g) > g for g in (0.5, 1.0, 1.5, 2.0)],
+ "stationary $\\pi$": [brock_π(g) for g in (0.5, 1.0, 1.5, 2.0)],
+ "return on currency $1/\\pi$": [1/brock_π(g) for g in (0.5, 1.0, 1.5, 2.0)],
+}).round(5)
+```
+
+But the model *also* has a continuum of non-stationary equilibria, indexed by initial price
+levels below the stationary one, in all of which the gross inflation rate converges to $\beta$
+— that is, the economy ends in **deflation**, with the return on currency approaching
+$1/\beta = 1.05$ and real balances exploding.
+
+These exist because the demand for currency implied by $\gamma \ln(m/p)$ is so elastic with
+respect to the rate of return that the government can raise enough seigniorage to finance $g$
+even while paying interest on currency through deflation.
+
+Applying least squares learning here selects the *classical stationary* equilibrium, just as
+in the overlapping generations model.
+
+The difference is the welfare ranking: in the Brock model, the non-stationary equilibria all
+Pareto-dominate the classical stationary one that learning picks.
+
+So least squares dynamics reliably select the *classical* equilibrium.
+
+Whether that equilibrium is the good one is a separate question, and the answer depends on the
+model.
+
+## A government learning the Phillips curve
+
+The examples so far put adaptive *households* into a monetary economy.
+
+We close with an application in which the adaptive agent is the **government**, learning an
+econometric relationship it can act on.
+
+The environment is different from the overlapping generations model — it is a Phillips curve
+economy — but the machinery is the same: an agent runs a regression, acts on it, and thereby
+helps generate the very data it is fitting.
+
+This example matters for a further reason.
+
+It is the one in {cite:t}`Sargent1993` that most directly seeded later work: {cite:t}`Sims1988`
+and {cite:t}`Chung1990` studied it, and it led Sargent to the escape-dynamics model of
+{cite:t}`Sargent1999`, *The Conquest of American Inflation*, which the {doc}`Phillips curve lectures `
+study in depth.
+
+### Two stories about post-war inflation
+
+There is a well-known account of the rise and fall of U.S. inflation after World War II.
+
+In it, the private sector always had rational expectations and understood the natural-rate
+hypothesis, while the government, for a time, wrongly believed there was an *exploitable*
+Phillips curve, a lasting tradeoff between inflation and unemployment.
+
+The government estimated Phillips curves on 1960s data, saw a tradeoff, and tried to buy lower
+unemployment with higher inflation.
+
+The result was higher inflation with no lasting gain in employment: the Phillips curve shifted
+up adversely, as the natural-rate hypothesis says it must.
+
+But there is a counter-story, told by economists who fit Phillips curves at the time.
+
+They argued that their procedures were *adaptive*, that they detected the adverse shift
+quickly enough to give sound advice, and so need not have led the government astray.
+
+The virtue of the model below is that a single specification nests both stories, with a
+parameter deciding which one plays out.
+
+### The model
+
+Inflation is decomposed into an expected and an unexpected part,
+
+```{math}
+:label: phillips_inflation
+
+\pi_t = g_{t-1} + \eta_t,
+```
+
+where $g_{t-1}$ is the inflation the public expects (and which the government controls) and
+$\eta_t$ is the public's forecast error, orthogonal to earlier information.
+
+The *true* Phillips curve is the natural-rate one: only the **surprise** $\eta_t$ moves
+unemployment:
+
+```{math}
+:label: phillips_true
+
+U_t = U^* - \theta(\pi_t - g_{t-1}) + u_t = U^* - \theta \eta_t + u_t,
+\qquad \theta > 0.
+```
+
+The government does not know this.
+
+It believes unemployment depends on the *level* of inflation, and fits the non-expectational
+regression
+
+```{math}
+:label: phillips_perceived
+
+U_t = \alpha_{0} + \alpha_{1}\pi_t + \epsilon_t.
+```
+
+Each period it minimizes $\tfrac12 \mathbb{E}(U_t^2 + \pi_t^2)$ subject to its perceived model, which
+yields the myopic target
+
+```{math}
+:label: phillips_rule
+
+g_{t-1} = -\frac{\alpha_{0}\,\alpha_{1}}{1 + \alpha_{1}^2}.
+```
+
+The system is **self-referential**: the government's beliefs $\alpha$ set its policy $g$
+through {eq}`phillips_rule`; the policy shapes the joint distribution of $(\pi, U)$; and that
+distribution is what the government's regression {eq}`phillips_perceived` estimates.
+
+```{code-cell} ipython3
+U_star, θ_pc = 5.0, 1.0 # natural rate 5%, Phillips slope
+σ_η, σ_u = 0.3, 0.3 # inflation-surprise and unemployment shocks
+
+def govt_target(α):
+ "Government's myopic optimal target inflation."
+ α0, α1 = α
+ return -α0 * α1 / (1 + α1**2)
+```
+
+### The consistent equilibrium
+
+A self-confirming, or **consistent**, equilibrium is a belief vector $\alpha$ whose induced
+policy generates data whose regression returns the same $\alpha$.
+
+Under a constant policy $g$, the surprise $\eta_t$ is the only thing moving inflation around
+its mean, so the population regression of $U$ on $\pi$ has slope exactly $-\theta$ and
+intercept $U^* + \theta g$.
+
+Solving that fixed point with the decision rule {eq}`phillips_rule` gives Kydland and
+Prescott's time-consistent outcome,
+
+$$
+\alpha_0 = (1 + \theta^2) U^*, \qquad \alpha_1 = -\theta,
+\qquad g \to \theta U^* .
+$$
+
+```{code-cell} ipython3
+α_consistent = np.array([(1 + θ_pc**2) * U_star, -θ_pc])
+g_consistent = θ_pc * U_star
+print(f"consistent equilibrium beliefs α = {α_consistent}, target inflation g = {g_consistent}")
+print(f"optimal (Ramsey) outcome: target inflation g = 0")
+```
+
+This is the **inflation bias**.
+
+The government inflates at rate $\theta U^*$ and gets nothing for it: because only surprises
+move unemployment, average unemployment is $U^*$ either way.
+
+Had the government understood the true model {eq}`phillips_true`, it would have chosen $g = 0$
+: the same unemployment with none of the inflation.
+
+### Constant-coefficient beliefs: learning the inflation bias
+
+Suppose the government believes its coefficients are constant and estimates them by least
+squares, updating a little each period.
+
+The deterministic path that least squares learning follows on average — its **mean
+dynamics** — moves beliefs toward the regression their current policy would induce.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Least squares learning converges to the inflation bias"
+ name: fig-olg-mean-dynamics
+---
+def induced_beliefs(α):
+ "The OLS fit that the current beliefs' policy would induce in a stationary sample."
+ g = govt_target(α)
+ return np.array([U_star + θ_pc * g, -θ_pc]) # slope is exactly -θ
+
+def mean_dynamics(α0=(0.0, 0.0), lr=0.2, T=60):
+ α = np.array(α0, float)
+ path = np.empty((T, 3))
+ for t in range(T):
+ α = α + lr * (induced_beliefs(α) - α)
+ path[t] = [α[0], α[1], govt_target(α)]
+ return path
+
+path = mean_dynamics()
+
+fig, ax = plt.subplots(figsize=(7.5, 4))
+ax.plot(path[:, 2], 'C0', lw=1.6, label="target inflation $g$")
+ax.plot(path[:, 0], 'C1', lw=1.2, label=r"belief $\alpha_0$")
+ax.plot(path[:, 1], 'C2', lw=1.2, label=r"belief $\alpha_1$")
+ax.axhline(g_consistent, color='C0', ls='--', lw=0.8)
+ax.set_xlabel("iteration")
+ax.legend(frameon=False)
+plt.show()
+```
+
+Least squares learning walks the government straight into the consistent equilibrium: beliefs
+settle at $\alpha = ((1+\theta^2)U^*, -\theta)$ and target inflation at $\theta U^* = 5$.
+
+The government keeps inflating, period after period, at a rate that buys it nothing, because
+its own policy keeps producing data that confirm the tradeoff it believes in.
+
+### Random-coefficient beliefs: escaping toward the optimum
+
+{cite:t}`Sims1988` and {cite:t}`Chung1990` then asked what happens if the government is *less*
+sure its coefficients are constant.
+
+They let the government suspect the coefficients drift — a random walk in $\alpha$ — and
+estimate them with a Kalman filter, which discounts old data in favor of recent data.
+
+That is a **constant-gain** algorithm: instead of a $1/t$ gain that eventually freezes the
+estimates, it keeps a fixed gain forever, so the government never stops paying attention to
+what just happened.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Constant-gain learning and recurrent escapes to zero inflation"
+ name: fig-olg-escapes
+---
+def constant_gain(T=40_000, gain=0.02, seed=1):
+ "Government estimates a drifting-coefficient Phillips curve; returns target inflation g_t."
+ rng = np.random.default_rng(seed)
+ α = α_consistent.copy() # start at the inflation bias
+ R = np.eye(2)
+ g_path = np.empty(T)
+ for t in range(T):
+ g = govt_target(α)
+ η = σ_η * rng.standard_normal()
+ π = g + η
+ U = U_star - θ_pc * η + σ_u * rng.standard_normal() # true Phillips curve
+ x = np.array([1.0, π])
+ R = R + gain * (np.outer(x, x) - R)
+ α = α + gain * np.linalg.solve(R, x * (U - x @ α))
+ g_path[t] = g
+ return g_path
+
+g_path = constant_gain()
+
+fig, ax = plt.subplots(figsize=(9, 4))
+ax.plot(g_path, lw=0.5)
+ax.axhline(g_consistent, color='C3', ls='--', lw=1, label="consistent equilibrium (inflation bias)")
+ax.axhline(0, color='C2', ls=':', lw=1, label="Ramsey / zero inflation")
+ax.set_xlabel("$t$")
+ax.set_ylabel("target inflation $g_t$")
+ax.legend(frameon=False)
+plt.show()
+```
+
+The constant-gain government does not settle at the inflation bias.
+
+It climbs toward it, then abruptly **escapes** toward zero inflation, then drifts back up, then
+escapes again, in a recurring sawtooth.
+
+The mechanism is the one Sims and Chung identified.
+
+Because the government discounts old data, a run of observations in which inflation surprises
+are small and unemployment stays near $U^*$ regardless of inflation quickly persuades it that
+the tradeoff is not there, and it stops exploiting a tradeoff it no longer believes in,
+dropping inflation toward zero.
+
+The willingness to entertain drifting coefficients lets the government learn the truth of the
+natural-rate hypothesis *without having to sit through a long inflation to do it*.
+
+How often the escapes happen depends on the gain — the government's degree of doubt about its
+own model — which is exactly the parameter {cite:t}`Sims1988` found to select between the two
+stories.
+
+A small gain keeps the economy near the inflation bias; a larger gain sends it toward the
+optimum more often ({ref}`olg_ex4`).
+
+### From escape dynamics to the *Conquest of American Inflation*
+
+The two regimes are formalizations of the two stories we began with.
+
+The consistent-equilibrium path is the natural-rate story: a government that trusts its
+mis-specified model inflates and stays stuck.
+
+The escaping path is the counter-story: a government alert to model change detects the poor
+tradeoff and backs away from it.
+
+This is the seed of a large literature.
+
+{cite:t}`Sims1988` and {cite:t}`Chung1990` showed both that the model could nest the two
+stories and — in Chung's case — that it could be *econometrically estimated* on post-war U.S.
+data, imputing a genuine estimation procedure to the government inside the model.
+
+Their work led {cite:t}`Sargent1999` to the **escape-route** model of *The Conquest of American
+Inflation*, in which recurrent escapes from the high-inflation Nash equilibrium toward the
+Ramsey outcome offer a theory of *how* U.S. inflation was actually conquered in the early
+1980s.
+
+The QuantEcon lectures {doc}`phillips_escaping_nash` and {doc}`phillips_learning` develop that
+model and its escape dynamics in full; the sawtooth above is a first glimpse of what they
+study.
+
+## Concluding remarks
+
+Part 1 showed that a system of adaptive agents can find a rational expectations equilibrium it
+was never told about, and made visible how much the *complexity of the environment* governs
+the speed: a rare deficit state is learned about slowly, and a large state space forces a
+parametric shortcut that may put the equilibrium out of reach entirely.
+
+Part 2 showed something stronger.
+
+Where the model has two equilibria, the learning dynamics do not merely find one; they
+systematically pick out the one that the rational expectations dynamics reject, and laboratory
+subjects go the same way.
+
+That is the best case in {cite:t}`Sargent1993` for treating adaptive dynamics as an
+equilibrium selection device.
+
+The Brock counterexample marks its limit.
+
+The selection is a fact about the algorithm, not a welfare theorem, and Sargent is candid about
+the resulting discomfort:
+
+> I know that it is inconsistent to doubt the real-time dynamics but keep the equilibria
+> selected by them. I confess that my affection for the selection performed in the monetary
+> models described in Chapter 6 is partly driven by my prior conviction that the selected
+> equilibria seem sensible to me.
+
+The Phillips curve application sharpened the point in two ways.
+
+It showed that *which* learning scheme finds the good outcome is model-dependent: here least
+squares walks into the bad equilibrium, and it is the constant-gain government — the one that
+never stops doubting — that escapes toward the good one.
+
+And it turned the *gain* into the object of interest, foreshadowing the escape-dynamics
+research of {cite:t}`Sargent1999` taken up in {doc}`phillips_escaping_nash` and
+{doc}`phillips_learning`.
+
+{doc}`marimon_mcgrattan_sargent` takes the selection idea in a different direction, into a
+setting where the competing equilibria are not high and low inflation but different *monetary
+institutions*: which good becomes money.
+
+## Exercises
+
+```{exercise-start}
+:label: olg_ex1
+```
+
+Because agents learn a separate saving rate for each deficit state, they can only learn about
+a state as fast as that state occurs.
+
+Compare two economies with $\bar G = (0.8, 0)$ that differ only in the transition matrix: one
+where the two states are equally likely, and one where the high-deficit state is rare.
+
+For each, compute the stationary equilibrium, run the learning simulation, and report how far
+each state's learned saving rate is from its target along with how often that state was
+visited.
+
+```{exercise-end}
+```
+
+```{solution-start} olg_ex1
+:class: dropdown
+```
+
+```{code-cell} ipython3
+G_ex = np.array([0.8, 0.0])
+cases = {"equally likely": np.array([[0.5, 0.5], [0.5, 0.5]]),
+ "high deficit rare": np.array([[0.5, 0.5], [0.03, 0.97]])}
+
+rows = []
+for name, P in cases.items():
+ s_star, _ = stationary_equilibrium(G_ex, P)
+ ergodic = qe.MarkovChain(P).stationary_distributions[0]
+ hist, _ = simulate(G_ex, P, T=30_000, s0=start)
+ learned = hist[-1].mean(axis=0)
+ for k in (0, 1):
+ rows.append([name, k+1, ergodic[k], s_star[k], learned[k],
+ abs(learned[k] - s_star[k])])
+
+pd.DataFrame(rows, columns=["chain", "state", "ergodic prob.",
+ "equilibrium $s_i$", "learned $s_i$", "error"]).round(4)
+```
+
+When the two states are equally likely the two saving rules are learned about equally well.
+
+When the high-deficit state occurs only about six percent of the time, the error in its rule is
+an order of magnitude larger than the error in the rule for the common state, even though both
+have had 30,000 periods of calendar time to settle.
+
+Sargent's comment is worth keeping in mind: in terms of *unconditional expected utility*,
+failing to learn what to do in a rarely visited state may cost very little.
+
+The cost of slow learning is not proportional to the size of the error.
+
+Note also that changing the transition matrix changes the equilibrium itself — beliefs about
+the persistence of deficits feed straight into {eq}`olg_equilibrium` — so the comparison has to
+be against a separately recomputed target for each chain.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: olg_ex2
+```
+
+The gain sequence $\gamma_\tau = 1/\tau$ is what makes {eq}`robbins_monro` converge.
+
+An agent who suspected the environment might be drifting would use a **constant gain**
+instead, weighting recent experience more heavily forever.
+
+Modify the simulation to use a constant gain and describe what happens to the limiting behavior
+of the saving rules.
+
+Compare gains of $0.02$ and $0.005$ against the $1/\tau$ benchmark.
+
+```{exercise-end}
+```
+
+```{solution-start} olg_ex2
+:class: dropdown
+```
+
+```{code-cell} ipython3
+def simulate_constant_gain(G, P, gain, T=30_000, s0=None, seed=0, band=0.4):
+ rng = np.random.default_rng(seed)
+ n = len(G)
+ s = np.array(s0, float)
+ M = np.array([[U_double(s[j, k], 1.0) for k in range(n)] for j in range(2)])
+ floor = G/N + band
+ hist = np.empty((T, 2, n))
+ i, prev = 0, None
+ for t in range(T):
+ j = t % 2
+ s_t = s[j, i]
+ if prev is not None:
+ j_p, i_p, s_p = prev
+ R = (N*s_t - G[i]) / (N*s_p)
+ M[j_p, i_p] += gain * (U_double(s_p, R) - M[j_p, i_p])
+ step = gain * U_prime(s_p, R) / M[j_p, i_p]
+ s[j_p, i_p] = np.clip(s[j_p, i_p] - step, floor[i_p], w1 - 0.4)
+ prev = (j, i, s_t)
+ hist[t] = s
+ i = rng.choice(n, p=P[i])
+ return hist
+
+rows = []
+for label, hist in [("gain = 0.02", simulate_constant_gain(G2, P2, 0.02, s0=start)),
+ ("gain = 0.005", simulate_constant_gain(G2, P2, 0.005, s0=start)),
+ ("gain = 1/τ", simulate(G2, P2, T=30_000, s0=start)[0])]:
+ tail = hist[-5_000:].mean(axis=1) # average the two classes
+ rows.append([label, *tail.mean(axis=0), *tail.std(axis=0)])
+
+pd.DataFrame(rows, columns=["", "mean $s_1$", "mean $s_2$",
+ "s.d. $s_1$", "s.d. $s_2$"]).set_index("").round(5)
+```
+
+```{code-cell} ipython3
+fig, ax = plt.subplots(figsize=(8, 4))
+for gain, colour in [(0.02, 'C1'), (0.005, 'C2')]:
+ h = simulate_constant_gain(G2, P2, gain, s0=start)
+ ax.plot(h[:, 0, 0], color=colour, lw=0.7, label=f"gain = {gain}")
+h, _ = simulate(G2, P2, T=30_000, s0=start)
+ax.plot(h[:, 0, 0], color='C0', lw=1.2, label=r"gain = $1/\tau$")
+ax.axhline(s2[0], color='k', lw=0.8, ls='--', label="equilibrium")
+ax.set_xlim(0, 30_000)
+ax.set_ylim(3.8, 5.0)
+ax.set_xlabel("$t$")
+ax.set_ylabel("saving rate, state 1")
+ax.legend(frameon=False, fontsize=9)
+plt.show()
+```
+
+The constant-gain rules do not converge.
+
+They settle into a *neighborhood* of the equilibrium and then keep rattling around inside it
+forever, because each new observation is always given the same weight and so never stops
+moving the estimate.
+
+Shrinking the gain tightens the band but does not eliminate it.
+
+Only a gain that declines to zero — like $1/\tau$ — delivers convergence to a point, which is
+why the convergence theorems for these systems all require it.
+
+What the constant-gain agent buys in exchange is adaptability: if the deficit process were to
+change, the $1/\tau$ agent would barely notice, while the constant-gain agent would track the
+change.
+
+That trade-off reappears throughout this literature.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: olg_ex3
+```
+
+The claim that the high-inflation steady state sits on "the wrong side of a Laffer curve" can
+be made precise.
+
+In a stationary equilibrium with gross inflation $\pi$, real balances are
+$h = (w_1 - w_2\pi)/2$ and the government collects seigniorage $h(1 - 1/\pi)$.
+
+Plot seigniorage as a function of $\pi$, mark the two stationary equilibria, and use the
+picture to explain why raising $G$ raises the low stationary inflation rate but *lowers* the
+high one, and what happens when $G$ gets too large.
+
+```{exercise-end}
+```
+
+```{solution-start} olg_ex3
+:class: dropdown
+```
+
+```{code-cell} ipython3
+def seigniorage(π):
+ return (w1_d - w2_d*π)/2 * (1 - 1/π)
+
+π_grid = np.linspace(1.01, 5.5, 500)
+peak = π_grid[np.argmax(seigniorage(π_grid))]
+
+fig, ax = plt.subplots(figsize=(7.5, 4.5))
+ax.plot(π_grid, seigniorage(π_grid), lw=1.6)
+for G_try, style in [(1.0, '-'), (1.03, '--'), (1.0505, ':')]:
+ ax.axhline(G_try, color='C3', lw=0.9, ls=style, label=f"$G = {G_try}$")
+ax.plot([π_low, π_high], [seigniorage(π_low), seigniorage(π_high)], 'ko', ms=6)
+ax.axvline(peak, color='gray', lw=0.8, ls=':')
+ax.set_xlabel(r"gross inflation $\pi$")
+ax.set_ylabel("seigniorage")
+ax.set_ylim(0, 1.5)
+ax.legend(frameon=False)
+plt.show()
+
+print(f"peak of the Laffer curve at π = {peak:.4f}, raising {seigniorage(peak):.4f}")
+print(f"the two steady states raise {seigniorage(π_low):.4f} and {seigniorage(π_high):.4f}")
+```
+
+The two stationary equilibria are the two inflation rates at which the inflation-tax Laffer
+curve crosses the required revenue $G$, one on each side of the peak.
+
+The low one sits on the rising branch, so financing a larger deficit there requires *more*
+inflation: the classical comparative static.
+
+The high one sits on the falling branch, where more inflation raises *less* revenue, so a
+larger deficit is financed by a *lower* stationary inflation rate, the anti-classical
+comparative static that makes the high equilibrium so awkward.
+
+As $G$ rises the horizontal line moves up and the two intersections converge.
+
+Past the peak of the curve there is no stationary equilibrium at all: the deficit exceeds the
+maximum revenue the inflation tax can raise.
+
+```{code-cell} ipython3
+rows = []
+for G_try in (0.8, 1.0, 1.04, 1.05, 1.0505, 1.06):
+ A1_try = (w1_d + w2_d - 2*G_try)/w2_d
+ disc = A1_try**2 - 4*A2
+ if disc < 0:
+ rows.append([G_try, np.nan, np.nan, "none"])
+ else:
+ r = np.sort(np.roots([1, -A1_try, A2]))
+ rows.append([G_try, r[0], r[1], "two"])
+
+pd.DataFrame(rows, columns=["$G$", "low $\\pi$", "high $\\pi$",
+ "stationary equilibria"]).round(4)
+```
+
+The two roots move toward each other as $G$ rises and collide at $G \approx 1.0505$, after
+which the model has no stationary equilibrium at all.
+
+That number is not a coincidence.
+
+Setting the discriminant of {eq}`inflation_map`'s fixed-point equation to zero gives a double
+root at $\pi = \sqrt{A_2} = \sqrt{w_1/w_2}$, and the deficit at which it occurs is exactly the
+peak of the Laffer curve computed above.
+
+```{code-cell} ipython3
+G_max = (w1_d + w2_d - 2*np.sqrt(A2)*w2_d) / 2
+print(f"roots collide at G = {G_max:.6f}, where π = √(w1/w2) = {np.sqrt(A2):.6f}")
+print(f"peak of the Laffer curve = {seigniorage(np.sqrt(A2)):.6f} at π = {np.sqrt(A2):.6f}")
+```
+
+The largest deficit the government can finance is the maximum of the inflation tax, and at
+that deficit the two stationary equilibria have merged into one.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: olg_ex4
+```
+
+In the Phillips curve application, the constant-gain government escapes toward zero inflation,
+and the lecture claims that *how often* it escapes depends on the gain, the government's degree
+of doubt about its own model.
+
+Make the claim quantitative.
+
+For a range of gains, simulate the constant-gain economy and measure both the average target
+inflation and the fraction of time spent near the Ramsey outcome (say, $g_t < 1$).
+
+Which way does more doubt push the economy, and how does this connect to the two stories about
+post-war inflation?
+
+```{exercise-end}
+```
+
+```{solution-start} olg_ex4
+:class: dropdown
+```
+
+```{code-cell} ipython3
+rows = []
+for gain in (0.005, 0.01, 0.02, 0.04, 0.08):
+ tails = [constant_gain(T=30_000, gain=gain, seed=s)[-20_000:] for s in range(4)]
+ mean_g = np.mean([g.mean() for g in tails])
+ near_ramsey = np.mean([np.mean(g < 1) for g in tails])
+ rows.append([gain, mean_g, near_ramsey])
+
+pd.DataFrame(rows, columns=["gain", "mean target inflation", "fraction of time $g < 1$"]
+ ).set_index("gain").round(3)
+```
+
+More doubt — a larger gain — pushes the economy away from the inflation bias and toward the
+optimum: mean inflation falls and the economy spends a larger fraction of its time near the
+Ramsey outcome.
+
+The intuition is that a higher gain discounts old data more heavily, so the government reacts
+faster to the run of observations that reveals the tradeoff to be illusory, and escapes sooner
+and more often.
+
+This is exactly the parameter {cite:t}`Sims1988` found to select between the two stories.
+
+A government confident in its constant-coefficient model (small gain) is the natural-rate
+story's government, stuck inflating; a government alert to model change (large gain) is the
+counter-story's, detecting the poor tradeoff and stepping away from it.
+
+The same gain reappears as the central object in {cite:t}`Sargent1999`, where the *rate* of
+escape from the high-inflation equilibrium governs how quickly an inflation can be conquered.
+
+```{solution-end}
+```
diff --git a/lectures/phillips_adaptive.md b/lectures/phillips_adaptive.md
index d67d9b4b2..f6f24407a 100644
--- a/lectures/phillips_adaptive.md
+++ b/lectures/phillips_adaptive.md
@@ -22,6 +22,9 @@ kernelspec:
# Adaptive Expectations and the Phelps Problem
+```{index} single: Phillips Curve; Adaptive Expectations
+```
+
```{contents} Contents
:depth: 2
```
@@ -47,6 +50,13 @@ This lecture continues the study of Phillips curve tradeoffs begun in {doc}`phil
It follows chapter 5 of {cite}`Sargent1999`.
+{doc}`phillips_credible_policies` pursued better-than-Nash outcomes while keeping *everyone*
+rational, and found too many of them: a continuum of sustainable values, with nothing inside the
+theory to choose among them.
+
+Here we retreat from that perfection in the smallest way we can, by keeping the government
+rational and giving the public a mechanical forecasting rule.
+
We describe
* the Cagan-Friedman adaptive expectations hypothesis,
@@ -102,6 +112,12 @@ The weights $(1 - \lambda)\lambda^{i-1}$ sum to one, so a permanently maintained
Let's confirm the induction property numerically.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Adaptive expectations converge to a constant inflation policy: the induction property"
+ name: fig-adapt-induction
+---
def adaptive_forecast(y, λ, x0=0.0):
"Simulate x_t = λ x_{t-1} + (1-λ) y_{t-1}."
T = len(y)
@@ -155,26 +171,26 @@ Because expected inflation $x_t$ is predetermined at the start of period $t$, it
The government's problem is therefore a discounted LQ control problem with
-* state $s_t = \begin{bmatrix} 1 & x_t \end{bmatrix}'$,
+* state $s_t = \begin{bmatrix} 1 & x_t \end{bmatrix}^\top$,
* control $y_t$, and
* transition $x_{t+1} = \lambda x_t + (1 - \lambda) y_t$.
### Casting the problem in LQ form
-Write $U_t = a' s_t - \theta y_t$ with $a = \begin{bmatrix} U^* & \theta \end{bmatrix}'$.
+Write $U_t = a^\top s_t - \theta y_t$ with $a = \begin{bmatrix} U^* & \theta \end{bmatrix}^\top$.
The per-period loss $\tfrac{1}{2}(U_t^2 + y_t^2)$ then equals
$$
-\frac{1}{2}\left[ s_t'(a a') s_t + (\theta^2 + 1) y_t^2 - 2 \theta \, y_t \, (a' s_t) \right] .
+\frac{1}{2}\left[ s_t^\top (a a^\top) s_t + (\theta^2 + 1) y_t^2 - 2 \theta \, y_t \, (a^\top s_t) \right] .
$$
-Matching this to the QuantEcon LQ loss $s_t' R s_t + y_t' Q y_t + 2 y_t' N s_t$, and matching the transition to $s_{t+1} = A s_t + B y_t$, gives
+Matching this to the QuantEcon LQ loss $s_t^\top R s_t + y_t^\top Q y_t + 2 y_t^\top N s_t$, and matching the transition to $s_{t+1} = A s_t + B y_t$, gives
$$
-R = \tfrac{1}{2} a a', \quad
+R = \tfrac{1}{2} a a^\top, \quad
Q = \tfrac{1}{2}(\theta^2 + 1), \quad
-N = -\tfrac{1}{2}\theta\, a', \quad
+N = -\tfrac{1}{2}\theta\, a^\top, \quad
A = \begin{bmatrix} 1 & 0 \\ 0 & \lambda \end{bmatrix}, \quad
B = \begin{bmatrix} 0 \\ 1 - \lambda \end{bmatrix} .
$$
@@ -277,6 +293,12 @@ In the undiscounted case ($\delta = 1$), inflation is driven all the way to the
Let's plot the full disinflation paths.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Optimal disinflation paths from the Phelps problem, discounted and undiscounted
+ name: fig-adapt-disinflation
+---
fig, axes = plt.subplots(1, 2, figsize=(11, 4.5))
for δ, ax in zip([0.96, 1.0], axes):
@@ -303,23 +325,23 @@ It is useful to state a more general version of the Phelps problem in which the
Define the vectors
$$
-X_{U,t} = \begin{bmatrix} U_{t-1} & \cdots & U_{t-m_U} \end{bmatrix}',
+X_{U,t} = \begin{bmatrix} U_{t-1} & \cdots & U_{t-m_U} \end{bmatrix}^\top,
\qquad
-X_{y,t} = \begin{bmatrix} y_{t-1} & \cdots & y_{t-m_y} \end{bmatrix}',
+X_{y,t} = \begin{bmatrix} y_{t-1} & \cdots & y_{t-m_y} \end{bmatrix}^\top,
$$
-and the state vector $X_t = \begin{bmatrix} X_{U,t}' & X_{y,t}' & 1 \end{bmatrix}'$, which collects information dated $t-1$ and earlier.
+and the state vector $X_t = \begin{bmatrix} X_{U,t}^\top & X_{y,t}^\top & 1 \end{bmatrix}^\top$, which collects information dated $t-1$ and earlier.
We can write two reduced-form Phillips curves that differ only in their *direction of fit*:
$$
-\text{Classical:} \quad U_t = \gamma' X_{C,t} + \varepsilon_{C,t},
-\qquad X_{C,t} = \begin{bmatrix} y_t & X_{t-1}' \end{bmatrix}',
+\text{Classical:} \quad U_t = \gamma^\top X_{C,t} + \varepsilon_{C,t},
+\qquad X_{C,t} = \begin{bmatrix} y_t & X_{t-1}^\top \end{bmatrix}^\top,
$$
$$
-\text{Keynesian:} \quad y_t = \beta' X_{K,t} + \varepsilon_{K,t},
-\qquad X_{K,t} = \begin{bmatrix} U_t & X_{t-1}' \end{bmatrix}' .
+\text{Keynesian:} \quad y_t = \beta^\top X_{K,t} + \varepsilon_{K,t},
+\qquad X_{K,t} = \begin{bmatrix} U_t & X_{t-1}^\top \end{bmatrix}^\top.
$$
The subscripts $C$ and $K$ stand for *Classical* (regress $U$ on $y$) and *Keynesian* (regress $y$ on $U$).
@@ -341,7 +363,11 @@ Once that substitution is made, the state variable $x_t$ disappears from view.
These objects — $\gamma$, $\beta$, $h(\gamma)$, and the two directions of fit — are exactly the ingredients we will need to define **self-confirming equilibria** in {doc}`phillips_self_confirming`.
```{note}
-The **induction hypothesis** is the restriction that, in the Keynesian Phillips curve $y_t = \beta' X_{K,t} + \varepsilon_{K,t}$, the weights on lagged $y$'s sum to unity (equivalently, in the classical form the weights on current and lagged $y$'s sum to zero). Under adaptive expectations this holds because the weights in {eq}`pa_geom` sum to one.
+The **induction hypothesis** is the restriction that, in the Keynesian Phillips curve $y_t = \beta^\top X_{K,t} + \varepsilon_{K,t}$,
+the weights on lagged $y$'s sum to unity (equivalently, in the classical form the weights on
+current and lagged $y$'s sum to zero).
+
+Under adaptive expectations this holds because the weights in {eq}`pa_geom` sum to one.
```
## Testing the natural-rate hypothesis
@@ -417,9 +443,10 @@ for δ in δ_grid:
y_inf.append(y[-1])
fig, ax = plt.subplots(figsize=(8, 4.5))
-ax.plot(δ_grid, y_inf, 'o-')
+ax.plot(δ_grid, y_inf, 'o-', lw=2)
ax.set_xlabel(r'discount factor $\delta$')
ax.set_ylabel(r'limiting inflation $y_\infty$')
+ax.set_title('Limiting inflation falls as patience rises')
plt.show()
```
diff --git a/lectures/phillips_credibility.md b/lectures/phillips_credibility.md
index 3f712f695..e2349a652 100644
--- a/lectures/phillips_credibility.md
+++ b/lectures/phillips_credibility.md
@@ -22,6 +22,9 @@ kernelspec:
# The Credibility Problem
+```{index} single: Phillips Curve; Credibility Problem
+```
+
```{contents} Contents
:depth: 2
```
@@ -30,7 +33,7 @@ kernelspec:
This lecture describes a basic expectational Phillips curve model of the sort studied by {cite}`KydlandPrescott1977` and Robert Barro and David Gordon.
-It is the first in a suite of lectures based on chapters of {cite}`Sargent1999`.
+It is the first *modeling* lecture in a suite based on chapters of {cite}`Sargent1999`, following {doc}`phillips_two_stories`, which lays out the two stories and reviews the Lucas Critique.
Those chapters formalize
@@ -107,7 +110,9 @@ We work with the following objects.
**Nash equilibrium:** a pair $(x, y)$ satisfying (i) $x = y$, and (ii) $y = B(x)$.
-**Ramsey problem:** $\max_y r(y, y)$. The *Ramsey outcome* is the value of $y$ that attains the maximum.
+**Ramsey problem:** $\max_y r(y, y)$.
+
+The *Ramsey outcome* is the value of $y$ that attains the maximum.
**Best response dynamics:** the dynamical system $y_t = B(y_{t-1})$, $y_0$ given.
@@ -206,6 +211,12 @@ For a given expectation $x$, the Phillips curve {eq}`pc_phillips` is a downward-
The government's best response for $y$, given $x$, occurs where an indifference curve is tangent to the Phillips curve indexed by $x$.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Phillips curves indexed by expected inflation, government indifference curves, and the Nash and Ramsey outcomes
+ name: fig-cred-nash-ramsey
+---
fig, ax = plt.subplots(figsize=(7, 6))
U_grid = np.linspace(0, 12, 200)
@@ -218,7 +229,7 @@ for x in [0.0, y_N / 2, y_N]:
# government indifference curves (circles U^2 + y^2 = const)
ξ = np.linspace(0, 2 * np.pi, 200)
-for R in [y_R, np.hypot(cm.U_star, y_N)]:
+for R in [np.hypot(cm.U_star, y_R), np.hypot(cm.U_star, y_N)]:
if R > 0:
ax.plot(R * np.cos(ξ), R * np.sin(ξ), 'C1--', lw=1)
@@ -265,6 +276,12 @@ y_path = best_response_path(cm, y0=0.0, T=20)
```
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Best response dynamics cobwebbing up to the Nash inflation rate
+ name: fig-cred-best-response
+---
fig, ax = plt.subplots(figsize=(6, 6))
x_grid = np.linspace(0, y_N + 1, 100)
@@ -321,7 +338,7 @@ Actual inflation is a disturbed version of the best response mapping evaluated a
y_t = B(x_t) + \eta_t ,
```
-where $\eta_t$ is an i.i.d. mean-zero term that represents the government's imperfect control of inflation.
+where $\eta_t$ is an IID mean-zero term that represents the government's imperfect control of inflation.
Substituting {eq}`pc_expect2` into {eq}`pc_expect1` gives the stochastic recursion
@@ -362,12 +379,18 @@ def ls_learning(cm, T=2000, σ_η=1.0, seed=0):
for t in range(1, T + 1):
η = σ_η * rng.standard_normal()
y[t] = cm.B(x[t - 1]) + η
- gain = 1.0 / (t + 1) # decreasing gain
+ gain = 1.0 / t # the 1/(t-1) gain of equation (7), reindexed
x[t] = x[t - 1] + gain * (cm.B(x[t - 1]) - x[t - 1] + η)
return x, y
```
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Least squares learning of expected inflation converges to the Nash outcome
+ name: fig-cred-ls-learning
+---
x, y = ls_learning(cm, T=2000, σ_η=1.0)
fig, ax = plt.subplots(figsize=(9, 5))
@@ -398,9 +421,11 @@ Better outcomes can occur if the government plans for the future.
Subsequent lectures describe three ways of modeling foresight, which impute varying amounts of rationality and predict different qualities of outcomes:
-1. A reputational approach that attributes rational expectations to both the government and the public. Many outcomes are sustainable, ranging from repetition of the Ramsey outcome to paths worse than repetition of the Nash outcome.
-2. An approach that keeps the government rational but gives the public *adaptive* expectations in the original Cagan-Friedman sense. This is the subject of {doc}`phillips_adaptive`. Depending on a comparison between a discount factor and an adaptation parameter, this setup can improve outcomes and possibly sustain repetition of the Ramsey outcome.
-3. An approach that attributes adaptive behavior to both the government and the public. This is the subject of {doc}`phillips_misspecified` and {doc}`phillips_self_confirming`.
+1. A reputational approach that attributes rational expectations to both the government and the public, the subject of {doc}`phillips_credible_policies`.
+ - Many outcomes are sustainable, ranging from repetition of the Ramsey outcome to paths worse than repetition of the Nash outcome, a multiplicity that turns out to be the theory's chief lesson.
+2. An approach that keeps the government rational but gives the public *adaptive* expectations in the original Cagan-Friedman sense, the subject of {doc}`phillips_adaptive`.
+ - Depending on a comparison between a discount factor and an adaptation parameter, this setup can improve outcomes and possibly sustain repetition of the Ramsey outcome.
+3. An approach that attributes adaptive behavior to both the government and the public, the subject of {doc}`phillips_misspecified` and {doc}`phillips_self_confirming`.
## Appendix: stochastic approximation
@@ -424,7 +449,7 @@ Rewrite the recursion as
x_{n+1} = x_n + a_n \left[ B(x_n) - x_n + \eta_n \right] ,
```
-where $\eta_n$ is i.i.d. with mean zero and finite variance.
+where $\eta_n$ is IID with mean zero and finite variance.
Introduce the transformed time scale $t_0 = 0$, $t_n = \sum_{i=0}^{n-1} a_i$, and interpolate the discrete sequence $\{x_n\}$ into a continuous-time process $x^0(t)$.
@@ -486,6 +511,7 @@ for θ in [0.5, 1.0, 2.0]:
ax.axhline(0, color='k', lw=0.8)
ax.set_xlabel('$t$')
ax.set_ylabel('$x_t - \\theta U^*$')
+ax.set_title('Least squares learning by Phillips slope')
ax.legend()
plt.show()
```
diff --git a/lectures/phillips_credible_policies.md b/lectures/phillips_credible_policies.md
new file mode 100644
index 000000000..3c85eea52
--- /dev/null
+++ b/lectures/phillips_credible_policies.md
@@ -0,0 +1,1136 @@
+---
+jupytext:
+ text_representation:
+ extension: .md
+ format_name: myst
+ format_version: 0.13
+ jupytext_version: 1.16.7
+kernelspec:
+ display_name: Python 3 (ipykernel)
+ language: python
+ name: python3
+---
+
+(phillips_credible_policies)=
+```{raw} jupyter
+
+```
+
+# Credible Government Policies
+
+```{index} single: Phillips Curve; Credible Policies
+```
+
+```{contents} Contents
+:depth: 2
+```
+
+## Overview
+
+{doc}`phillips_credibility` left the government in a trap.
+
+Without a technology for committing itself, a government that re-optimizes every period ends
+at the Nash outcome, inflating at rate $\theta U^*$ and getting nothing for it.
+
+That was a one-period story, and it invited an obvious objection: a government that expects to
+face the same public again tomorrow has a *reputation* to protect.
+
+This lecture, following chapter 4 of {cite}`Sargent1999`, takes the objection seriously.
+
+We repeat the Kydland–Prescott economy forever, let the government's current action depend on
+the whole history of outcomes, and ask which outcome paths can be supported as **subgame
+perfect equilibria**.
+
+The answer is not the one the objection anticipated.
+
+Reputation does not rescue the Ramsey outcome, and it does not confirm the Nash outcome either.
+
+It delivers a *continuum* of equilibrium values, some better than Nash and some considerably
+worse, with no principle inside the model to select among them.
+
+That is the conclusion this lecture exists to establish, and Sargent states it in one sentence:
+
+> My conclusion is that the multiplicity of credible plans replaces pessimism with agnosticism.
+
+It matters for the argument of the whole suite.
+
+{doc}`phillips_two_stories` rests part of its case against the *triumph of natural-rate theory*
+on exactly this weakness: a theory with so many equilibria makes few predictions, so a story
+that depends on policy makers having learned the right one is doing work the theory cannot do
+for it.
+
+### What we build
+
+The machinery is the recursive method of Abreu, Pearce, and Stacchetti
+{cite}`Abreu,APS1990`, which computes not an optimal *value* but the whole *set* of equilibrium
+values.
+
+Ordinary dynamic programming iterates an operator that maps continuation *values* into values.
+
+APS iterate an operator that maps *sets* of continuation values into sets of values, and the
+set of subgame perfect equilibrium values is its largest fixed point.
+
+```{note}
+Making a promised value into a state variable, and then doing dynamic programming in that
+state, is the technique sometimes called **dynamic programming squared**.
+
+It is the organizing idea of the QuantEcon lectures on [Stackelberg
+plans](https://python-advanced.quantecon.org/dyn_stack.html) and of the recursive treatments of
+Ramsey problems in {cite}`Ljungqvist2012`.
+
+This lecture is a compact instance: the state is a promised value, and the object we compute is
+a set.
+```
+
+We then use the machinery three ways.
+
+We construct particular equilibria by guess-and-verify — infinite repetition of Nash, infinite
+repetition of something better, and Abreu's stick-and-carrot, which is *worse* than Nash.
+
+We compute the best and worst equilibrium values from two small programming problems, and check
+them against a direct iteration of the APS operator.
+
+And we exhibit three quite different equilibria that all attain the same worst value, which is
+the sharpest illustration of how little the equilibrium concept pins down.
+
+```{note}
+Sargent's own advice to the reader is worth passing on: this chapter is difficult if you have
+not seen the theory before, and a reader willing to accept its verdict — that the theory of
+credible policy yields agnosticism rather than a prediction — can go straight to
+{doc}`phillips_adaptive` without losing the thread of the argument.
+```
+
+Let's start with our imports:
+
+```{code-cell} ipython3
+import matplotlib.pyplot as plt
+import numpy as np
+from typing import NamedTuple
+```
+
+## The repeated economy
+
+The one-period economy is the one from {doc}`phillips_credibility`.
+
+Let $(U, y, x)$ be unemployment, inflation, and the public's expectation of inflation.
+
+Unemployment obeys the expectations-augmented Phillips curve $U = U^* - \theta(y - x)$, and the
+government's one-period payoff, written as a function of $(x, y)$, is
+
+```{math}
+:label: cp_r
+
+r(x, y) = -\frac{1}{2}\left[\bigl(U^* - \theta(y - x)\bigr)^2 + y^2\right].
+```
+
+The government's one-period best response to an expectation $x$ is
+
+```{math}
+:label: cp_B
+
+B(x) = \frac{\theta\,(U^* + \theta x)}{\theta^2 + 1},
+```
+
+the Nash outcome is $y^N = \theta U^*$, and the Ramsey outcome is $y^R = 0$.
+
+Two things are new.
+
+First, the economy repeats for $t = 1, 2, \ldots$, and the government ranks outcome paths
+$(x, y) = \{x_t, y_t\}_{t=1}^\infty$ by
+
+```{math}
+:label: cp_value
+
+V^g(x, y) = (1 - \delta)\sum_{t=1}^{\infty}\delta^{t-1} r(x_t, y_t),
+\qquad \delta \in (0, 1).
+```
+
+The factor $(1-\delta)$ puts $V^g$ in the same units as a one-period payoff, so values and
+one-period returns can be compared directly.
+
+Second, inflation is confined to a bounded interval $Y = [0, y^\#]$.
+
+The lower bound is the Ramsey outcome; the upper bound $y^\#$ is what makes the government's
+problem non-trivial, as we shall see when we look for the *worst* equilibrium.
+
+```{code-cell} ipython3
+class Model(NamedTuple):
+ θ: float = 1.25 # slope of the Phillips curve
+ U_star: float = 5.5 # natural rate of unemployment
+ y_max: float = 10.0 # y^#, the highest admissible inflation rate
+ δ: float = 0.95 # discount factor
+
+
+def r(m, x, y):
+ "One-period government payoff when the public expects x and inflation is y."
+ U = m.U_star - m.θ * (y - x)
+ return -0.5 * (U**2 + y**2)
+
+
+def B(m, x):
+ "The government's one-period best response to expected inflation x."
+ return m.θ * (m.U_star + m.θ * x) / (m.θ**2 + 1)
+
+
+def y_nash(m):
+ return m.θ * m.U_star
+```
+
+Two payoff schedules do all the work below, both read as functions of a single inflation rate
+$y$ that the public has come to expect.
+
+The first is the **rational-expectations payoff** $r(y, y)$: what the government gets if it
+delivers the inflation the public expects.
+
+The second is the **deviation payoff** $r(y, B(y))$: what it gets if the public expects $y$ but
+the government yields to temptation and best-responds.
+
+Both have closed forms worth recording, because they explain everything that follows:
+
+```{math}
+:label: cp_schedules
+
+r(y, y) = -\frac{1}{2}\left(U^{*2} + y^2\right),
+\qquad
+r\bigl(y, B(y)\bigr) = -\frac{\left(U^* + \theta y\right)^2}{2\left(1 + \theta^2\right)} .
+```
+
+```{code-cell} ipython3
+def r_keep(m, y):
+ "Payoff from delivering the expected inflation rate: r(y, y)."
+ return -0.5 * (m.U_star**2 + y**2)
+
+
+def r_cheat(m, y):
+ "Payoff from best-responding to an expectation of y: r(y, B(y))."
+ return -(m.U_star + m.θ * y)**2 / (2 * (1 + m.θ**2))
+
+
+m = Model()
+ys = np.linspace(0, m.y_max, 7)
+print("closed forms agree with the definitions:",
+ np.allclose(r_keep(m, ys), r(m, ys, ys)),
+ np.allclose(r_cheat(m, ys), r(m, ys, B(m, ys))))
+```
+
+Both schedules fall as $y$ rises, but at different rates, and they cross exactly at the Nash
+rate, where the temptation to deviate vanishes because $B(y^N) = y^N$.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "The two payoff schedules: delivering expected inflation against deviating from it"
+ name: fig-cpol-schedules
+---
+grid = np.linspace(0, m.y_max, 400)
+yN = y_nash(m)
+v_R, v_N = r_keep(m, 0.0), r_keep(m, yN)
+v_lo = r_cheat(m, m.y_max)
+
+fig, ax = plt.subplots(figsize=(7.5, 5.5))
+ax.plot(grid, r_keep(m, grid), 'C0', lw=1.6, label='$r(y, y)$: deliver what is expected')
+ax.plot(grid, r_cheat(m, grid), 'C1', lw=1.6, label='$r(y, B(y))$: deviate')
+ax.plot(0.0, v_R, 'ko', ms=6)
+ax.annotate('$v^R$', (0.0, v_R), textcoords='offset points', xytext=(8, 4))
+ax.plot(yN, v_N, 'ko', ms=6)
+ax.annotate('$v^N$', (yN, v_N), textcoords='offset points', xytext=(8, 4))
+ax.plot(m.y_max, v_lo, 'ko', ms=6)
+ax.annotate('$v_{min}$', (m.y_max, v_lo), textcoords='offset points',
+ xytext=(-34, 4))
+ax.axvline(yN, color='k', lw=0.6, ls=':')
+ax.set_xlabel('inflation rate $y$ expected by the public')
+ax.set_ylabel('one-period payoff')
+ax.set_title('Figure 4.1: the two payoff schedules and the worst equilibrium value')
+ax.legend(loc='lower left')
+plt.show()
+```
+
+The vertical gap between the curves is the government's one-period temptation to inflate less
+than expected.
+
+It is zero at $y^N$, and it widens as expected inflation rises above the Nash rate — which is
+why the *worst* equilibrium will turn out to live at the top of $Y$.
+
+## Recursive strategies and promised values
+
+A government strategy must be able to condition today's action on the entire past.
+
+Carrying whole histories around is unmanageable, so we follow {cite}`Sargent1999` in restricting
+attention to strategies with a recursive representation — a restriction that costs nothing,
+since it excludes no equilibrium *values*.
+
+```{prf:definition} Recursive government strategy
+:label: cp_strategy
+
+A recursive government strategy is a pair of functions $\sigma = (\sigma_1, \sigma_2)$ together
+with an initial condition $v_1$, with the structure
+
+$$
+v_1 \in \mathbb{R} \text{ given}, \qquad
+y_t = \sigma_1(v_t), \qquad
+v_{t+1} = \sigma_2(v_t, x_t, y_t),
+$$
+
+where $v_t$ is a state variable that summarizes the history of outcomes before $t$.
+```
+
+Each member of the private sector knows $v_t$ and knows $\sigma$, and so forecasts
+
+```{math}
+:label: cp_expect
+
+x_t = \sigma_1(v_t) .
+```
+
+Equation {eq}`cp_expect` builds rational expectations into the private sector: the public is
+never surprised along the equilibrium path.
+
+A strategy $(\sigma, v_1)$ generates an entire outcome path, and hence a value
+$V^g(\sigma, v_1)$ through {eq}`cp_value`.
+
+The theory of credible policy now does something that at first looks like a trick.
+
+It ties the past to the future by requiring the state variable to *be* the value it generates,
+
+```{math}
+:label: cp_fixedpt
+
+v = V^g(\sigma, v).
+```
+
+So $v$ works two shifts at once.
+
+In $v_{t+1} = \sigma_2(v_t, x_t, y_t)$ it is a bookkeeping device that records what has
+happened.
+
+In {eq}`cp_fixedpt` it is a **promised value** — the discounted future the government is owed
+on entering the period.
+
+Reading it the second way is what makes the machinery go: the strategy hands the government
+current and future outcomes that make it *want* to do what is expected of it.
+
+```{note}
+Two readings of $\sigma$ are available and, within an equilibrium, impossible to tell apart.
+
+It can be a decision rule that the government chooses, or a description of a system of public
+expectations to which the government conforms.
+
+We return to this ambiguity in the interpretation section, because it is where the theory's
+agnosticism comes from.
+```
+
+### Historical antecedents
+
+Formalizing credibility was an achievement of the 1980s, but a sophisticated grasp of the idea
+is much older.
+
+In 1784 the finance minister Jacques Necker explained to Louis XVI why an absolute monarch, free
+to renege on debt payments, found it hard to borrow {cite:p}`SargentVelde1995`:
+
+> Therefore one can rekindle or sustain public trust only by giving reassurances on the
+> sovereign's intentions, and by proving that no motive can incite him to fail his obligations.
+
+Every clause of that sentence is in the modern definition.
+
+A king proves he will never fail his obligations by never *wanting* to fail them — by
+conforming to a system of expectations that carries its own incentive constraints, so that at
+every date and in every contingency his current payoff plus continuation value is higher if he
+confirms expectations than if he disappoints them.
+
+```{prf:definition} Subgame perfect equilibrium
+:label: cp_spe
+
+A recursive strategy with promised value $v$ is a **subgame perfect equilibrium** (SPE) if and
+only if
+
+(a) $\sigma_2(v, \sigma_1(v), \eta)$ is itself a value attainable by a subgame perfect
+equilibrium, for every $\eta \in Y$; and
+
+(b) writing $y = \sigma_1(v)$,
+
+$$
+v = (1-\delta)\, r(y, y) + \delta\, \sigma_2(v, y, y)
+ \;\geq\; (1-\delta)\, r(y, \eta) + \delta\, \sigma_2(v, y, \eta),
+ \qquad \forall\, \eta \in Y .
+$$
+```
+
+The definition attaches four objects to an equilibrium: a promised value $v$; a first-period
+outcome $(y, y)$; a continuation value $v'$ if the prescribed outcome is observed; and another
+continuation value $\tilde v$ if it is not.
+
+In those terms condition (b) reads
+
+$$
+v = (1-\delta) r(y, y) + \delta v' \;\geq\; (1-\delta) r(y, \eta) + \delta \tilde v,
+\qquad \forall \eta \in Y,
+$$
+
+which simply says the government does better adhering than departing.
+
+Condition (a) says the continuation values must themselves be equilibrium values.
+
+The definition is circular — equilibrium values appear on both sides — and that circularity is
+exactly what recursivity buys, and what APS learned to exploit.
+
+## The Abreu–Pearce–Stacchetti method
+
+Dynamic programming computes an optimal value function by iterating the Bellman equation, a map
+that turns tomorrow's value function into today's.
+
+APS adapted the idea to equilibrium *sets*.
+
+Start with a candidate set $W \subset \mathbb{R}$ of continuation values.
+
+Pick a first-period rational-expectations outcome $(y, y)$ and two continuation values from
+$W$: a $w_1$ to reward adherence and a $w_2$ to punish deviation.
+
+If $w_1$ is high enough and $w_2$ low enough, the pair supports $y$, and delivers the value
+
+$$
+w = (1-\delta) r(y, y) + \delta w_1 \;\geq\; (1-\delta) r(y, \eta) + \delta w_2,
+\qquad \forall \eta \in Y .
+$$
+
+```{prf:definition} Admissibility
+:label: cp_admissible
+
+A pair $(y, w)$ is **admissible** with respect to a set of continuation values $W$ if there
+exist $w_1, w_2 \in W$ with
+$w = (1-\delta) r(y,y) + \delta w_1 \geq (1-\delta) r(y,\eta) + \delta w_2$ for all
+$\eta \in Y$.
+```
+
+Let $B(W)$ collect the $w$ components of all admissible pairs.
+
+This construction builds in condition (b) of {prf:ref}`cp_spe` but ignores condition (a),
+because the continuation values were drawn from an arbitrary set.
+
+```{prf:definition} Self-generation
+:label: cp_selfgen
+
+A set $W$ of prospective continuation values is **self-generating** if $W \subseteq B(W)$.
+```
+
+Every value in a self-generating set is supported by continuation values drawn from that same
+set — which is condition (a).
+
+APS proved that the set $V$ of SPE values is the largest self-generating set, that $B$ maps
+compact sets into compact sets, that $B$ is monotone ($W_2 \subseteq W_1$ implies
+$B(W_2) \subseteq B(W_1)$), and that starting from any $W_0$ with $B(W_0) \subseteq W_0$ the
+iteration $W_j = B(W_{j-1})$ converges monotonically to $V = B(V)$.
+
+### Computing the operator
+
+Two simplifications make $B$ easy to evaluate here.
+
+Because the government's temptation is worst when it best-responds, the constraint in
+{prf:ref}`cp_admissible` binds hardest at $\eta = B(y)$, so we may replace "for all $\eta$" by
+that single deviation.
+
+And because a lower punishment relaxes the constraint, we may always set $w_2$ to the smallest
+element of $W$.
+
+Writing $W = [\underline w, \overline w]$, the outcome $y$ is then admissible provided some
+$w_1 \in W$ satisfies
+
+```{math}
+:label: cp_bound
+
+w_1 \;\geq\; \frac{(1-\delta)\left[r(y, B(y)) - r(y, y)\right]}{\delta} + \underline w
+\;\equiv\; \ell(y) ,
+```
+
+and the values it generates sweep out the interval
+$\left[(1-\delta) r(y,y) + \delta \max(\underline w, \ell(y)),\;
+(1-\delta) r(y,y) + \delta \overline w\right]$.
+
+```{code-cell} ipython3
+def B_operator(m, W, n_grid=4001):
+ """
+ One application of the APS operator to an interval W = [w_lo, w_hi].
+
+ Returns the new interval, or None if no first-period outcome is admissible.
+ """
+ w_lo, w_hi = W
+ y = np.linspace(0.0, m.y_max, n_grid)
+ ℓ = (1 - m.δ) * (r_cheat(m, y) - r_keep(m, y)) / m.δ + w_lo # equation (10)
+ ok = ℓ <= w_hi # admissible outcomes
+ if not ok.any():
+ return None
+ keep = (1 - m.δ) * r_keep(m, y)
+ lows = keep[ok] + m.δ * np.maximum(w_lo, ℓ[ok])
+ highs = keep[ok] + m.δ * w_hi
+ return lows.min(), highs.max()
+
+
+def solve_aps(m, W0=None, tol=1e-12, max_iter=10_000):
+ "Iterate the APS operator to its largest fixed point."
+ W = (r_keep(m, m.y_max), 0.0) if W0 is None else W0
+ for it in range(max_iter):
+ W_new = B_operator(m, W)
+ if W_new is None:
+ return None, it
+ if max(abs(W_new[0] - W[0]), abs(W_new[1] - W[1])) < tol:
+ return W_new, it
+ W = W_new
+ return W, max_iter
+```
+
+We start from $W_0 = [r(y^\#, y^\#),\, 0]$, which is large enough to contain every equilibrium
+value, and iterate.
+
+```{code-cell} ipython3
+W, iters = solve_aps(m)
+print(f"converged in {iters} iterations")
+print(f"set of SPE values V = [{W[0]:.4f}, {W[1]:.4f}]")
+```
+
+## The best and the worst equilibrium
+
+APS's iteration is general but tells us little about *why* the set ends where it does.
+
+For this economy both endpoints can be found by hand, and the arguments are instructive.
+
+### The worst
+
+The worst equilibrium value solves
+
+$$
+\underline v = \min_{y \in Y,\; v_1 \in V}\ \left[(1-\delta) r(y,y) + \delta v_1\right]
+\quad\text{subject to}\quad
+(1-\delta) r(y,y) + \delta v_1 \geq (1-\delta) r(y, B(y)) + \delta \underline v ,
+$$
+
+where the *worst* value is used as the continuation in the event of a deviation — the harshest
+punishment available.
+
+The minimum is attained where the constraint binds, and at that point the two sides collapse to
+$\underline v = r(y, B(y))$ for some $y$.
+
+So finding the worst equilibrium reduces to a one-dimensional problem:
+
+```{math}
+:label: cp_worst
+
+\underline v = \min_{y \in Y}\ r\bigl(y, B(y)\bigr)
+= -\frac{\left(U^* + \theta y^\#\right)^2}{2\left(1 + \theta^2\right)} ,
+```
+
+where the closed form follows from {eq}`cp_schedules`, and the minimizing action is $y^\#$
+because the deviation payoff falls monotonically in $y$.
+
+Now the role of the upper bound $y^\#$ is clear.
+
+It is what keeps the worst punishment finite; without it the government could be threatened
+with arbitrarily bad outcomes and almost anything could be supported.
+
+```{prf:proposition} The worst SPE is self-enforcing
+:label: cp_prop_worst
+
+At the worst equilibrium the continuation value following a deviation equals the initial
+promised value, so a deviation simply restarts the equilibrium.
+```
+
+### The best
+
+Given $\underline v$ as the threat, the best value solves
+
+```{math}
+:label: cp_best
+
+\overline v = \max_{y \in Y}\ r(y, y)
+\quad\text{subject to}\quad
+r(y, y) \geq (1-\delta) r\bigl(y, B(y)\bigr) + \delta \underline v ,
+```
+
+where we have used the fact that the best value must reward adherence with itself, so that
+$\overline v = (1-\delta) r(y,y) + \delta \overline v = r(y,y)$.
+
+```{prf:proposition} The best SPE is self-rewarding
+:label: cp_prop_best
+
+At the best equilibrium the continuation value following adherence equals the promised value.
+```
+
+```{code-cell} ipython3
+def worst_value(m):
+ "The worst SPE value and the action that attains it."
+ return r_cheat(m, m.y_max), m.y_max
+
+
+def best_value(m, v_lo, n_grid=200_001):
+ "The best SPE value, given the worst value as the punishment threat."
+ y = np.linspace(0.0, m.y_max, n_grid)
+ feasible = r_keep(m, y) >= (1 - m.δ) * r_cheat(m, y) + m.δ * v_lo
+ if not feasible.any():
+ return None, None
+ vals = np.where(feasible, r_keep(m, y), -np.inf)
+ k = vals.argmax()
+ return vals[k], y[k]
+
+
+v_lo, y_sharp = worst_value(m)
+v_hi, y_best = best_value(m, v_lo)
+
+print(f"worst value v_lo = {v_lo:.4f} attained with y = {y_sharp:.2f}")
+print(f"best value v_hi = {v_hi:.4f} attained with y = {y_best:.2f}")
+print(f"APS iteration gave [{W[0]:.4f}, {W[1]:.4f}]")
+print(f"the two agree: {np.allclose(W, (v_lo, v_hi), atol=1e-6)}")
+```
+
+The programming problems and the set iteration agree, and at this discount factor the best
+equilibrium is Ramsey itself.
+
+## Examples of recursive equilibria
+
+We now construct particular equilibria by guess-and-verify, in the order in which the literature
+found them.
+
+### Infinite repetition of Nash
+
+The easiest equilibrium repeats the one-period Nash outcome forever.
+
+Take $v_1 = v^N = r(y^N, y^N)$, $\sigma_1(v) = y^N$ for every $v$, and
+$\sigma_2(v, x, y) = v^N$ for every $(v, x, y)$.
+
+Condition (a) holds by construction, and condition (b) collapses to
+$r(y^N, y^N) \geq r(y^N, B(y^N))$, which holds *with equality* because $y^N$ is a fixed point of
+the best response map.
+
+Nothing deters the government here, and nothing needs to: it is already doing what it most
+wants to do.
+
+### Infinite repetition of something better
+
+Let $v^b = r(y^b, y^b) > v^N$ be a better-than-Nash value, and suppose
+
+```{math}
+:label: cp_bg
+
+r\bigl(y^b, B(y^b)\bigr) - r(y^b, y^b)
+\;\leq\; \frac{\delta}{1-\delta}\left(v^b - v^N\right).
+```
+
+The left side is the one-period gain from cheating; the right side is the discounted loss from
+reverting to Nash forever.
+
+When {eq}`cp_bg` holds, the strategy that prescribes $y^b$ while the promised value is $v^b$ and
+reverts to Nash after any deviation is an SPE.
+
+{cite:t}`BarroGordon1983` studied the case $y^b = y^R$: anticipated reversion to Nash supports
+Ramsey forever.
+
+Whether it does depends on patience, and for this economy the threshold has a remarkably clean
+form.
+
+Setting $y^b = 0$ in {eq}`cp_bg` and using the closed forms {eq}`cp_schedules`, every $U^*$ and
+every scale factor cancels, leaving
+
+```{math}
+:label: cp_cutoff
+
+\delta \;\geq\; \delta^\star = \frac{1}{2 + \theta^2} .
+```
+
+```{code-cell} ipython3
+def supports_ramsey_by_nash(m):
+ "Does reversion to Nash sustain Ramsey forever?"
+ gain = r_cheat(m, 0.0) - r_keep(m, 0.0)
+ loss = m.δ / (1 - m.δ) * (r_keep(m, 0.0) - r_keep(m, y_nash(m)))
+ return gain <= loss
+
+
+δ_star = 1 / (2 + m.θ**2)
+print(f"closed-form cutoff δ* = 1/(2 + θ²) = {δ_star:.6f}\n")
+for δ in (0.20, 0.27, δ_star - 1e-6, δ_star + 1e-6, 0.50, 0.95):
+ md = m._replace(δ=δ)
+ print(f" δ = {δ:.8f}: Nash reversion supports Ramsey? "
+ f"{supports_ramsey_by_nash(md)}")
+```
+
+Above $\delta^\star \approx 0.28$ the threat of reverting to Nash is enough to hold the
+government at zero inflation; below it, the threat is too weak.
+
+### Something worse: Abreu's stick and carrot
+
+When reversion to Nash is not a strong enough threat, {cite:t}`Abreu` asked whether some
+*worse* equilibrium could be used as the punishment.
+
+His device is a stick-and-carrot strategy: the government is required to inflate at the maximum
+rate $y^\#$ for one period — the stick — after which it is rewarded with the Ramsey outcome
+forever.
+
+The value is
+
+```{math}
+:label: cp_abreu
+
+\tilde v = (1-\delta)\, r(y^\#, y^\#) + \delta\, v^R ,
+```
+
+and the punishment for refusing the stick is to *restart* the whole strategy, so the deviation
+continuation value is $\tilde v$ itself.
+
+That self-referential choice is what makes the arithmetic so simple.
+
+Substituting $\tilde v$ on both sides of the incentive constraint, the $\delta$ terms cancel and
+the strategy is an equilibrium precisely when
+$\tilde v \geq r\bigl(y^\#, B(y^\#)\bigr) = \underline v$.
+
+```{code-cell} ipython3
+def abreu_value(m):
+ "Value of the stick-and-carrot strategy: one period at y^#, then Ramsey."
+ return (1 - m.δ) * r_keep(m, m.y_max) + m.δ * r_keep(m, 0.0)
+
+
+for δ in (0.20, 0.95):
+ md = m._replace(δ=δ)
+ va, vn = abreu_value(md), r_keep(md, y_nash(md))
+ print(f"δ = {δ}: v_abreu = {va:8.4f} v^N = {vn:8.4f} "
+ f"worse than Nash? {va < vn} is an SPE? {va >= worst_value(md)[0]}")
+```
+
+At $\delta = 0.95$ the stick-and-carrot value is *better* than Nash, because a single bad period
+is heavily outweighed by an eternity of Ramsey.
+
+At $\delta = 0.2$ it is far *worse* than Nash — and that is the point.
+
+A punishment worse than Nash is a stronger deterrent than reversion to Nash, so it can support
+outcomes that Nash reversion cannot.
+
+```{code-cell} ipython3
+m2 = m._replace(δ=0.20)
+
+def supports_ramsey_with_threat(m, threat):
+ "Does the threat value sustain Ramsey forever?"
+ return r_keep(m, 0.0) >= (1 - m.δ) * r_cheat(m, 0.0) + m.δ * threat
+
+print(f"at δ = 0.2:")
+print(f" threat = Nash ({r_keep(m2, y_nash(m2)):8.4f}): "
+ f"supports Ramsey? {supports_ramsey_with_threat(m2, r_keep(m2, y_nash(m2)))}")
+print(f" threat = Abreu ({abreu_value(m2):8.4f}): "
+ f"supports Ramsey? {supports_ramsey_with_threat(m2, abreu_value(m2))}")
+```
+
+At $\delta = 0.2$, an impatient government cannot be held at Ramsey by the prospect of Nash, but
+*can* be held there by the prospect of Abreu's stick.
+
+Two definitions name the structures we have just met.
+
+```{prf:definition} Self-enforcing and self-rewarding
+:label: cp_selfenf
+
+A recursive SPE is **self-enforcing** if the continuation value following a deviation equals the
+initial promised value, and **self-rewarding** if the continuation value following adherence
+equals the promised value.
+```
+
+Abreu's stick-and-carrot is self-enforcing; infinite repetition of a better-than-Nash outcome is
+self-rewarding.
+
+{prf:ref}`cp_prop_worst` and {prf:ref}`cp_prop_best` say that these two structures are not
+curiosities: they are exactly what the extreme equilibria look like.
+
+## Multiplicity
+
+Two layers of multiplicity inhabit this theory.
+
+There is a continuum of equilibrium *values*, and — the sharper point — many different outcome
+*paths* attain the same value.
+
+To see the second, we construct three equilibria that all deliver the worst value
+$\underline v$.
+
+All three begin the same way: the first-period promised value is $\underline v$, the prescribed
+action is $y^\#$, and the continuation value $v_2$ solves
+$\underline v = (1-\delta) r(y^\#, y^\#) + \delta v_2$.
+
+They differ in what they do afterwards.
+
+```{code-cell} ipython3
+def next_value(m, v):
+ "Solve v = (1-δ) r(y^#, y^#) + δ v' for the continuation value v'."
+ return (v - (1 - m.δ) * r_keep(m, m.y_max)) / m.δ
+
+
+def action_delivering(m, v):
+ "The inflation rate ỹ with r(ỹ, ỹ) = v."
+ return np.sqrt(max(-2.0 * v - m.U_star**2, 0.0))
+
+
+def method_1(m, v_lo, v_hi, max_steps=500):
+ "Climb at y^# until one more step would overshoot v_hi, then settle at v_hi."
+ v, y = [v_lo], []
+ for _ in range(max_steps):
+ nxt = next_value(m, v[-1])
+ if nxt > v_hi:
+ break
+ y.append(m.y_max)
+ v.append(nxt)
+ target = (v[-1] - m.δ * v_hi) / (1 - m.δ) # r(ỹ, ỹ) for the switching period
+ y.append(action_delivering(m, target))
+ v.append(v_hi)
+ y.append(action_delivering(m, v_hi)) # stay at v_hi forever after
+ return np.array(v), np.array(y)
+
+
+def method_2(m, v_lo, max_steps=500):
+ "Climb at y^# only until the promised value first exceeds v^N, then freeze."
+ v_N = r_keep(m, y_nash(m))
+ v, y = [v_lo], []
+ for _ in range(max_steps):
+ nxt = next_value(m, v[-1])
+ y.append(m.y_max)
+ v.append(nxt)
+ if nxt > v_N:
+ break
+ y.append(action_delivering(m, v[-1])) # freeze at v** forever
+ return np.array(v), np.array(y)
+
+
+def method_3(m, v_lo):
+ "One period at y^#, then freeze immediately at v_2."
+ v_2 = next_value(m, v_lo)
+ return (np.array([v_lo, v_2, v_2]),
+ np.array([m.y_max, action_delivering(m, v_2),
+ action_delivering(m, v_2)]))
+```
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Three subgame perfect equilibria that attain the same worst value
+ name: fig-cpol-three-methods
+---
+paths = {'method 1': method_1(m, v_lo, v_hi),
+ 'method 2': method_2(m, v_lo),
+ 'method 3': method_3(m, v_lo)}
+
+fig, axes = plt.subplots(2, 3, figsize=(13, 6.5), sharex='col')
+for k, (name, (v, y)) in enumerate(paths.items()):
+ axes[0, k].plot(range(1, len(v) + 1), v, 'C0o-', ms=3.5, lw=1)
+ axes[0, k].axhline(v_lo, color='k', ls='--', lw=0.8)
+ axes[0, k].axhline(v_hi, color='C2', ls=':', lw=0.8)
+ axes[0, k].set_title(f'{name}: continuation values', fontsize=10)
+ axes[1, k].plot(range(1, len(y) + 1), y, 'C1o-', ms=3.5, lw=1)
+ axes[1, k].set_ylim(-0.4, m.y_max + 0.4)
+ axes[1, k].set_xlabel('$t$')
+ axes[1, k].set_title(f'{name}: inflation', fontsize=10)
+axes[0, 0].set_ylabel('promised value $v_t$')
+axes[1, 0].set_ylabel('inflation $y_t$')
+fig.suptitle('Figures 4.2-4.4: three equilibria that all attain the worst '
+ 'value $v_{min}$')
+plt.tight_layout()
+plt.show()
+```
+
+```{code-cell} ipython3
+print(f"Nash value {r_keep(m, y_nash(m)):.4f} at inflation {y_nash(m):.4f}; "
+ f"worst value {v_lo:.4f}\n")
+for name, (v, y) in paths.items():
+ print(f"{name:9s}: {len(y):3d} periods before settling, "
+ f"first-period value {v[0]:.4f}, "
+ f"terminal value {v[-1]:8.4f} at inflation {y[-1]:.4f}")
+```
+
+The three paths could hardly be less alike.
+
+The first climbs at maximum inflation for about sixty periods before easing down to Ramsey.
+
+The second climbs for a shorter spell and then freezes forever at a value just *better* than
+Nash, sustained by an inflation rate just below the Nash rate.
+
+The third inflates at the maximum for a single period and then settles immediately, at a value
+barely above the worst one.
+
+All three are subgame perfect, and all three deliver exactly the same value to the government.
+
+The equilibrium concept has nothing to say about which of them describes the world.
+
+## Numerical example
+
+Here is the numerical example reported in chapter 4 of {cite}`Sargent1999`, computed from
+scratch.
+
+```{code-cell} ipython3
+import pandas as pd
+
+rows = [
+ ("$\\theta$", m.θ, ""),
+ ("$U^*$", m.U_star, ""),
+ ("$y^\\#$", m.y_max, ""),
+ ("$\\delta$", m.δ, ""),
+ ("$y^N$ (Nash inflation)", y_nash(m), "6.8750"),
+ ("$y^R$ (Ramsey inflation)", 0.0, "0"),
+ ("$v^R$", r_keep(m, 0.0), "-15.1250"),
+ ("$v^N$", r_keep(m, y_nash(m)), "-38.7578"),
+ ("$\\underline v$", v_lo, "-63.2195"),
+ ("$v_{\\rm abreu}$", abreu_value(m), "-17.6250"),
+]
+pd.DataFrame(rows, columns=["object", "computed", "reported in the book"]).set_index("object")
+```
+
+```{code-cell} ipython3
+print(f"cutoff discount factor δ* : {1 / (2 + m.θ**2):.4f} (book reports 0.2807)")
+print(f"Abreu stick-and-carrot at δ = 0.2: {abreu_value(m._replace(δ=0.2)):.4f}"
+ f" (book reports -55.125)")
+```
+
+Every entry reproduces the book.
+
+## Interpretations
+
+The literature on credible plans bears mixed news for the *triumph of natural-rate theory*
+story of {doc}`phillips_two_stories`.
+
+It certainly rescues the government from the pessimism of {doc}`phillips_credibility`: better
+outcomes than Nash are available, and at plausible discount factors Ramsey itself is an
+equilibrium.
+
+But it rescues too much.
+
+Values worse than Nash are equilibria too, and between the extremes lies a continuum.
+
+The multitude of outcomes mutes the model empirically, and it undoes the very thing early
+rational-expectations researchers wanted from the hypothesis — the elimination of free
+parameters describing expectations.
+
+### Whose expectations are they?
+
+In his 1979 review of an OECD report edited by Paul McCracken, Lucas protested the report's
+recommendation that "governments should try to promote good expectations," as though
+expectations were an extra set of policy instruments.
+
+In 1979 that protest was well aimed: rational-expectations models then took government policy as
+exogenous and made expectations a *function* of it, linked by cross-equation restrictions.
+
+The theory of credible policy changes the picture in a way that vindicates neither side
+cleanly.
+
+It turns systems of expectations into free parameters that influence outcomes — but into
+parameters that nobody inside the model gets to choose.
+
+The government complies with equilibrium expectations about its own behavior.
+
+Within an equilibrium, the government's strategy is simultaneously a decision rule and a
+description of the public's expectations, and the two cannot be disentangled.
+
+As Sargent puts it: the authors of the McCracken report believed in multiplicity and
+manipulation, while Lucas doubted both; the literature on credible plans supports multiplicity
+but not manipulation.
+
+### Remedies
+
+Reputation alone is a weak foundation for anti-inflation policy, and that weakness has inspired
+proposals to change the game rather than to hope for a good equilibrium within it.
+
+{cite:t}`Rogoff1985` proposed delegating monetary policy to someone who cares less about
+unemployment than society does.
+
+Assigning it to someone unaware even of a *temporary* tradeoff would be equally effective, and
+Alan Blinder later suggested a related device: an authority who knows the natural rate and never
+wants unemployment to differ from it.
+
+Maintaining a pool of potential central bankers with different inflation–unemployment
+preferences can also improve outcomes {cite:p}`BarroGordon1983`.
+
+### Where this leaves us
+
+Every remedy just listed changes the *institutions*, not the theory.
+
+Within the theory, the multiplicity is irreducible, and Sargent locates its source precisely: it
+stems from the rationality imputed to *everyone* in the system.
+
+Perfection is what generates the continuum, because a system of expectations sophisticated
+enough to support one equilibrium is sophisticated enough to support many.
+
+So the book retreats from perfection.
+
+The remaining lectures replace fully rational participants with ones whose understanding of the
+economy is limited — first the public in {doc}`phillips_adaptive`, then both sides in
+{doc}`phillips_misspecified` and {doc}`phillips_self_confirming`, and finally a government that
+estimates its model in real time in {doc}`phillips_learning`.
+
+Those models are closer to what Lucas had in mind when he criticized the McCracken report, and,
+unlike the theory of this lecture, they make predictions.
+
+## Exercises
+
+```{exercise-start}
+:label: cpol_ex1
+```
+
+The cutoff {eq}`cp_cutoff` claims that reversion to Nash sustains Ramsey forever if and only if
+$\delta \geq 1/(2 + \theta^2)$ — a threshold that depends on the slope of the Phillips curve
+alone, and not on the natural rate $U^*$ or the upper bound $y^\#$.
+
+Verify both claims numerically.
+
+For a grid of $\theta$ values, find the smallest $\delta$ on a fine grid at which reversion to
+Nash supports Ramsey, and compare it with $1/(2+\theta^2)$.
+
+Then check that changing $U^*$ leaves the answer untouched.
+
+```{exercise-end}
+```
+
+```{solution-start} cpol_ex1
+:class: dropdown
+```
+
+```{code-cell} ipython3
+δ_grid = np.linspace(0.01, 0.99, 9801)
+
+rows = []
+for θ in (0.5, 1.0, 1.25, 2.0, 3.0):
+ for U_star in (5.5, 11.0):
+ md = Model(θ=θ, U_star=U_star)
+ ok = [δ for δ in δ_grid
+ if supports_ramsey_by_nash(md._replace(δ=δ))]
+ rows.append((θ, U_star, min(ok), 1 / (2 + θ**2)))
+
+pd.DataFrame(rows, columns=["$\\theta$", "$U^*$", "smallest $\\delta$ found",
+ "$1/(2+\\theta^2)$"]).round(4)
+```
+
+The grid search matches the closed form to within the grid spacing, and doubling $U^*$ changes
+nothing.
+
+The reason both $U^*$ and $y^\#$ drop out is visible in {eq}`cp_schedules`: the one-period
+temptation at $y = 0$ and the Nash–Ramsey value gap are both proportional to $U^{*2}$, so the
+scale cancels from the inequality, and $y^\#$ never enters because neither side involves the
+upper bound.
+
+A steeper Phillips curve — a larger $\theta$ — makes the Nash outcome worse relative to Ramsey,
+which strengthens the threat and lets a *less* patient government be held at zero inflation.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: cpol_ex2
+```
+
+The set of equilibrium values shrinks as the government becomes impatient.
+
+Use `solve_aps` to compute the set $V = [\underline v, \overline v]$ for a grid of discount
+factors, and plot both endpoints against $\delta$, marking the Nash and Ramsey values.
+
+At roughly what discount factor does the best equilibrium value stop being Ramsey?
+
+What happens to the set as $\delta \to 0$, and why is that the answer you should expect?
+
+```{exercise-end}
+```
+
+```{solution-start} cpol_ex2
+:class: dropdown
+```
+
+```{code-cell} ipython3
+δs = np.linspace(0.02, 0.98, 49)
+lows, highs = [], []
+for δ in δs:
+ Wδ, _ = solve_aps(m._replace(δ=δ))
+ lows.append(Wδ[0])
+ highs.append(Wδ[1])
+
+fig, ax = plt.subplots(figsize=(8.5, 5))
+ax.fill_between(δs, lows, highs, alpha=0.2, color='C0', label='set of SPE values $V$')
+ax.plot(δs, highs, 'C0', lw=1.4)
+ax.plot(δs, lows, 'C0', lw=1.4)
+ax.axhline(r_keep(m, 0.0), color='C2', ls=':', lw=1.2, label='Ramsey $v^R$')
+ax.axhline(r_keep(m, y_nash(m)), color='k', ls='--', lw=1, label='Nash $v^N$')
+ax.set_xlabel(r'discount factor $\delta$')
+ax.set_ylabel('value')
+ax.set_title('SPE values by discount factor')
+ax.legend()
+plt.show()
+```
+
+```{code-cell} ipython3
+best_is_ramsey = [δ for δ, h in zip(δs, highs)
+ if np.isclose(h, r_keep(m, 0.0), atol=1e-6)]
+print(f"best value equals Ramsey for δ ≥ {min(best_is_ramsey):.3f}")
+print(f"closed-form cutoff for Nash reversion: δ* = {1/(2+m.θ**2):.3f}")
+```
+
+The best equilibrium value equals Ramsey down to a discount factor well *below* the
+$\delta^\star \approx 0.28$ of {eq}`cp_cutoff`, and the reason is Abreu.
+
+The cutoff $\delta^\star$ was derived using reversion to *Nash* as the threat; the APS set uses
+the worst equilibrium as the threat, which is a good deal harsher, so Ramsey survives at lower
+discount factors than Barro and Gordon's argument alone would suggest.
+
+As $\delta \to 0$ the set collapses toward the single point $v^N$.
+
+That is what it must do: an entirely impatient government cares only about the current period,
+promises about the future carry no weight, and the only outcome that can be sustained is the one
+the government would choose myopically — the Nash outcome.
+
+```{solution-end}
+```
+
+```{exercise-start}
+:label: cpol_ex3
+```
+
+Method 3 in the multiplicity section is the boldest of the three: it inflates at $y^\#$ for a
+single period and then freezes forever at a constant inflation rate.
+
+Because it settles immediately, its incentive constraint is the tightest of the three, so it is
+worth checking rather than assuming.
+
+Verify directly that method 3 is a subgame perfect equilibrium: confirm that its first-period
+promise is honoured, that its frozen continuation value lies inside $V$, and that the government
+prefers adherence to deviation in *both* phases, using $\underline v$ as the punishment.
+
+```{exercise-end}
+```
+
+```{solution-start} cpol_ex3
+:class: dropdown
+```
+
+```{code-cell} ipython3
+v_2 = next_value(m, v_lo)
+y_tilde = action_delivering(m, v_2)
+
+# phase 1: promised v_lo, prescribed action y^#, continuation v_2 on adherence
+lhs_1 = (1 - m.δ) * r_keep(m, m.y_max) + m.δ * v_2
+rhs_1 = (1 - m.δ) * r_cheat(m, m.y_max) + m.δ * v_lo
+
+# phase 2: promised v_2, prescribed action ỹ, continuation v_2 on adherence
+lhs_2 = (1 - m.δ) * r_keep(m, y_tilde) + m.δ * v_2
+rhs_2 = (1 - m.δ) * r_cheat(m, y_tilde) + m.δ * v_lo
+
+print(f"v_2 = {v_2:.4f}, ỹ = {y_tilde:.4f}")
+print(f"v_2 lies inside V = [{v_lo:.4f}, {v_hi:.4f}]: "
+ f"{v_lo <= v_2 <= v_hi}")
+print(f"phase 1 delivers the promise: {np.isclose(lhs_1, v_lo)}")
+print(f"phase 1 incentive: {lhs_1:.4f} >= {rhs_1:.4f} -> {lhs_1 >= rhs_1}")
+print(f"phase 2 delivers the promise: {np.isclose(lhs_2, v_2)}")
+print(f"phase 2 incentive: {lhs_2:.4f} >= {rhs_2:.4f} -> {lhs_2 >= rhs_2}")
+print(f"slack in phase 2: {lhs_2 - rhs_2:.6f}")
+```
+
+All the conditions hold, and the first-period promise is delivered exactly.
+
+The margin in the second phase is very thin, and that is not an accident.
+
+The frozen value $v_2$ is only barely above $\underline v$, so the punishment for deviating is
+only barely worse than the equilibrium itself — which is precisely what it means to be near the
+bottom of the set of equilibrium values.
+
+Push the construction any lower and the incentive constraint fails, which is another way of
+seeing why $\underline v$ is where it is.
+
+```{solution-end}
+```
diff --git a/lectures/phillips_drifts_volatilities.md b/lectures/phillips_drifts_volatilities.md
index becf74ee7..a0bb34753 100644
--- a/lectures/phillips_drifts_volatilities.md
+++ b/lectures/phillips_drifts_volatilities.md
@@ -27,6 +27,9 @@ kernelspec:
# Drifts and Volatilities
+```{index} single: Phillips Curve; Drifts and Volatilities
+```
+
```{contents} Contents
:depth: 2
```
@@ -82,10 +85,9 @@ for state-space models by MCMC in {doc}`ar1_bayes` and {doc}`ar1_turningpts`.
We work through the data transformation, prior, sampler, and main empirical
results.
-Let's start with some imports and the path to the data.
+Let's start with some imports and the URL for the data.
```{code-cell} ipython3
-from pathlib import Path
import time
import matplotlib.pyplot as plt
@@ -97,19 +99,11 @@ from scipy.special import expit
from scipy.stats import invwishart
-def locate_data_assets():
- """Find assets from either a MyST build or the repository root."""
- relative = Path('_static/lecture_specific/phillips_drifts_volatilities')
- candidates = (relative, Path('lectures') / relative)
- for candidate in candidates:
- if (candidate / 'NEWQDATA.csv').is_file():
- return candidate
- searched = ', '.join(str(path.resolve()) for path in candidates)
- raise FileNotFoundError(f'NEWQDATA.csv was not found; searched {searched}')
-
-
-asset_path = locate_data_assets()
-data_path = asset_path / 'NEWQDATA.csv'
+data_url = (
+ 'https://raw.githubusercontent.com/QuantEcon/lecture-python.myst/'
+ 'main/lectures/_static/lecture_specific/phillips_drifts_volatilities/'
+ 'NEWQDATA.csv'
+)
```
## Bad policy or bad luck?
@@ -155,13 +149,14 @@ changing volatility for coefficient drift.
So Cogley and Sargent build a model that has room for *both* channels at once,
and they let a Bayesian posterior sort out how much of each the data call for.
+(csdv-model)=
## A VAR with drifting coefficients and stochastic volatility
Let the variables be ordered as nominal interest, transformed unemployment, and
inflation,
$$
-y_t = \begin{bmatrix} i_t & u_t & \pi_t \end{bmatrix}'.
+y_t = \begin{bmatrix} i_t & u_t & \pi_t \end{bmatrix}^\top.
$$
(Here $u_t$ is not the raw unemployment rate but its logit, we define this transformation in the data section below.)
@@ -170,9 +165,9 @@ The measurement equation is a VAR with two lags and date-specific coefficients,
```{math}
:label: csdv_measurement
-y_t = X_t'\theta_t + \varepsilon_t,
+y_t = X_t^\top \theta_t + \varepsilon_t,
\qquad
-X_t' = I_3 \otimes \begin{bmatrix} 1 & y_{t-1}' & y_{t-2}' \end{bmatrix}.
+X_t^\top = I_3 \otimes \begin{bmatrix} 1 & y_{t-1}^\top & y_{t-2}^\top \end{bmatrix}.
```
Each equation has an intercept and six lag coefficients, so $\theta_t$ contains
@@ -358,7 +353,7 @@ directly from the series.
```{code-cell} ipython3
def prepare_data(source, ordering=('i', 'u', 'pi')):
"""Transform a quarterly table and construct the VAR data."""
- if isinstance(source, (str, Path)):
+ if isinstance(source, str):
table = pd.read_csv(source)
else:
table = source.copy()
@@ -393,7 +388,7 @@ def prepare_data(source, ordering=('i', 'u', 'pi')):
}
-data = prepare_data(data_path)
+data = prepare_data(data_url)
data_summary = pd.Series(
{
@@ -569,6 +564,7 @@ prior_summary = pd.Series(
prior_summary.to_frame()
```
+(csdv-sampler)=
## A Metropolis-within-Gibbs sampler
We simulate the posterior by cycling through five parameter blocks used by
@@ -899,7 +895,7 @@ Gaussian truncated to $\mathcal A$,
$$
\pi_{\mathcal A}(z\mid\lambda,Y^T)
= \frac{N(z;m,C)\,\mathbb{1}_{\mathcal A}(z)}
- {\Pr(z\in\mathcal A\mid\lambda,Y^T)}.
+ {\mathbb{P}\{z\in\mathcal A\mid\lambda,Y^T\}}.
$$
That normalizing probability is difficult to compute, but the elliptical
@@ -1174,10 +1170,11 @@ def validate_posterior_arrays(result, periods):
expected_shapes = validate_posterior_arrays(posterior, len(data['dates']))
```
+(csdv-results)=
## What the data say
-We summarize the posterior by its mean coefficient path $E(\theta_t\mid T)$ and
-mean covariance path $E(R_t\mid T)$, and then interpret them
+We summarize the posterior by its mean coefficient path $\mathbb{E}(\theta_t\mid T)$ and
+mean covariance path $\mathbb{E}(R_t\mid T)$, and then interpret them
in the context of the question we asked.
### The rate and structure of drift
@@ -1616,11 +1613,11 @@ f_{\pi\pi}(\omega,t)
s_\pi
(I-A_{t\mid T}e^{-i\omega})^{-1}
\mathcal R_t
-(I-A_{t\mid T}'e^{i\omega})^{-1}
-s_\pi',
+(I-A_{t\mid T}^\top e^{i\omega})^{-1}
+s_\pi^\top,
```
-where $\mathcal R_t$ embeds $E(R_t\mid T)$ in the companion system.
+where $\mathcal R_t$ embeds $\mathbb{E}(R_t\mid T)$ in the companion system.
Low-frequency power depends on both the autoregressive coefficients and the
innovation covariance.
@@ -1921,8 +1918,8 @@ rule,
```{math}
:label: csdv_policy_rule
i_t = \beta_0
-+ \beta_1 E_t\bar\pi_{t,t+h_\pi}
-+ \beta_2 E_t\bar u_{t,t+h_u}
++ \beta_1 \mathbb{E}_t\bar\pi_{t,t+h_\pi}
++ \beta_2 \mathbb{E}_t\bar u_{t,t+h_u}
+ \beta_3 i_{t-1}
+ \nu_t.
```
@@ -2285,7 +2282,7 @@ if len(incomplete_quarters):
else:
complete_extension = quarterly_unfilled
-cs_sample = pd.read_csv(data_path)
+cs_sample = pd.read_csv(data_url)
overlap_date = pd.Timestamp('2000-10-01')
cs_sample_overlap = cs_sample.iloc[-1]
latest_overlap = latest_quarterly.loc[overlap_date]
@@ -3251,6 +3248,7 @@ require a separate $Q=0$ model.
The 2025Q3 natural-rate and policy-margin estimates remain imprecise, especially
because not every posterior draw satisfies $|\beta_3|<1$.
+(csdv-verdict)=
## Bad policy or bad luck? A verdict
The Bayesian VAR delivers a nuanced answer to the question that opened this
diff --git a/lectures/phillips_escaping_nash.md b/lectures/phillips_escaping_nash.md
index 79917d436..19efa82a1 100644
--- a/lectures/phillips_escaping_nash.md
+++ b/lectures/phillips_escaping_nash.md
@@ -22,6 +22,9 @@ kernelspec:
# Escaping Nash Inflation
+```{index} single: Phillips Curve; Escaping Nash Inflation
+```
+
```{contents} Contents
:depth: 2
```
@@ -73,7 +76,7 @@ U_n = u - \theta(\pi_n - \hat x_n) + \sigma_1 W_{1n},
\hat x_n = x_n,
```
-with $\theta, u > 0$ and $W_n = (W_{1n}, W_{2n})'$ i.i.d. standard Gaussian.
+with $\theta, u > 0$ and $W_n = (W_{1n}, W_{2n})^\top$ IID standard Gaussian.
The government does not know {eq}`en_truth`.
@@ -87,6 +90,15 @@ U_n = \gamma_1 \pi_n + \gamma_{-1} + \eta_n ,
with beliefs $\gamma = (\gamma_1, \gamma_{-1})$ (slope and intercept), and it treats $\eta_n$ as exogenous.
+```{note}
+The slope comes first here, and the regressors are $\Phi = (\pi, 1)$, matching
+{cite}`ChoWilliamsSargent2002` and the $\gamma_1, \gamma_{-1}$ convention of
+{doc}`phillips_self_confirming`.
+
+{doc}`phillips_priors` studies the same model with the intercept first; see the warning there
+for the translation.
+```
+
Believing {eq}`en_belief`, the government solves the {doc}`Phelps problem `, whose static best response sets inflation to the constant
```{math}
@@ -131,7 +143,7 @@ A self-confirming equilibrium is a belief that reproduces itself: the population
Writing $T(\gamma)$ for those population coefficients, CWS show
$$
-\bar g(\gamma) \equiv E\left[\Phi(U - \Phi'\gamma)\right] = \bar M \left(T(\gamma) - \gamma\right),
+\bar g(\gamma) \equiv \mathbb{E}\left[\Phi(U - \Phi^\top \gamma)\right] = \bar M \left(T(\gamma) - \gamma\right),
$$
so a self-confirming equilibrium solves $\bar g(\gamma) = 0$.
@@ -150,9 +162,15 @@ print(f"check g_bar = {model.g_bar(γ_sce)}")
```
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: The unique self-confirming equilibrium in belief space
+ name: fig-esc-sce
+---
γ1_grid = np.linspace(-2, 1, 200)
fig, ax = plt.subplots(figsize=(7, 6))
-ax.plot(γ1_grid, model.u * (1 + γ1_grid**2), label=r'$\gamma_{-1} = u(1+\gamma_1^2)$')
+ax.plot(γ1_grid, model.u * (1 + γ1_grid**2), label=r'$\gamma_{-1} = u(1+\gamma_1^2)$', lw=2)
ax.axvline(-model.θ, color='C1', ls='--', label=r'$\gamma_1 = -\theta$')
ax.plot(γ_sce[0], γ_sce[1], 'ko', ms=8)
ax.annotate('SCE (Nash)', γ_sce, (γ_sce[0] + 0.1, γ_sce[1] + 1))
@@ -183,6 +201,12 @@ A rest point of {eq}`en_mean` is a self-confirming equilibrium, and CWS show thi
So under the mean dynamics alone, the adaptive government is drawn to Nash inflation.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Mean dynamics: from a perturbed start, beliefs return to Nash"
+ name: fig-esc-mean-dynamics
+---
def mean_ode(t, z, model):
γ, R = z[:2], z[2:].reshape(2, 2)
return np.concatenate([np.linalg.inv(R) @ model.g_bar(γ),
@@ -221,7 +245,7 @@ Drawing on {cite}`Williams2019`, CWS reduce this to a clean control problem: the
```{math}
:label: en_control
-\bar S = \inf_{v(\cdot),\, T} \; \frac12 \int_0^T v(s)' Q(\gamma(s), R(s))^{-1} v(s)\, ds
+\bar S = \inf_{v(\cdot),\, T} \; \frac12 \int_0^T v(s)^\top Q(\gamma(s), R(s))^{-1} v(s)\, ds
```
subject to the *perturbed* mean dynamics
@@ -291,6 +315,12 @@ infl = -intercept * slope / (1 + slope**2)
```
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: The dominant escape path and the inflation rate along it
+ name: fig-esc-dominant-path
+---
fig, axes = plt.subplots(1, 2, figsize=(12, 5))
axes[0].plot(esc.t, intercept, label='intercept $\\gamma_{-1}$')
@@ -338,15 +368,18 @@ R0 = model.M(γ_sce)
R0_inv = np.linalg.inv(R0)
σ1σ2 = model.σ1 * model.σ2
+x_sce = model.x(γ_sce)
+
+# each pair of shock realizations induces its own escape forcing
candidates = {
- "{(1,1),(-1,-1)} → Ramsey": np.array([σ1σ2, 0.0]),
+ "{(1,1),(-1,-1)} → Ramsey": np.array([σ1σ2, 0.0]),
"{(1,-1),(-1,1)} → higher π": np.array([-σ1σ2, 0.0]),
- "{(1,1),(1,-1)}": R0 @ (R0_inv @ np.array([model.x(γ_sce) * model.σ1, model.σ1])),
- "{(-1,1),(-1,-1)}": R0 @ (R0_inv @ np.array([-model.x(γ_sce) * model.σ1, -model.σ1])),
+ "{(1,1),(1,-1)}": np.array([x_sce * model.σ1, model.σ1]),
+ "{(-1,1),(-1,-1)}": np.array([-x_sce * model.σ1, -model.σ1]),
}
for name, force in candidates.items():
- v = R0_inv @ force
+ v = R0_inv @ force # belief velocity along this candidate
print(f" {name:28s} |velocity| = {np.linalg.norm(v):.3f}")
```
@@ -354,7 +387,7 @@ The pair $\{(1,1), (-1,-1)\}$ produces a velocity far larger than the last two,
Its mirror image $\{(1,-1),(-1,1)\}$ has the same speed but points the wrong way — toward *higher* inflation — where the mean dynamics oppose it and quickly pull it back.
-So the winner of the race is the Ramsey-ward path, and the escape forcing it induces is the $R^{-1}(\sigma_1\sigma_2, 0)'$ of {eq}`en_force`.
+So the winner of the race is the Ramsey-ward path, and the escape forcing it induces is the $R^{-1}(\sigma_1\sigma_2, 0)^\top$ of {eq}`en_force`.
## Mean dynamics reinforce the escape
@@ -367,6 +400,12 @@ But CWS (their Figures 8-9) show that once beliefs have moved a little way out a
We can see this by plotting the mean-dynamics vector field in belief space.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Mean-dynamics vector field with the escape path superimposed
+ name: fig-esc-vector-field
+---
gs = np.linspace(-1.2, 0.1, 16) # slope
gi = np.linspace(4.5, 10.5, 16) # intercept
GS, GI = np.meshgrid(gs, gi)
@@ -421,7 +460,13 @@ The full dynamic model of {doc}`phillips_learning` — with lagged unemployment
A richer model lets the government detect the subtler distributed-lag ("induction-hypothesis") version of the natural-rate hypothesis, so it escapes toward Ramsey more readily.
```{note}
-The escape dynamics inherit the same "near determinism" that makes the mean dynamics useful: for small gains, the stochastic simulations of {doc}`phillips_learning` hug the deterministic escape path derived here. The next lecture, {doc}`phillips_priors`, shows that the government's *prior* about how its coefficients drift reshapes both dynamics — and can even make the escape a deterministic *cycle*.
+The escape dynamics inherit the same "near determinism" that makes the mean dynamics useful:
+for small gains, the stochastic simulations of {doc}`phillips_learning` hug the deterministic
+escape path derived here.
+
+The next lecture, {doc}`phillips_priors`, shows that the government's *prior* about how its
+coefficients drift reshapes both dynamics — and can even make the escape a deterministic
+*cycle*.
```
## Escaping volatile inflation
@@ -441,7 +486,7 @@ In their model the expected inflation volatility a private agent faces is
```{math}
:label: en_vol
-E(\sigma_\pi \mid \gamma) = \left[ \sigma_2^2 + \left(\frac{\gamma_1}{1 + \gamma_1^2}\right)^2 \sigma_3^2 \right]^{1/2} .
+\mathbb{E}(\sigma_\pi \mid \gamma) = \left[ \sigma_2^2 + \left(\frac{\gamma_1}{1 + \gamma_1^2}\right)^2 \sigma_3^2 \right]^{1/2} .
```
At the self-confirming equilibrium $\gamma_1 = -\theta$, the government believes policy is effective and leans against $W_3$ aggressively, so inflation is *volatile*.
@@ -451,6 +496,12 @@ Along an escape, $\gamma_1 \to 0$: the government stops believing it can exploit
Applying {eq}`en_vol` to the belief path we already computed shows the level and volatility of inflation escaping *in tandem*.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: The level and volatility of inflation escape in tandem
+ name: fig-esc-volatility
+---
σ3 = 0.9 # size of the stabilizable shock
infl_vol = np.sqrt(model.σ2**2 + (slope / (1 + slope**2))**2 * σ3**2)
@@ -480,6 +531,14 @@ The more shocks the government can offset, the more complex the escape-triggerin
Taken literally, this says an economy is more likely to escape to low inflation precisely when there are *few* shocks to stabilize — a suggestive link between the arrival of the mid-1980s calm and the disinflation that accompanied it.
+It is also a claim about data, and it points to the empirical lecture that closes this suite.
+
+If the calm and the disinflation arrived together, then a statistical model must be able to
+separate *shrinking shocks* from *shifting dynamics* before it can say which caused which.
+
+{doc}`phillips_drifts_volatilities` builds a model with room for both channels and lets the
+data apportion them.
+
## Exercises
```{exercise-start}
@@ -515,7 +574,7 @@ for σ in [0.2, 0.3, 0.4, 0.5]:
print(f"σ = {σ}: exit time along the escape path = {s.t[-1]:.2f}")
```
-A larger $\sigma$ makes the escape forcing $R^{-1}(\sigma_1\sigma_2, 0)'$ stronger, so beliefs travel the escape route faster (a shorter exit *time* along the deterministic path).
+A larger $\sigma$ makes the escape forcing $R^{-1}(\sigma_1\sigma_2, 0)^\top$ stronger, so beliefs travel the escape route faster (a shorter exit *time* along the deterministic path).
Note that this is distinct from the *frequency* of escapes, which is governed by the action $\bar S$ and the gain $\varepsilon$; a noisier economy travels a given escape route more quickly once the escape is under way.
diff --git a/lectures/phillips_learning.md b/lectures/phillips_learning.md
index 460e3c83e..083be17b3 100644
--- a/lectures/phillips_learning.md
+++ b/lectures/phillips_learning.md
@@ -22,15 +22,21 @@ kernelspec:
# Adaptive Learning and Escape Dynamics
+```{index} single: Phillips Curve; Adaptive Learning and Escape Dynamics
+```
+
```{contents} Contents
:depth: 2
```
## Overview
-This lecture is the culmination of the *Phillips curve tradeoffs* suite.
+This lecture is the culmination of the theory built by {cite}`Sargent1999`.
-It follows chapter 8 of {cite}`Sargent1999`, the most ambitious chapter of the book.
+It follows chapter 8, the most ambitious chapter of the book, and everything after it in this
+suite either sharpens its analytics ({doc}`phillips_escaping_nash`, {doc}`phillips_priors`),
+carries its tools to a later episode ({doc}`phillips_lost_conquest`), or puts its central claim
+to an empirical test ({doc}`phillips_drifts_volatilities`).
In {doc}`phillips_self_confirming` a government held *fixed* beliefs about the Phillips curve — beliefs that were confirmed by the data those beliefs generated.
@@ -46,7 +52,7 @@ We ask whether such an adaptive government converges to a self-confirming equili
The answer depends on a single parameter — the **gain** that governs how fast old data are discounted:
* With a *decreasing* gain that implements least squares, the mean dynamics pull the economy to a self-confirming equilibrium, and we get nothing new: the system is stuck near the Nash outcome.
-* With a *constant* gain, agents discount past data, convergence is arrested, and **new outcomes emerge**. The system recurrently *escapes* the self-confirming equilibrium toward the Ramsey (zero-inflation) outcome — spontaneous stabilizations that resemble the arrival of Volcker.
+* With a *constant* gain, agents discount past data, convergence is arrested, and **new outcomes emerge** as the system recurrently *escapes* the self-confirming equilibrium toward the Ramsey (zero-inflation) outcome, in spontaneous stabilizations that resemble the arrival of Volcker.
These escapes are the heart of the *vindication of econometric policy evaluation* story from {doc}`phillips_two_stories`: an adaptive government, learning a Solow-Tobin distributed-lag version of the natural-rate hypothesis, is led by chance observations to stabilize inflation.
@@ -74,7 +80,7 @@ The material is technical, and readers who want the punchline can skip to the si
A self-confirming equilibrium under the classical identification is pinned down by the government's beliefs about some population moments and the regression coefficients they imply.
-Under the classical identification these beliefs are measured by the triple $(\gamma, \, E X_{C} X_{C}', \, E U X_{C})$, where $\gamma$ is the vector of Phillips-curve coefficients.
+Under the classical identification these beliefs are measured by the triple $(\gamma, \, E X_{C} X_{C}^\top, \, E U X_{C})$, where $\gamma$ is the vector of Phillips-curve coefficients.
In the adaptive models of this lecture the *time-$t$ values* of these objects are among the economy's state variables; they disappear as state variables in a self-confirming equilibrium only because there they are constants.
@@ -84,8 +90,8 @@ A self-confirming equilibrium under the classical identification satisfies the m
:label: pl_scemoments
\begin{aligned}
-E\, R_{XC}^{-1}(\gamma)\left[ U_t X_{Ct}' - \left(X_{Ct} X_{Ct}'\right)\gamma \right] &= 0, \\
-E\, X_{Ct} X_{Ct}' - R_{XC}(\gamma) &= 0,
+\mathbb{E}\, R_{XC}^{-1}(\gamma)\left[ U_t X_{Ct}^\top - \left(X_{Ct} X_{Ct}^\top \right)\gamma \right] &= 0, \\
+\mathbb{E}\, X_{Ct} X_{Ct}^\top - R_{XC}(\gamma) &= 0,
\end{aligned}
```
@@ -110,9 +116,9 @@ The moment conditions {eq}`pl_scemoments` then take the compact form
```{math}
:label: pl_bdef
-E\left[F(\phi, \zeta)\right] = 0,
+\mathbb{E}\left[F(\phi, \zeta)\right] = 0,
\qquad
-b(\phi) \equiv E\left[F(\phi, \zeta)\right],
+b(\phi) \equiv \mathbb{E}\left[F(\phi, \zeta)\right],
```
where $\zeta$ is a random vector and the expectation is over its distribution (which, again, depends on $\phi$).
@@ -141,7 +147,7 @@ where the distribution used to evaluate the expectation defining $b(\phi_k)$ in
This is the relaxation algorithm used to compute self-confirming equilibria in {doc}`phillips_self_confirming`.
-Each step requires evaluating the mathematical expectation $b(\phi) = E[F(\phi, \zeta)]$ — which is exactly why we needed the moment (Lyapunov) formulas there.
+Each step requires evaluating the mathematical expectation $b(\phi) = \mathbb{E}[F(\phi, \zeta)]$ — which is exactly why we needed the moment (Lyapunov) formulas there.
### Stochastic approximation
@@ -170,7 +176,17 @@ One then approximates $\phi^o(t)$ by a continuous-time process as $n \to \infty$
Different rates of decrease of the gain sequence $\{a_n\}$ produce different approximating processes, because they change the mapping {eq}`pl_artificial` from real time $n$ to artificial time $t_n$.
```{note}
-Recursive stochastic approximation originates with {cite}`RobbinsMonro1951`, who devised {eq}`pl_sa` to find the root of a regression function observed with noise, and with {cite}`KieferWolfowitz1952`, who adapted it to find the maximum of a regression function (the "K-W" algorithms referred to below). The "ODE method" for analyzing such recursions — approximating the interpolated process by the solution of a differential equation — is due to {cite}`Ljung1977`; book-length treatments are {cite}`BenvenisteMetivierPriouret1990` and {cite}`KushnerYin2003`. Its use to study learning in self-referential macroeconomic models is developed by {cite}`MarcetSargent1989` and, comprehensively, by {cite}`EvansHonkapohja2001`.
+Recursive stochastic approximation originates with {cite}`RobbinsMonro1951`, who devised
+{eq}`pl_sa` to find the root of a regression function observed with noise, and with
+{cite}`KieferWolfowitz1952`, who adapted it to find the maximum of a regression function (the
+"K-W" algorithms referred to below).
+
+The "ODE method" for analyzing such recursions — approximating the interpolated process by the
+solution of a differential equation — is due to {cite}`Ljung1977`; book-length treatments are
+{cite}`BenvenisteMetivierPriouret1990` and {cite}`KushnerYin2003`.
+
+Its use to study learning in self-referential macroeconomic models is developed by
+{cite}`MarcetSargent1989` and, comprehensively, by {cite}`EvansHonkapohja2001`.
```
### Mean dynamics
@@ -239,17 +255,21 @@ First, the log moment generating function of (an averaged version of) the innova
```{math}
:label: pl_mgf
-H(\theta, \phi) = \log E \exp\left(\theta' F(\phi, \zeta)\right),
+H(\theta, \phi) = \log \mathbb{E} \exp\left(\theta^\top F(\phi, \zeta)\right),
```
where the expectation is over the distribution of $\zeta$.
```{note}
-Equation {eq}`pl_mgf` is a heuristic shorthand. The object that actually enters the theory is a *time-averaged* limit; {cite}`DupuisKushner1987` and {cite}`KushnerYin2003` assume that for each $\delta > 0$ the following limit exists uniformly in $\phi_i, \alpha_i$ on any compact set:
+Equation {eq}`pl_mgf` is a heuristic shorthand.
+
+The object that actually enters the theory is a *time-averaged* limit;
+{cite}`DupuisKushner1987` and {cite}`KushnerYin2003` assume that for each $\delta > 0$ the
+following limit exists uniformly in $\phi_i, \alpha_i$ on any compact set:
$$
\sum_{i=0}^{T/\delta - 1} \delta\, H(\alpha_i, \phi_i)
= \lim_{N \to \infty} \frac{\delta}{N}
- \log E \exp \sum_{i=0}^{T/\delta - 1} \alpha_i'
+ \log \mathbb{E} \exp \sum_{i=0}^{T/\delta - 1} \alpha_i^\top
\sum_{j=iN}^{iN+N-1} F(\phi_i, \zeta_j) .
$$
The inner sum averages the innovations over a block of length $N$; the double limit lets us treat serially dependent innovations.
@@ -260,7 +280,7 @@ Second, the **Legendre transform** of $H$, which plays the role of a rate functi
```{math}
:label: pl_legendre
-L(\beta, \phi) = \sup_\theta \left[ \theta'\beta - H(\theta, \phi) \right] .
+L(\beta, \phi) = \sup_\theta \left[ \theta^\top \beta - H(\theta, \phi) \right] .
```
Third, the **action functional**, which measures the "cost" of a candidate escape path $\phi(\cdot)$:
@@ -322,7 +342,7 @@ $$
F(\phi, \zeta) = b(\phi) + \sigma(\phi)\, \zeta,
$$
-where $\zeta_n$ is stationary and Gaussian but not necessarily serially uncorrelated, and define $R = \sum_j E\, \zeta_t \zeta_{t-j}'$.
+where $\zeta_n$ is stationary and Gaussian but not necessarily serially uncorrelated, and define $R = \sum_j \mathbb{E}\, \zeta_t \zeta_{t-j}^\top$.
Then the action functional takes the quadratic form
@@ -330,22 +350,31 @@ Then the action functional takes the quadratic form
:label: pl_action2
S(T, \phi) = \frac{1}{2} \int_0^T
-\left(\tfrac{d}{ds}\phi - b(\phi)\right)'
-\left[\sigma(\phi)\, R\, \sigma(\phi)'\right]^{+}
+\left(\tfrac{d}{ds}\phi - b(\phi)\right)^\top
+\left[\sigma(\phi)\, R\, \sigma(\phi)^\top \right]^{+}
\left(\tfrac{d}{ds}\phi - b(\phi)\right)
h(s)\, ds ,
```
-where $(\cdot)^{+}$ is the Moore-Penrose generalized inverse (used to handle possible stochastic singularity of $\sigma R \sigma'$).
+where $(\cdot)^{+}$ is the Moore-Penrose generalized inverse (used to handle possible stochastic singularity of $\sigma R \sigma^\top$).
The weight $h(s)$ depends on the gain: $h(s) = \exp(s)$ when $\gamma = 1$ in $a_n = a_0 / n^\gamma$, and $h(s) = 1$ when $\gamma < 1$.
-Read {eq}`pl_action2` as a cost that penalizes departures of the *realized drift* $\tfrac{d}{ds}\phi$ from the *mean drift* $b(\phi)$, weighting each direction by the inverse of the local noise covariance $\sigma R \sigma'$.
+Read {eq}`pl_action2` as a cost that penalizes departures of the *realized drift* $\tfrac{d}{ds}\phi$ from the *mean drift* $b(\phi)$, weighting each direction by the inverse of the local noise covariance $\sigma R \sigma^\top$.
The least-action escape therefore threads the beliefs through regions where the mean dynamics are weak and the noise is informative — which, in our model, is the direction of the *induction hypothesis*.
```{note}
-This quadratic action functional is precisely the object minimized in {cite}`ChoWilliamsSargent2002`, the published treatment of the model of this lecture. They solve the control problem {eq}`pl_escapeproblem` for the Nash self-confirming equilibrium and show, analytically, that the least-action escape drives the sum of weights on inflation toward the value that activates the induction hypothesis — that is, toward the Ramsey outcome. {cite}`SargentWilliams2005` study how the government's prior (equivalently, the covariance structure of the gain algorithm, our $P_0$ and forgetting factor) reshapes the escape, and {cite}`Kasa2004` applies the same large-deviation machinery to recurrent currency crises.
+This quadratic action functional is precisely the object minimized in
+{cite}`ChoWilliamsSargent2002`, the published treatment of the model of this lecture.
+
+They solve the control problem {eq}`pl_escapeproblem` for the Nash self-confirming equilibrium
+and show, analytically, that the least-action escape drives the sum of weights on inflation
+toward the value that activates the induction hypothesis — that is, toward the Ramsey outcome.
+
+{cite}`SargentWilliams2005` study how the government's prior (equivalently, the covariance
+structure of the gain algorithm, our $P_0$ and forgetting factor) reshapes the escape, and
+{cite}`Kasa2004` applies the same large-deviation machinery to recurrent currency crises.
```
### From computation to adaptation
@@ -360,7 +389,25 @@ Two facts organize everything below:
2. gain sequences that fall off more slowly — in the limit, constant gains that discount the past — *arrest* that pull and increase the frequency with which the escape dynamics influence outcomes.
```{note}
-A brief intellectual history. {cite}`Lucas_Prescott_1971` dismissed iterating on the moment conditions {eq}`pl_scezero` as a computational strategy, but {cite}`Townsend1983` used it. {cite}`Woodford1990` and {cite}`MarcetSargent1989` used the mean dynamics {eq}`pl_ode` to establish conditions for the convergence of least squares learning to rational expectations in models with self-reference, both requiring continuity of $b(\phi)$. In-Koo Cho studied problems with *discontinuous* $b(\phi)$ inherited from discontinuous decision rules (trigger strategies in credibility and search problems); to make least squares learning approach rational expectations he used gains satisfying $\tfrac{1}{\log n} < a_n < \tfrac{1}{\sqrt n}$, which yield a *diffusion* approximation to {eq}`pl_sa` that promotes enough experimentation to discover an equilibrium. {cite}`KandoriMailathRob1993` use related mathematics to select long-run equilibria in games via mutation, and Roger Myerson applied an escape-route calculation to a voting problem. The modern synthesis of these learning methods is {cite}`EvansHonkapohja2001`.
+A brief intellectual history.
+
+{cite}`Lucas_Prescott_1971` dismissed iterating on the moment conditions {eq}`pl_scezero` as a
+computational strategy, but {cite}`Townsend1983` used it.
+
+{cite}`Woodford1990` and {cite}`MarcetSargent1989` used the mean dynamics {eq}`pl_ode` to
+establish conditions for the convergence of least squares learning to rational expectations in
+models with self-reference, both requiring continuity of $b(\phi)$.
+
+In-Koo Cho studied problems with *discontinuous* $b(\phi)$ inherited from discontinuous
+decision rules (trigger strategies in credibility and search problems); to make least squares
+learning approach rational expectations he used gains satisfying $\tfrac{1}{\log n} < a_n < \tfrac{1}{\sqrt n}$,
+which yield a *diffusion* approximation to {eq}`pl_sa` that promotes enough experimentation to
+discover an equilibrium.
+
+{cite}`KandoriMailathRob1993` use related mathematics to select long-run equilibria in games
+via mutation, and Roger Myerson applied an escape-route calculation to a voting problem.
+
+The modern synthesis of these learning methods is {cite}`EvansHonkapohja2001`.
```
## The adaptive model
@@ -374,9 +421,9 @@ The government believes in a distributed-lag Phillips curve
```{math}
:label: pl_belief
-U_t = \gamma' X_{C,t} + \varepsilon_{C,t},
+U_t = \gamma^\top X_{C,t} + \varepsilon_{C,t},
\qquad
-X_{C,t} = \begin{bmatrix} y_t & U_{t-1} & U_{t-2} & y_{t-1} & y_{t-2} & 1 \end{bmatrix}' .
+X_{C,t} = \begin{bmatrix} y_t & U_{t-1} & U_{t-2} & y_{t-1} & y_{t-2} & 1 \end{bmatrix}^\top.
```
Arriving at time $t$ with an estimate $\gamma_{t-1}$, it sets the systematic part of inflation by solving the Phelps problem *as if* $\gamma_{t-1}$ will govern the Phillips curve forever:
@@ -386,7 +433,7 @@ Arriving at time $t$ with an estimate $\gamma_{t-1}$, it sets the systematic par
y_t = h(\gamma_{t-1}) X_{t-1} + v_{2t},
\qquad
-X_{t-1} = \begin{bmatrix} U_{t-1} & U_{t-2} & y_{t-1} & y_{t-2} & 1 \end{bmatrix}' .
+X_{t-1} = \begin{bmatrix} U_{t-1} & U_{t-2} & y_{t-1} & y_{t-2} & 1 \end{bmatrix}^\top.
```
It then updates its beliefs by **recursive least squares** (RLS):
@@ -395,8 +442,8 @@ It then updates its beliefs by **recursive least squares** (RLS):
:label: pl_rls
\begin{aligned}
-\gamma_t &= \gamma_{t-1} + g_t R_{XC,t}^{-1} X_{C,t}\left(U_t - \gamma_{t-1}' X_{C,t}\right), \\
-R_{XC,t} &= R_{XC,t-1} + g_t\left(X_{C,t} X_{C,t}' - R_{XC,t-1}\right),
+\gamma_t &= \gamma_{t-1} + g_t R_{XC,t}^{-1} X_{C,t}\left(U_t - \gamma_{t-1}^\top X_{C,t}\right), \\
+R_{XC,t} &= R_{XC,t-1} + g_t\left(X_{C,t} X_{C,t}^\top - R_{XC,t-1}\right),
\end{aligned}
```
@@ -416,14 +463,14 @@ $$
Given a belief $\gamma$, the decision rule $h(\gamma)$ solves an LQ control problem.
-Write the believed Phillips curve as $U_t = \gamma_0 y_t + c' s_t$, where $\gamma_0$ is the coefficient on current inflation and $c$ collects the coefficients on the state $s_t = X_{t-1}$.
+Write the believed Phillips curve as $U_t = \gamma_0 y_t + c^\top s_t$, where $\gamma_0$ is the coefficient on current inflation and $c$ collects the coefficients on the state $s_t = X_{t-1}$.
-The government minimizes $E\sum_t \delta^t (U_t^2 + y_t^2)$, so the per-period loss is $s_t' (cc') s_t + (\gamma_0^2 + 1) y_t^2 + 2\gamma_0\, y_t\, c' s_t$, and the state evolves as $s_{t+1} = A s_t + B y_t$ with
+The government minimizes $\mathbb{E}\sum_t \delta^t (U_t^2 + y_t^2)$, so the per-period loss is $s_t^\top (cc^\top) s_t + (\gamma_0^2 + 1) y_t^2 + 2\gamma_0\, y_t\, c^\top s_t$, and the state evolves as $s_{t+1} = A s_t + B y_t$ with
$$
s_{t+1} = \begin{bmatrix} U_t \\ U_{t-1} \\ y_t \\ y_{t-1} \\ 1 \end{bmatrix},
\qquad
-U_t = c' s_t + \gamma_0 y_t .
+U_t = c^\top s_t + \gamma_0 y_t .
$$
We solve the discounted LQ problem with `scipy`'s discrete algebraic Riccati equation.
@@ -548,6 +595,12 @@ The classical self-confirming equilibrium has serially uncorrelated $(U, y)$ flu
First, least squares (a decreasing gain).
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Classical adaptive model under least squares: inflation hugs the self-confirming value"
+ name: fig-learn-ls
+---
ls = simulate(model, λ=1.0, T_prior=5000, n=1000, seed=1)
fig, ax = plt.subplots(figsize=(9, 4.5))
@@ -556,7 +609,6 @@ ax.axhline(5, color='k', ls='--', lw=1, label='self-confirming (Nash)')
ax.axhline(0, color='C2', ls=':', lw=1, label='Ramsey')
ax.set_xlabel('$t$')
ax.set_ylabel('inflation $y_t$')
-ax.set_title('Figure 8.1: classical adaptive model, least squares')
ax.legend()
plt.show()
```
@@ -570,6 +622,12 @@ We get nothing new — the government is stuck near the Nash outcome.
Now give the government a *constant* gain, $\lambda = 0.975$, so it discounts past data.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Classical adaptive model under a constant gain: recurrent escapes toward Ramsey"
+ name: fig-learn-cgain
+---
cg = simulate(model, λ=0.975, T_prior=300, n=1000, seed=1)
fig, ax = plt.subplots(figsize=(9, 4.5))
@@ -578,14 +636,10 @@ ax.axhline(5, color='k', ls='--', lw=1, label='self-confirming (Nash)')
ax.axhline(0, color='C2', ls=':', lw=1, label='Ramsey')
ax.set_xlabel('$t$')
ax.set_ylabel('inflation $y_t$')
-ax.set_title('Figure 8.2: classical adaptive model, constant gain '
- r'($\lambda = 0.975$)')
ax.legend()
plt.show()
```
-The picture is completely different.
-
Inflation starts near the self-confirming value of 5, then drops almost to zero and stays there for a long time, before slowly heading back toward 5 only to be propelled toward zero again.
The mean dynamics that pull the system toward the self-confirming equilibrium are opposed by a recurrent force that sends inflation close to the Ramsey outcome.
@@ -606,6 +660,12 @@ The answer is the **induction hypothesis** of {doc}`phillips_adaptive`: when the
Let's plot inflation together with that sum of weights.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Stabilizations coincide with the sum of weights on inflation rising toward zero
+ name: fig-learn-escape-route
+---
fig, axes = plt.subplots(2, 1, figsize=(9, 7), sharex=True)
axes[0].plot(cg['y'], lw=0.8)
@@ -632,6 +692,12 @@ When it reaches zero, the induction hypothesis is (temporarily) satisfied, the P
We can see the escape route directly by plotting the joint path of the constant and the sum of weights in the estimated Phillips curve.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: The escape route in belief space, coloured by time
+ name: fig-learn-belief-path
+---
fig, ax = plt.subplots(figsize=(8, 6))
sc = ax.scatter(cg['constant'], cg['sumweights'], c=np.arange(len(cg['y'])),
cmap='viridis', s=6)
@@ -661,6 +727,12 @@ The recurrent stabilizations toward Ramsey depend on the discount factor $\delta
Lowering $\delta$ raises the inflation rate observed during the low-inflation episodes, consistent with the workings of the Phelps problem under the induction hypothesis.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Escapes toward Ramsey deepen as the government becomes more patient
+ name: fig-learn-discount
+---
fig, ax = plt.subplots(figsize=(9, 4.5))
for δ in [0.90, 0.95, 0.98]:
m = AdaptivePhillips(δ=δ)
@@ -670,7 +742,6 @@ ax.axhline(0, color='k', lw=0.5)
ax.set_xlabel('$t$')
ax.set_ylabel('inflation $y_t$')
ax.legend()
-ax.set_title('Escapes toward Ramsey deepen as the government becomes patient')
plt.show()
```
@@ -700,20 +771,37 @@ Adaptation makes the government's beliefs a hidden state that imparts serial cor
In this sense the adaptive models contain the underpinnings for vindicating econometric policy evaluation — the second of the two stories of {doc}`phillips_two_stories`.
+That last observation is a testable one, and it is worth recording what the model predicts before anyone looks.
+
+If the government's beliefs really are a hidden drifting state, then a reduced-form description of post-war inflation and unemployment ought to show *drifting coefficients*, not merely drifting shock variances.
+
+{doc}`phillips_drifts_volatilities` fits exactly such a model to the data and reports the verdict — including one prediction of the escape mechanism that the data decline to confirm.
+
## Exercises
```{exercise-start}
:label: pl_ex1
```
-Build the **Keynesian** adaptive model, in which the government fits the Phillips curve in the reverse direction, regressing inflation on unemployment.
-
-The regressors are $X_{K,t} = \begin{bmatrix} U_t & U_{t-1} & U_{t-2} & y_{t-1} & y_{t-2} & 1 \end{bmatrix}'$, and the government estimates $\beta$ in $y_t = \beta' X_{K,t} + \varepsilon_{K,t}$, then inverts to $\gamma$ before solving the Phelps problem.
+Everything above hinges on the constant gain, so it is worth seeing how much.
-Rather than re-derive everything, explore the *classical* model's sensitivity to the constant gain: simulate with $\lambda \in \{0.99, 0.975, 0.95\}$ and compare how often inflation escapes toward Ramsey.
+Simulate the classical adaptive model with $\lambda \in \{0.99, 0.975, 0.95\}$ and compare how often inflation escapes toward Ramsey.
How does a larger gain (smaller $\lambda$, faster discounting of the past) affect the frequency of escapes?
+```{note}
+A more ambitious version of this exercise is to build the **Keynesian** adaptive model, in
+which the government fits the Phillips curve in the reverse direction, regressing inflation on
+unemployment.
+
+The regressors are $X_{K,t} = \begin{bmatrix} U_t & U_{t-1} & U_{t-2} & y_{t-1} & y_{t-2} & 1 \end{bmatrix}^\top$;
+the government estimates $\beta$ in $y_t = \beta^\top X_{K,t} + \varepsilon_{K,t}$, then inverts to
+$\gamma$ using the inversion formulas $\gamma_1 = \beta_1^{-1}$, $\gamma_{-1} = -\beta_{-1}/\beta_1$
+of {doc}`phillips_self_confirming` before solving the Phelps problem.
+
+{cite}`Sargent1999` reports that this variant escapes less readily than the classical one.
+```
+
```{exercise-end}
```
@@ -731,6 +819,7 @@ for λ in [0.99, 0.975, 0.95]:
ax.axhline(0, color='k', lw=0.5)
ax.set_xlabel('$t$')
ax.set_ylabel('inflation $y_t$')
+ax.set_title('Escape frequency by constant gain')
ax.legend()
plt.show()
```
@@ -772,4 +861,4 @@ Because that equilibrium is only marginally stable — the mean dynamics have an
A tighter prior keeps the gain small throughout, so least squares reliably hugs the self-confirming equilibrium.
```{solution-end}
-```
\ No newline at end of file
+```
diff --git a/lectures/phillips_lost_conquest.md b/lectures/phillips_lost_conquest.md
index 3cf10176a..9fc63bcbb 100644
--- a/lectures/phillips_lost_conquest.md
+++ b/lectures/phillips_lost_conquest.md
@@ -22,6 +22,9 @@ kernelspec:
# The Lost Conquest: Fed Policy in the 2020s
+```{index} single: Phillips Curve; Fed Policy in the 2020s
+```
+
```{contents} Contents
:depth: 2
```
@@ -78,19 +81,32 @@ from scipy.linalg import solve_discrete_are
The first two elements are among the most documented facts in modern macroeconomics.
-**Declining persistence.**
-From the 1970s into the 1980s inflation was highly persistent; a shock raised inflation for years.
-{cite}`CogleySargentConquest2005` and {cite}`StockWatson2007` document a marked decline in persistence after the mid-1980s — inflation began reverting to target much faster.
+**Declining persistence.** From the 1970s into the 1980s inflation was highly persistent; a
+shock raised inflation for years.
+
+{cite}`CogleySargentConquest2005` and {cite}`StockWatson2007` document a marked decline in
+persistence after the mid-1980s — inflation began reverting to target much faster.
+
+**A flatter Phillips curve.** Since the 1990s, estimates of the Phillips curve's slope have
+trended toward zero — the "missing disinflation" after the Great Recession being the leading
+example.
-**A flatter Phillips curve.**
-Since the 1990s, estimates of the Phillips curve's slope have trended toward zero — the "missing disinflation" after the Great Recession being the leading example.
Both facts were on policy makers' minds.
-Former Fed Chair Janet Yellen observed in 2019 that "the slope of the Phillips curve … has diminished very significantly since the 1960s … and … inflation has become much less persistent."
-And, as {cite}`Bernanke2022` writes, "a flat Phillips curve means that inflation is a less reliable indicator of economic overheating [and] the costs, in terms of unemployment, of bringing inflation back down to target could be higher than in the past."
-**Real-time uncertainty.**
-The third element, emphasized by {cite}`Orphanides2001`, is that the output gap is badly mismeasured in *real time*, especially at business-cycle turning points.
-Through 2020–2023 the real-time gap was persistently *below* the later-revised measure, so the Fed perceived more slack — reinforcing the belief that inflation would fade on its own.
+Former Fed Chair Janet Yellen observed in 2019 that "the slope of the Phillips curve … has
+diminished very significantly since the 1960s … and … inflation has become much less
+persistent."
+
+And, as {cite}`Bernanke2022` writes, "a flat Phillips curve means that inflation is a less
+reliable indicator of economic overheating [and] the costs, in terms of unemployment, of
+bringing inflation back down to target could be higher than in the past."
+
+**Real-time uncertainty.** The third element, emphasized by {cite}`Orphanides2001`, is that the
+output gap is badly mismeasured in *real time*, especially at business-cycle turning points.
+
+Through 2020–2023 the real-time gap was persistently *below* the later-revised measure, so the
+Fed perceived more slack — reinforcing the belief that inflation would fade on its own.
+
We use current-vintage data below and return to the real-time distinction in the conclusion.
## The Fed's drifting-coefficients beliefs
@@ -108,12 +124,12 @@ where $\pi_t$ is inflation, $x_t$ the output gap, $\rho_t$ the *perceived persis
The Fed updates $\theta_t = (\alpha_{0,t}, \rho_t, \kappa_t)$ by constant-gain recursive least squares — exactly the algorithm of {doc}`phillips_learning` and {doc}`phillips_priors`, with gain $\gamma$ discounting the past so the estimates can *track* drift:
$$
-\theta_{t+1} = \theta_t + \gamma R_t^{-1} X_t\left(\pi_t - X_t'\theta_t\right),
+\theta_{t+1} = \theta_t + \gamma R_t^{-1} X_t\left(\pi_t - X_t^\top \theta_t\right),
\qquad
-R_{t+1} = R_t + \gamma\left(X_t X_t' - R_t\right),
+R_{t+1} = R_t + \gamma\left(X_t X_t^\top - R_t\right),
$$
-with $X_t = (1, \pi_{t-1}, x_t)'$.
+with $X_t = (1, \pi_{t-1}, x_t)^\top$.
We download quarterly PCE inflation, the CBO output gap, and the federal funds rate from FRED.
@@ -157,6 +173,12 @@ beliefs = estimate_beliefs(data)
```
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: The Fed's perceived inflation persistence and Phillips curve slope
+ name: fig-lc-beliefs
+---
fig, axes = plt.subplots(2, 1, figsize=(10, 6), sharex=True)
axes[0].plot(beliefs['rho'])
axes[0].axhline(1, color='k', lw=0.5, ls=':')
@@ -173,8 +195,6 @@ plt.tight_layout()
plt.show()
```
-The two panels tell the story.
-
Perceived **persistence** $\rho_t$ is near one through the high-inflation 1970s and 1980s, then drifts down after the mid-1980s, reaching a post-2008 trough — and then *jumps back up* toward one when belief updating resumes in 2022, just as the Fed abandoned the "transitory" characterization and began to tighten.
Perceived **slope** $\kappa_t$ trends toward zero over the 2010s: the Phillips curve flattens.
@@ -185,7 +205,7 @@ By 2019 the Fed's model said inflation was *not persistent* and only *weakly lin
Each period the Fed sets its policy rate by solving a linear-quadratic Phelps problem, taking its *current* estimates as if they will hold forever — the anticipated-utility assumption of {cite}`Kreps1998` that we met in {doc}`phillips_learning`.
-Pairing the belief Phillips curve {eq}`lc_pc` with a fixed "IS curve" $x_t = b_0 + b_1 x_{t-1} + g(i_{t-1} - \pi_{t-1}) + \varepsilon^x_t$ gives linear state dynamics for $X_t = (1, \pi_t, x_t, i_{t-1})'$,
+Pairing the belief Phillips curve {eq}`lc_pc` with a fixed "IS curve" $x_t = b_0 + b_1 x_{t-1} + g(i_{t-1} - \pi_{t-1}) + \varepsilon^x_t$ gives linear state dynamics for $X_t = (1, \pi_t, x_t, i_{t-1})^\top$,
$$
X_{t+1} = A_t X_t + B_t\, i_t + C \varepsilon_{t+1},
@@ -215,7 +235,7 @@ b0, b1, g = np.linalg.lstsq(X_is, x[1:], rcond=None)[0]
β, π_star, λ_x, η = 0.95, 2.0, 0.2, 0.5
-def phelps_rate(θ, state):
+def phelps_rate(θ, state, η=η):
"Optimal (subjectively) funds rate given beliefs θ=(α₀,ρ,κ) and state."
α0, ρ, κ = θ
A = np.array([[1, 0, 0, 0],
@@ -255,13 +275,18 @@ optimal = pd.Series(opt, index=data.index[1:])
```
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: The belief-driven Phelps rule against the actual federal funds rate
+ name: fig-lc-phelps-rate
+---
fig, ax = plt.subplots(figsize=(10, 4.5))
window = slice('1991', None)
ax.plot(optimal[window], 'C0', label="Phelps problem's recommended rate")
ax.plot(data['i'][window], 'C3', lw=1, label='actual federal funds rate')
ax.set_xlabel('year')
ax.set_ylabel('percent')
-ax.set_title("The belief-driven Phelps rule vs. actual policy")
ax.legend()
plt.show()
@@ -278,6 +303,12 @@ Crucially, around the 2021 surge the recommended rate barely moves — the belie
To isolate the role of the drifting beliefs, we recompute the Phelps recommendations holding beliefs *fixed* at their January 2000 values — when inflation was still perceived as persistent and the Phillips curve as steeper.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Counterfactual: a Fed that had not updated its beliefs since 2000"
+ name: fig-lc-counterfactual
+---
counterfactual = pd.Series(
[phelps_rate(θ_2000, np.array([1.0, pi[t], x[t], i_[t - 1]]))
for t in range(1, n)],
@@ -290,13 +321,10 @@ ax.plot(counterfactual[w], 'C1--', label='counterfactual (beliefs frozen at 2000
ax.plot(data['i'][w], 'C3', lw=1, alpha=0.7, label='actual funds rate')
ax.set_xlabel('year')
ax.set_ylabel('percent')
-ax.set_title('Counterfactual: a Fed that had not updated its beliefs since 2000')
ax.legend()
plt.show()
```
-The contrast is stark.
-
A Fed with year-2000 beliefs — perceiving persistent inflation and a steeper Phillips curve — would have tightened *immediately and sharply* in 2021, driving the funds rate well above 4% before the actual Fed had moved at all.
The muted, delayed response was not a change in objectives; it was a change in *beliefs*.
@@ -342,6 +370,12 @@ Under sufficient conditions (a small lagged-inflation term and a sufficiently ag
Let's reproduce {prf:ref}`lc_prop1` by solving the cubic for the stable root.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Aggressive policy makes inflation look less persistent
+ name: fig-lc-persistence
+---
def measured_persistence(φ_π, β=0.99, γ_b=0.5, κ=0.1, σ=1.0):
"Stable MSV root λ(φ_π): the persistence an econometrician would measure."
coeffs = [β, -(1 + β + κ * σ), 1 + γ_b + κ * σ * φ_π, -γ_b]
@@ -354,10 +388,9 @@ def measured_persistence(φ_π, β=0.99, γ_b=0.5, κ=0.1, σ=1.0):
λ_path = [measured_persistence(φ) for φ in φ_grid]
fig, ax = plt.subplots(figsize=(8, 4.5))
-ax.plot(φ_grid, λ_path)
+ax.plot(φ_grid, λ_path, lw=2)
ax.set_xlabel(r'Taylor-rule aggressiveness $\phi_\pi$')
ax.set_ylabel(r'measured persistence $\lambda$')
-ax.set_title('Aggressive policy makes inflation look less persistent')
plt.show()
```
@@ -384,9 +417,18 @@ When large post-pandemic shocks finally hit, those self-confirming beliefs told
This is a modern replay of the *Conquest*'s recurrent dynamics, one level up: the mechanism now works through the perceived slope and persistence of the Phillips curve and the policy-rate instrument, and the misspecification is not about expectations but about the *policy endogeneity* of the reduced-form Phillips curve that the Fed treats as structural — ignoring the Lucas Critique in just the way the "vindication" story of {doc}`phillips_two_stories` describes.
```{note}
-As {cite}`SargentWilliams2025` note, the drifting-coefficients model is a purely descriptive "Kepler stage" model, not a structural "Newton stage" one. The paper also acknowledges an alternative reading in which the 2020s accommodation was fiscal in origin — see the fiscal-theory accounts it cites — a very different rationalization of the same policy path.
+As {cite}`SargentWilliams2025` note, the drifting-coefficients model is a purely descriptive
+"Kepler stage" model, not a structural "Newton stage" one.
+
+The paper also acknowledges an alternative reading in which the 2020s accommodation was fiscal
+in origin — see the fiscal-theory accounts it cites — a very different rationalization of the
+same policy path.
```
+This lecture imputed drifting beliefs to the Fed and then asked what policy they would recommend.
+
+The closing lecture of the suite, {doc}`phillips_drifts_volatilities`, comes at the same period from the opposite direction: it asks the data whether the *reduced form* of the economy drifted at all, and how much of what looks like drifting beliefs is really drifting shock variances.
+
## Exercises
```{exercise-start}
@@ -407,10 +449,8 @@ How does a larger smoothing penalty change the character of the recommended poli
```
```{code-cell} ipython3
-def recommend(θ_path_fn, η_val):
- global η
- η_save = η
- η = η_val
+def recommend(η_val):
+ "Phelps recommendations along the whole sample, for a given smoothing weight."
θ, R = np.array([0.5, 0.9, 0.05]), np.diag([1.0, 10.0, 5.0])
out = []
for t in range(1, n):
@@ -418,17 +458,18 @@ def recommend(θ_path_fn, η_val):
X = np.array([1.0, pi[t - 1], x[t]])
R = R + g_t * (np.outer(X, X) - R)
θ = θ + g_t * np.linalg.solve(R, X * (pi[t] - X @ θ))
- out.append(phelps_rate(θ, np.array([1.0, pi[t], x[t], i_[t - 1]])))
- η = η_save
+ out.append(phelps_rate(θ, np.array([1.0, pi[t], x[t], i_[t - 1]]),
+ η=η_val))
return pd.Series(out, index=data.index[1:])
fig, ax = plt.subplots(figsize=(10, 4.5))
w = slice('2015', None)
for η_val in [0.1, 0.5, 2.0]:
- ax.plot(recommend(None, η_val)[w], lw=1, label=rf'$\eta = {η_val}$')
+ ax.plot(recommend(η_val)[w], lw=1, label=rf'$\eta = {η_val}$')
ax.plot(data['i'][w], 'k:', lw=1.5, label='actual')
ax.set_xlabel('year')
ax.set_ylabel('percent')
+ax.set_title('Phelps recommendations by smoothing weight')
ax.legend()
plt.show()
```
diff --git a/lectures/phillips_misspecified.md b/lectures/phillips_misspecified.md
index 58b079e3e..cd2d01d45 100644
--- a/lectures/phillips_misspecified.md
+++ b/lectures/phillips_misspecified.md
@@ -22,6 +22,9 @@ kernelspec:
# Optimal Misspecified Beliefs
+```{index} single: Phillips Curve; Optimal Misspecified Beliefs
+```
+
```{contents} Contents
:depth: 2
```
@@ -32,6 +35,16 @@ This lecture continues the study of Phillips curve tradeoffs.
It follows chapter 6 of {cite}`Sargent1999`.
+In {doc}`phillips_adaptive` the public forecast inflation with a fixed adaptive rule whose
+parameter $\lambda$ we simply chose.
+
+That was unsatisfying in a way the *vindication* story of {doc}`phillips_two_stories` cannot
+afford: a free parameter describing expectations is exactly what rational expectations was meant
+to eliminate.
+
+Here we take the first step toward earning that parameter, by letting agents choose it to fit
+the data their own beliefs generate.
+
We describe three conceptual issues that recur throughout this suite of lectures:
1. how to formulate equilibria in which agents share a common *misspecified* least squares forecasting model,
@@ -67,7 +80,7 @@ Following {cite}`Bray1982`, assume that
p_t = a + b \, p_{t+1}^e + u_t ,
```
-where $u_t$ is i.i.d. with mean zero and variance $\sigma_u^2$, $a > 0$, $b \in (0, 1)$, $p_t$ is the market price, and $p_{t+1}^e$ is the market's expectation of next period's price.
+where $u_t$ is IID with mean zero and variance $\sigma_u^2$, $a > 0$, $b \in (0, 1)$, $p_t$ is the market price, and $p_{t+1}^e$ is the market's expectation of next period's price.
The rational expectations equilibrium has $p_{t+1}^e = \frac{a}{1-b}$ and $p_t = \frac{a}{1-b} + u_t$.
@@ -271,14 +284,20 @@ print(f"actual one-step forecast error std σ̄_ε = {σ_bar:.4f}")
For the equilibrium $C$, we plot the equilibrium spectral densities of the true and approximating models.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Spectral densities of the true price process and of the agents' approximating model
+ name: fig-mis-spectra
+---
F = bray.true_spectrum(C_star)
σ_ε2 = fitted_sigma2(bray, C_star, c_star)
G = bray.approx_spectrum(c_star, σ_ε2)
half = bray.N // 2
fig, ax = plt.subplots(figsize=(8, 5))
-ax.plot(bray.ω[:half], np.log(F[:half]), 'C0', label='true model')
-ax.plot(bray.ω[:half], np.log(G[:half]), 'C1--', label='forecasting model')
+ax.plot(bray.ω[:half], np.log(F[:half]), 'C0', label='true model', lw=2)
+ax.plot(bray.ω[:half], np.log(G[:half]), 'C1--', label='forecasting model', lw=2)
ax.set_xlabel(r'angular frequency $\omega$')
ax.set_ylabel('log spectral density')
ax.legend()
@@ -296,6 +315,12 @@ The true spectral density decreases sharply with frequency — Granger's "typica
We compare the impulse response functions of the two models by feeding a unit shock through each moving-average representation.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Impulse responses of the true and approximating models
+ name: fig-mis-irf
+---
def impulse_response(num_roots, den_roots, T=25):
"IRF of (1 - num L)/(1 - den L): coefficients of the ratio of lag polys."
h = np.empty(T)
@@ -311,8 +336,8 @@ irf_true = scale * impulse_response(1 - C_star, φ) # f(L)
irf_approx = impulse_response(1 - c_star, bray.ρ) # g(L)
fig, ax = plt.subplots(figsize=(8, 5))
-ax.plot(irf_true, 'C0o-', ms=4, label='true model')
-ax.plot(irf_approx, 'C1s--', ms=4, label='approximating model')
+ax.plot(irf_true, 'C0o-', ms=4, label='true model', lw=2)
+ax.plot(irf_approx, 'C1s--', ms=4, label='approximating model', lw=2)
ax.set_xlabel('lag')
ax.set_ylabel('response')
ax.legend()
@@ -361,9 +386,10 @@ b_grid = np.arange(0.1, 0.85, 0.1)
C_of_b = [solve_equilibrium(BrayModel(a=1.0, b=b, σ_u=1.0)) for b in b_grid]
fig, ax = plt.subplots(figsize=(8, 4.5))
-ax.plot(b_grid, C_of_b, 'o-')
+ax.plot(b_grid, C_of_b, 'o-', lw=2)
ax.set_xlabel('feedback parameter $b$')
ax.set_ylabel('equilibrium belief $C$')
+ax.set_title('Equilibrium belief by expectational feedback')
plt.show()
```
@@ -397,6 +423,7 @@ ax.annotate('equilibrium', (C_star, C_star),
(C_star + 0.05, C_star - 0.03))
ax.set_xlabel('$C$')
ax.set_ylabel('$B(C)$')
+ax.set_title('The best-estimate map and its fixed point')
ax.legend()
plt.show()
```
diff --git a/lectures/phillips_priors.md b/lectures/phillips_priors.md
index 13d47cb7f..2a2c8a0ab 100644
--- a/lectures/phillips_priors.md
+++ b/lectures/phillips_priors.md
@@ -22,6 +22,9 @@ kernelspec:
# Priors, Escapes, and Learning Cycles
+```{index} single: Phillips Curve; Priors and Learning Cycles
+```
+
```{contents} Contents
:depth: 2
```
@@ -78,7 +81,7 @@ U_n &= u - (\pi_n - \hat x_n) + \sigma_1 W_{1n}, \qquad u > 0, \\
\end{aligned}
```
-where $U_n$ is unemployment, $\pi_n$ is inflation, $x_n$ is the systematic part of inflation set by the government, $\hat x_n$ is the public's (rational) forecast, and $W_n = (W_{1n}, W_{2n})'$ is i.i.d. standard Gaussian noise.
+where $U_n$ is unemployment, $\pi_n$ is inflation, $x_n$ is the systematic part of inflation set by the government, $\hat x_n$ is the public's (rational) forecast, and $W_n = (W_{1n}, W_{2n})^\top$ is IID standard Gaussian noise.
Since $\pi_n - \hat x_n = \sigma_2 W_{2n}$, the true unemployment rate is $U_n = u - \sigma_2 W_{2n} + \sigma_1 W_{1n}$ — it fluctuates around the natural rate $u$ regardless of systematic policy.
@@ -94,7 +97,22 @@ U_n = a + b\, \pi_n + \eta_n ,
with belief vector $\gamma = (a, b)$ (intercept and slope), and it treats $\eta_n$ as an exogenous shock.
-Believing {eq}`pp_belief`, the government solves the Phelps problem — minimize $\hat E \sum_n \delta^n (U_n^2 + \pi_n^2)$ — whose static best response sets inflation to the constant
+```{warning}
+{doc}`phillips_escaping_nash` studies this same static model but orders the belief vector the
+other way round, as $\gamma = (\gamma_1, \gamma_{-1})$ with the *slope* first and regressors
+$\Phi = (\pi, 1)$, following {cite}`ChoWilliamsSargent2002`.
+
+Here we put the *intercept* first, with $\Phi = (1, \pi)$, following
+{cite}`SargentWilliams2005`.
+
+The two are the same model in different coordinates: $a = \gamma_{-1}$ and $b = \gamma_1$, so
+the self-confirming belief $(2u, -1)$ of this lecture is the $(-1, u(1+\theta^2))$ of the last
+one, with $\theta = 1$.
+
+Matrices such as $M$, $V$, and $P$ have their rows and columns correspondingly transposed.
+```
+
+Believing {eq}`pp_belief`, the government solves the Phelps problem — minimize $\hat{\mathbb{E}} \sum_n \delta^n (U_n^2 + \pi_n^2)$ — whose static best response sets inflation to the constant
```{math}
:label: pp_bestresp
@@ -130,7 +148,7 @@ class StaticPhillips:
Three belief vectors are worth naming.
-* **Belief 1 (Nash):** $b = -1$ with an intercept that makes the government set $x = u$. This is the time-consistent outcome of {cite}`KydlandPrescott1977`.
+* **Belief 1 (Nash):** $b = -1$ with an intercept that makes the government set $x = u$, the time-consistent outcome of {cite}`KydlandPrescott1977`.
* **Belief 2 (Ramsey):** $b = 0$, so the government perceives *no* tradeoff and sets $x = 0$.
* **Belief 3 (induction):** in a dynamic version, coefficients on current and lagged inflation summing to zero, which for a patient government also sends inflation toward $0$.
@@ -140,7 +158,7 @@ A self-confirming equilibrium is a belief $\bar\gamma$ that reproduces itself: t
For the static model this is easy to solve by hand.
-The slope is $b = \operatorname{cov}(U, \pi)/\operatorname{var}(\pi) = -\sigma_2^2/\sigma_2^2 = -1$, and matching means gives the intercept $a = u + x(\bar\gamma)$.
+The slope is $b = \operatorname{cov}(U, \pi)/\mathbb{V}[\pi] = -\sigma_2^2/\sigma_2^2 = -1$, and matching means gives the intercept $a = u + x(\bar\gamma)$.
Substituting the best response {eq}`pp_bestresp` with $b = -1$ gives $x = a/2$, so $a = u + a/2$, i.e. $a = 2u$.
@@ -175,13 +193,13 @@ and it forms its estimate $\gamma_n = \hat\alpha_{n \mid n-1}$ by the Kalman fil
The covariance matrix $V$ is the government's **prior about parameter drift** — the object we set free.
-With regressors $\Phi_n = (1, \pi_n)'$, a large-sample approximation to the Kalman filter (see {cite}`BenvenisteMetivierPriouret1990`) is
+With regressors $\Phi_n = (1, \pi_n)^\top$, a large-sample approximation to the Kalman filter (see {cite}`BenvenisteMetivierPriouret1990`) is
```{math}
:label: pp_kalman
\begin{aligned}
-\gamma_{n+1} &= \gamma_n + P_n \Phi_n\left(U_n - \Phi_n' \gamma_n\right), \\
+\gamma_{n+1} &= \gamma_n + P_n \Phi_n\left(U_n - \Phi_n^\top \gamma_n\right), \\
P_{n+1} &= P_n - P_n M(\gamma_n) P_n + \sigma^{-2} V ,
\end{aligned}
```
@@ -287,6 +305,12 @@ V(\lambda) = \begin{bmatrix} V^*_{11} & \sqrt\lambda\, V^*_{12} \\ \sqrt\lambda\
For each $\lambda$ we solve the Riccati equation and look at the largest real part among the eigenvalues of $\bar P(\lambda)\, \partial\bar g/\partial\gamma$: where it is positive, the self-confirming equilibrium is unstable.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Stability of the self-confirming equilibrium as the slope prior tightens
+ name: fig-pri-stability
+---
def V_tighten_slope(λ, V_star):
V = V_star.copy()
V[0, 1] = V[1, 0] = np.sqrt(λ) * V_star[0, 1]
@@ -306,7 +330,6 @@ ax.plot(λ_grid, max_re)
ax.axhline(0, color='k', lw=0.8)
ax.set_xlabel(r'prior-tightening parameter $\lambda$')
ax.set_ylabel('max real part of eigenvalue')
-ax.set_title(r'Figure 4: stability of the SCE as the slope prior tightens')
plt.show()
```
@@ -334,16 +357,22 @@ mask = sol.t > sol.t[-1] - 800
```
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Coefficients cycle, and the limit cycle in belief space
+ name: fig-pri-cycle
+---
fig, axes = plt.subplots(1, 2, figsize=(12, 5))
-axes[0].plot(sol.t[mask], a_path[mask], label='intercept')
-axes[0].plot(sol.t[mask], b_path[mask], label='slope')
+axes[0].plot(sol.t[mask], a_path[mask], label='intercept', lw=2)
+axes[0].plot(sol.t[mask], b_path[mask], label='slope', lw=2)
axes[0].set_xlabel('time')
axes[0].set_ylabel('coefficient')
axes[0].set_title('Figure 5a: coefficients cycle')
axes[0].legend()
-axes[1].plot(a_path[mask], b_path[mask])
+axes[1].plot(a_path[mask], b_path[mask], lw=2)
axes[1].plot(*γ_sce, 'kx', ms=10, label='SCE')
axes[1].set_xlabel('intercept')
axes[1].set_ylabel('slope')
@@ -359,13 +388,18 @@ The beliefs settle into a closed orbit around the self-confirming equilibrium.
Because inflation is a function of the coefficients through the best response {eq}`pp_bestresp`, the cycle in beliefs shows up as a cycle in inflation that oscillates between the Nash and Ramsey outcomes.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Inflation oscillates between Nash and Ramsey along the learning cycle
+ name: fig-pri-inflation-cycle
+---
fig, ax = plt.subplots(figsize=(9, 4.5))
ax.plot(sol.t[mask], x_path[mask])
ax.axhline(model.u, color='k', ls='--', lw=1, label='Nash')
ax.axhline(0, color='C2', ls=':', lw=1, label='Ramsey')
ax.set_xlabel('time')
ax.set_ylabel('inflation $x$')
-ax.set_title('Figure 6: inflation oscillates between Nash and Ramsey along the cycle')
ax.legend()
plt.show()
```
@@ -417,7 +451,37 @@ When $\sigma = \sigma_1$ — as in a self-confirming equilibrium, where the regr
Sims instead used $\sigma \neq \sigma_1$ (and did not shrink the gain), which *misallocates* the observed variation and produces prolonged, perhaps permanent, departures from the self-confirming equilibrium.
-Let's simulate the static model under both specifications.
+The mechanism is visible in the Riccati equation {eq}`pp_riccati` before we simulate anything.
+
+A government that attributes *too little* variance to its regression error must attribute the
+variation it sees to something else, and the only thing else available is drift in its own
+coefficients.
+
+So $\sigma < \sigma_1$ inflates the steady-state $P$, and $P$ is what multiplies each forecast
+error in {eq}`pp_kalman`.
+
+Understating the error variance therefore makes the government *learn faster* — which is
+precisely what weakens the pull of the mean dynamics.
+
+```{code-cell} ipython3
+def effective_gain(model, σ_govt, ε, λ=1.0):
+ "Steady-state weight the filter puts on a single forecast error at the SCE."
+ V = ε**2 * V_tighten_slope(λ, V_star)
+ P = solve_riccati(V, M_sce, σ_govt)
+ Φ = np.array([1.0, model.x(γ_sce)])
+ return (P @ Φ)[0] / (σ_govt**2 + Φ @ P @ Φ)
+
+ε_common = 0.0002
+for σ_govt, tag in ((model.σ1, 'σ = σ1 (correct) '), (0.1, 'σ ≠ σ1 (Sims) ')):
+ print(f"{tag}: effective gain = {effective_gain(model, σ_govt, ε_common):.5f}")
+```
+
+Understating the error variance multiplies the effective gain by more than twenty, even though
+the prior $V$ and the scale $\varepsilon$ are identical in the two runs.
+
+Let's now simulate the static model under both specifications, holding $\varepsilon$ fixed so
+that the *only* difference is the variance the government attributes to its own regression
+error.
```{code-cell} ipython3
def simulate(model, σ_govt, ε, λ=1.0, T=3000, seed=0):
@@ -442,35 +506,75 @@ def simulate(model, σ_govt, ε, λ=1.0, T=3000, seed=0):
infl[n] = π
return infl
-x_base = simulate(model, σ_govt=model.σ1, ε=0.05, seed=1) # σ = σ1
-x_sims = simulate(model, σ_govt=0.1, ε=0.20, seed=1) # σ ≠ σ1 (Sims-like)
+T_sim, seeds = 6000, range(10)
+
+def summarize(σ_govt):
+ "Mean inflation and time near Ramsey, across seeds."
+ paths = [simulate(model, σ_govt=σ_govt, ε=ε_common, T=T_sim, seed=s)
+ for s in seeds]
+ return (np.array([p.mean() for p in paths]),
+ np.array([(p < 2).mean() for p in paths]))
-print(f"σ = σ1 : mean inflation {x_base.mean():.2f}, "
- f"fraction near Ramsey {(x_base < 2).mean():.0%}")
-print(f"σ ≠ σ1 : mean inflation {x_sims.mean():.2f}, "
- f"fraction near Ramsey {(x_sims < 2).mean():.0%}")
+mean_base, ramsey_base = summarize(model.σ1)
+mean_sims, ramsey_sims = summarize(0.1)
+
+print(f"{'seed':>4} | {'σ = σ1: mean π':>15} {'near Ramsey':>12}"
+ f" | {'σ ≠ σ1: mean π':>15} {'near Ramsey':>12}")
+for s in seeds:
+ print(f"{s:>4} | {mean_base[s]:>15.2f} {ramsey_base[s]:>11.0%}"
+ f" | {mean_sims[s]:>15.2f} {ramsey_sims[s]:>11.0%}")
+print(f"\nnever escaped in {T_sim} periods: "
+ f"σ = σ1: {np.sum(ramsey_base < 0.01)}/{len(seeds)} paths, "
+ f"σ ≠ σ1: {np.sum(ramsey_sims < 0.01)}/{len(seeds)} paths")
```
+The two columns behave quite differently, and the difference is about *how often* the economy
+leaves Nash rather than about where it goes when it does.
+
+With the correct error variance, an escape is a genuinely rare event: several of the sample
+paths sit at the Nash rate for six thousand periods without ever escaping at all, and the rest
+escape at some point and are held near Ramsey afterwards.
+
+With Sims's misallocation, *every* path escapes, and every one of them spends most of its time
+near Ramsey.
+
+Three seeds make the point without any cherry-picking: seed 0 never escapes, seed 4 escapes
+repeatedly and is pulled back each time, and seed 6 escapes and stays away for a long spell.
+
```{code-cell} ipython3
-fig, axes = plt.subplots(2, 1, figsize=(9, 7), sharex=True)
-axes[0].plot(x_base, lw=0.6)
-axes[0].axhline(model.u, color='k', ls='--', lw=1)
-axes[0].set_ylabel('inflation')
-axes[0].set_title(r'$\sigma = \sigma_1$: recurrent escapes, pulled back to Nash')
-
-axes[1].plot(x_sims, lw=0.6, color='C1')
-axes[1].axhline(model.u, color='k', ls='--', lw=1)
+---
+mystnb:
+ figure:
+ caption: Escapes are rare under the correct error variance and pervasive under Sims's misallocation
+ name: fig-pri-sims
+---
+show = (0, 4, 6)
+
+fig, axes = plt.subplots(2, 1, figsize=(9.5, 7), sharex=True, sharey=True)
+for s in show:
+ axes[0].plot(simulate(model, σ_govt=model.σ1, ε=ε_common, T=T_sim, seed=s),
+ lw=0.5, label=f'seed {s}')
+ axes[1].plot(simulate(model, σ_govt=0.1, ε=ε_common, T=T_sim, seed=s),
+ lw=0.5, label=f'seed {s}')
+for ax, title in zip(axes, [r'$\sigma = \sigma_1$ (correct): escapes are rare events',
+ r'$\sigma \neq \sigma_1$ (Sims): every path escapes and lingers']):
+ ax.axhline(model.u, color='k', ls='--', lw=1)
+ ax.axhline(0, color='C2', ls=':', lw=1)
+ ax.set_ylabel('inflation')
+ ax.set_title(title)
+ ax.legend(frameon=False, fontsize=8, ncol=3, loc='lower right')
axes[1].set_xlabel('$n$')
-axes[1].set_ylabel('inflation')
-axes[1].set_title(r'$\sigma \neq \sigma_1$ (Sims): prolonged spells near Ramsey')
-
plt.tight_layout()
plt.show()
```
-With the correct error variance the mean dynamics reassert themselves and inflation is repeatedly pulled back toward Nash.
+With the correct error variance the mean dynamics dominate: the economy sits at Nash, and only
+an unusual run of shocks dislodges it — the rare-escape picture of {doc}`phillips_learning` and
+{doc}`phillips_escaping_nash`.
-With Sims's misallocation the pull is weakened, and the economy lingers near the Ramsey outcome — the government behaves as if it has *permanently* learned a good-enough version of the natural-rate hypothesis.
+With Sims's misallocation the government learns too fast for the mean dynamics to hold it, so
+it escapes almost immediately and behaves as if it had *permanently* learned a good-enough
+version of the natural-rate hypothesis.
As {cite}`SargentWilliams2005` put it, one can read the difference in two equivalent ways: either Sims allowed too much parameter drift to permit convergence, or he did not let the government attribute enough variation to its regression error.
@@ -506,7 +610,9 @@ The broader message is that *how* an adaptive government learns — the prior it
It determines whether the economy converges to Nash, cycles between Nash and Ramsey, or escapes to Ramsey and stays there.
-The final lecture, {doc}`phillips_lost_conquest`, carries these same tools — constant-gain learning, an anticipated-utility Phelps problem, and a self-confirming equilibrium — into the present, to rationalize the Federal Reserve's response to the inflation of the 2020s.
+{doc}`phillips_lost_conquest` carries these same tools — constant-gain learning, an anticipated-utility Phelps problem, and a self-confirming equilibrium — into the present, to rationalize the Federal Reserve's response to the inflation of the 2020s.
+
+{doc}`phillips_drifts_volatilities` then closes the suite by putting the whole account to an empirical test, fitting a drifting-coefficient, stochastic-volatility VAR to post-war data and asking how much of the Great Inflation was drifting beliefs and how much was simply bad luck.
## Exercises
@@ -552,6 +658,7 @@ ax.plot(λ_grid, max_re, ls='--', label='tighten slope (for comparison)')
ax.axhline(0, color='k', lw=0.8)
ax.set_xlabel(r'$\lambda$')
ax.set_ylabel('max real part of eigenvalue')
+ax.set_title('Stability under a tighter intercept prior')
ax.legend()
plt.show()
```
@@ -595,9 +702,16 @@ for name, V in [("baseline V*", V_star),
print(f"{name:24s}: direction {v.round(3)}, terminal {term.round(2)}")
```
-Both priors send beliefs toward a terminal point with slope $0$ — the Ramsey belief — but along different directions and to slightly different intercepts.
+Both priors send beliefs toward a terminal point with slope $0$ — the Ramsey belief — but along
+different directions, and the intercepts they arrive at differ considerably: the slope-tightened
+prior lands at an intercept a little over half the baseline's.
+
+The destination is robust in the sense that matters for policy: slope $0$ means the government
+perceives no exploitable tradeoff, and {eq}`pp_bestresp` then sets inflation to zero whatever the
+intercept.
-The destination (zero inflation) is a robust feature; the *route* depends on the shape of the prior.
+The *route*, and where along the Ramsey line the escape terminates, depend on the shape of the
+prior.
```{solution-end}
```
diff --git a/lectures/phillips_self_confirming.md b/lectures/phillips_self_confirming.md
index 2b5eb1b21..385174c38 100644
--- a/lectures/phillips_self_confirming.md
+++ b/lectures/phillips_self_confirming.md
@@ -22,6 +22,9 @@ kernelspec:
# Self-Confirming Equilibria
+```{index} single: Phillips Curve; Self-Confirming Equilibria
+```
+
```{contents} Contents
:depth: 2
```
@@ -36,9 +39,10 @@ In addition to what's in Anaconda, this lecture will use the following library:
## Overview
-This lecture completes the study of Phillips curve tradeoffs begun in {doc}`phillips_credibility`.
+This lecture completes the *equilibrium* half of the study begun in {doc}`phillips_credibility`.
-It follows chapter 7 of {cite}`Sargent1999`.
+It follows chapter 7 of {cite}`Sargent1999`, after which the suite turns from fixed beliefs to
+beliefs that are learned in real time.
We seek models that depart minimally from the basic {cite}`KydlandPrescott1977` model of {doc}`phillips_credibility` but that also let a government's *beliefs* be shaped by the data its own policies generate.
@@ -69,13 +73,13 @@ Recall from {doc}`phillips_adaptive` the ingredients of the general Phelps probl
The government believes in a reduced-form Phillips curve that it can fit in either direction:
$$
-\text{Classical:} \quad U_t = \gamma' X_{C,t} + \varepsilon_{C,t},
-\qquad X_{C,t} = \begin{bmatrix} y_t & X_{t-1}' \end{bmatrix}',
+\text{Classical:} \quad U_t = \gamma^\top X_{C,t} + \varepsilon_{C,t},
+\qquad X_{C,t} = \begin{bmatrix} y_t & X_{t-1}^\top \end{bmatrix}^\top,
$$
$$
-\text{Keynesian:} \quad y_t = \beta' X_{K,t} + \varepsilon_{K,t},
-\qquad X_{K,t} = \begin{bmatrix} U_t & X_{t-1}' \end{bmatrix}' .
+\text{Keynesian:} \quad y_t = \beta^\top X_{K,t} + \varepsilon_{K,t},
+\qquad X_{K,t} = \begin{bmatrix} U_t & X_{t-1}^\top \end{bmatrix}^\top.
$$
Solving the Phelps problem takes the government's beliefs $\gamma$ as given and delivers a decision rule $h(\gamma)$ for inflation.
@@ -98,7 +102,7 @@ The *actual* Phillips curve extends the one used in earlier lectures to allow fo
U_t = U^* - \frac{\theta}{1 - \rho_2 L}(y_t - x_t) + \frac{v_{1t}}{1 - \rho_1 L},
```
-with $|\rho_1| < 1$, $|\rho_2| < 1$, and $v_t = (v_{1t}, v_{2t})'$ a vector white noise, where $v_{2t} \equiv y_t - x_t$ is the surprise in inflation.
+with $|\rho_1| < 1$, $|\rho_2| < 1$, and $v_t = (v_{1t}, v_{2t})^\top$ a vector white noise, where $v_{2t} \equiv y_t - x_t$ is the surprise in inflation.
For much of this lecture we set $\rho_1 = \rho_2 = 0$ to make theoretical points, which reduces {eq}`sc_actual` to
@@ -122,7 +126,7 @@ A self-confirming equilibrium is a fixed belief vector $\gamma$, a government de
(c) unemployment is generated by the actual Phillips curve {eq}`sc_actual`; and
(d) the government's beliefs satisfy the least squares orthogonality conditions
-$E\left[U_t - \gamma' X_{C,t}\right] X_{C,t}' = 0$ (the **classical** direction of fit).
+$\mathbb{E}\left[U_t - \gamma^\top X_{C,t}\right] X_{C,t}^\top = 0$ (the **classical** direction of fit).
```
Condition (d) makes the government's beliefs depend on moment matrices that, through (a)–(c), themselves depend on the government's beliefs.
@@ -131,12 +135,19 @@ The government's beliefs imply behavior that produces data whose moments *confir
A distinct self-confirming equilibrium results from replacing (d) with the **Keynesian** direction of fit:
-> (d′) the government fits the Keynesian Phillips curve, $E\left[y_t - \beta' X_{K,t}\right] X_{K,t}' = 0$, then recovers $\gamma$ from the inversion formulas {eq}`sc_invert`.
+> (d′) the government fits the Keynesian Phillips curve, $\mathbb{E}\left[y_t - \beta^\top X_{K,t}\right] X_{K,t}^\top = 0$, then recovers $\gamma$ from the inversion formulas {eq}`sc_invert`.
Because the government's beliefs affect the whole probability distribution of the data, the direction of minimization affects outcomes.
```{note}
-Computing a self-confirming equilibrium in general means finding a fixed point of a map $\gamma = T(h(\gamma))$ (classical) or $\beta = S(h(\gamma(\beta)))$ (Keynesian). The moments in the orthogonality conditions are obtained from the state-space representation of the system by solving a discrete Lyapunov equation. In practice one iterates a relaxation algorithm $\beta_{j+1} = \kappa\beta_j + (1-\kappa) S(\beta_j)$, which resembles the least squares learning recursion of {doc}`phillips_credibility`.
+Computing a self-confirming equilibrium in general means finding a fixed point of a map $\gamma = T(h(\gamma))$
+(classical) or $\beta = S(h(\gamma(\beta)))$ (Keynesian).
+
+The moments in the orthogonality conditions are obtained from the state-space representation of
+the system by solving a discrete Lyapunov equation.
+
+In practice one iterates a relaxation algorithm $\beta_{j+1} = \kappa\beta_j + (1-\kappa) S(\beta_j)$,
+which resembles the least squares learning recursion of {doc}`phillips_credibility`.
```
## The special case solved by hand
@@ -148,9 +159,9 @@ With $X_{t-1} = 1$, the actual Phillips curve implies the second moments
```{math}
:label: sc_moments
-\operatorname{var}(U_t) = \theta^2 \sigma_2^2 + \sigma_1^2,
+\mathbb{V}[U_t] = \theta^2 \sigma_2^2 + \sigma_1^2,
\qquad
-\operatorname{var}(y_t) = \sigma_2^2,
+\mathbb{V}[y_t] = \sigma_2^2,
\qquad
\operatorname{cov}(U_t, y_t) = -\theta \sigma_2^2 .
```
@@ -158,7 +169,7 @@ With $X_{t-1} = 1$, the actual Phillips curve implies the second moments
**Classical direction of fit** ($U$ on $y$): the slope is
$$
-\gamma_1 = \frac{\operatorname{cov}(U_t, y_t)}{\operatorname{var}(y_t)} = -\theta,
+\gamma_1 = \frac{\operatorname{cov}(U_t, y_t)}{\mathbb{V}[y_t]} = -\theta,
$$
and the requirement that the means lie on the regression line gives the intercept $\gamma_{-1} = (\gamma_1^2 + 1) U^*$.
@@ -166,7 +177,7 @@ and the requirement that the means lie on the regression line gives the intercep
**Keynesian direction of fit** ($y$ on $U$): the slope is
$$
-\beta_1 = \frac{\operatorname{cov}(U_t, y_t)}{\operatorname{var}(U_t)}
+\beta_1 = \frac{\operatorname{cov}(U_t, y_t)}{\mathbb{V}[U_t]}
= \frac{-\theta \sigma_2^2}{\sigma_1^2 + \theta^2 \sigma_2^2},
$$
@@ -222,13 +233,19 @@ Under the Keynesian direction of fit, the government estimates a *flatter* Phill
Let's draw the two self-confirming Phillips curves, reproducing Figure 7.1.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: The two self-confirming Phillips curves, one for each direction of fit
+ name: fig-sce-two-curves
+---
fig, ax = plt.subplots(figsize=(7, 6))
U_grid = np.linspace(0, 12, 100)
# perceived Phillips curves U = γ_{-1} + γ_1 y => y = (U - γ_{-1}) / γ_1
-ax.plot(U_grid, (U_grid - γ0_C) / γ1_C, 'C0', label='P: classical fit')
-ax.plot(U_grid, (U_grid - γ0_K) / γ1_K, 'C1', label='Q: Keynesian fit')
+ax.plot(U_grid, (U_grid - γ0_C) / γ1_C, 'C0', label='P: classical fit', lw=2)
+ax.plot(U_grid, (U_grid - γ0_K) / γ1_K, 'C1', label='Q: Keynesian fit', lw=2)
ax.plot(sce.U_star, y_C, 'C0o')
ax.annotate('Nash', (sce.U_star, y_C), (sce.U_star + 0.4, y_C - 0.6))
@@ -278,7 +295,7 @@ x_t = C y_{t-1} + (1 - C) x_{t-1}, \quad C \in (0, 1),
where the public has constant-gain adaptive expectations with a parameter $C$ that it *tunes to fit the data*.
-Taking $x_t$ as a state variable, the government solves the Phelps problem: it maximizes $-E_0 \sum_{t=0}^\infty \delta^t\left[(U^* - \theta(y_t - x_t))^2 + y_t^2\right]$ by choice of a feedback rule $y_t = f_1 + f_2 x_t + v_{2t}$.
+Taking $x_t$ as a state variable, the government solves the Phelps problem: it maximizes $-\mathbb{E}_0 \sum_{t=0}^\infty \delta^t\left[(U^* - \theta(y_t - x_t))^2 + y_t^2\right]$ by choice of a feedback rule $y_t = f_1 + f_2 x_t + v_{2t}$.
This is exactly the LQ Phelps problem of {doc}`phillips_adaptive` with adaptation parameter $\lambda = 1 - C$.
@@ -386,6 +403,12 @@ The induction hypothesis embedded in the adaptive expectations scheme, together
As the government becomes more patient, the equilibrium mean inflation rate falls toward the Ramsey value of zero.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Mean inflation and the equilibrium gain as the government becomes more patient
+ name: fig-sce-patience
+---
δ_grid = np.array([0.95, 0.96, 0.97, 0.98, 0.99, 0.995])
C_vals, ν_vals = [], []
for δ in δ_grid:
@@ -400,7 +423,7 @@ axes[0].set_xlabel(r'discount factor $\delta$')
axes[0].set_ylabel('mean inflation')
axes[0].legend()
-axes[1].plot(δ_grid, C_vals, 'o-', color='C1')
+axes[1].plot(δ_grid, C_vals, 'o-', color='C1', lw=2)
axes[1].set_xlabel(r'discount factor $\delta$')
axes[1].set_ylabel('equilibrium gain $C$')
@@ -411,7 +434,12 @@ plt.show()
Mean inflation lies far below the Nash value at every discount factor and declines toward the Ramsey value of zero as $\delta \to 1$.
```{note}
-The precise equilibrium values depend on the near–unit-root approximation $\rho$ used to keep the perceived model's spectral density well defined, as discussed in {doc}`phillips_misspecified`. The qualitative conclusion — better-than-Nash outcomes that approach Ramsey as $\delta \to 1$ — is robust.
+The precise equilibrium values depend on the near–unit-root approximation $\rho$ used to keep
+the perceived model's spectral density well defined, as discussed in
+{doc}`phillips_misspecified`.
+
+The qualitative conclusion — better-than-Nash outcomes that approach Ramsey as $\delta \to 1$ —
+is robust.
```
### Spectra and impulse responses
@@ -419,6 +447,12 @@ The precise equilibrium values depend on the near–unit-root approximation $\rh
Let's compare the true and approximating inflation processes at the equilibrium, as in Figures 7.2 and 7.3 of {cite}`Sargent1999`.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Spectral densities of the true and approximating inflation processes at the equilibrium
+ name: fig-sce-spectra
+---
ν_star, F, _ = mp.true_process(C_star)
c_star = mp.best_estimate(C_star)
H = np.abs((1 - (1 - c_star) * mp.z) / (1 - mp.ρ * mp.z))**2
@@ -427,8 +461,8 @@ G = H * σ_ε2
half = mp.N // 2
fig, ax = plt.subplots(figsize=(8, 5))
-ax.plot(mp.ω[:half], np.log(F[:half]), 'C0', label='true model')
-ax.plot(mp.ω[:half], np.log(G[:half]), 'C1--', label='approximating model')
+ax.plot(mp.ω[:half], np.log(F[:half]), 'C0', label='true model', lw=2)
+ax.plot(mp.ω[:half], np.log(G[:half]), 'C1--', label='approximating model', lw=2)
ax.set_xlabel(r'angular frequency $\omega$')
ax.set_ylabel('log spectral density')
ax.legend()
@@ -440,6 +474,12 @@ The true and approximating spectral densities match well at all but the lowest f
The true inflation rate is only moderately serially correlated, and — as in the Bray model of {doc}`phillips_misspecified` — the approximating model uses a unit root to simulate a mean, capturing first moments with second moments.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Impulse responses of the true and approximating inflation processes
+ name: fig-sce-irf
+---
def ima_impulse(num, den, T=25):
"IRF of (1 - num L)/(1 - den L)."
h = np.empty(T)
@@ -454,8 +494,8 @@ irf_true = ima_impulse(1 - C_star, ψ)
irf_approx = ima_impulse(1 - c_star, mp.ρ)
fig, ax = plt.subplots(figsize=(8, 5))
-ax.plot(irf_true, 'C0o-', ms=4, label='true model')
-ax.plot(irf_approx, 'C1s--', ms=4, label='approximating model')
+ax.plot(irf_true, 'C0o-', ms=4, label='true model', lw=2)
+ax.plot(irf_approx, 'C1s--', ms=4, label='approximating model', lw=2)
ax.set_xlabel('lag')
ax.set_ylabel('response')
ax.legend()
@@ -507,6 +547,7 @@ ax.plot(σ1_grid, y_keynes, label='Keynesian mean inflation')
ax.axhline(5.0, color='k', ls='--', lw=1, label='Nash')
ax.set_xlabel(r'$\sigma_1$')
ax.set_ylabel('mean inflation')
+ax.set_title('Keynesian mean inflation by shock size')
ax.legend()
plt.show()
```
@@ -546,6 +587,7 @@ ax.plot(C_star, C_star, 'ko')
ax.annotate('equilibrium', (C_star, C_star), (C_star + 0.03, C_star - 0.03))
ax.set_xlabel('$C$')
ax.set_ylabel('$B(C)$')
+ax.set_title('The best-estimate map and the equilibrium gain')
ax.legend()
plt.show()
```
diff --git a/lectures/phillips_two_stories.md b/lectures/phillips_two_stories.md
index 0e380be5a..ce48acdf9 100644
--- a/lectures/phillips_two_stories.md
+++ b/lectures/phillips_two_stories.md
@@ -22,6 +22,9 @@ kernelspec:
# The Rise and Fall of U.S. Inflation
+```{index} single: Phillips Curve; Rise and Fall of U.S. Inflation
+```
+
```{contents} Contents
:depth: 2
```
@@ -51,13 +54,15 @@ In both stories, the Federal Reserve learns the natural-rate-of-unemployment the
The stories differ in how that theory is cast:
* **The triumph of natural-rate theory.** Academic economists discovered the natural-rate hypothesis, taught that any inflation-unemployment tradeoff is temporary, and eventually persuaded policy makers to pursue low inflation.
-* **The vindication of econometric policy evaluation.** Policy makers never abandoned the methods that Robert Lucas criticized in his famous Critique. Recurrently re-estimating a Phillips curve and using it to choose a target, they were led by the *data itself* — an adversely shifting empirical Phillips curve — toward lower inflation.
+* **The vindication of econometric policy evaluation.** Policy makers never abandoned the methods that Robert Lucas criticized in his famous Critique.
+ - Recurrently re-estimating a Phillips curve and using it to choose a target, they were led toward lower inflation by the *data itself*, an adversely shifting empirical Phillips curve.
This lecture presents the facts that motivate both stories, sketches the two interpretations, and reviews the Lucas Critique that chapter 2 both invokes and modifies.
The remaining lectures in the suite build the models:
* {doc}`phillips_credibility` — the one-period Kydland-Prescott credibility problem (chapter 3).
+* {doc}`phillips_credible_policies` — reputation in the repeated economy, and why the theory of credible policy replaces pessimism with agnosticism (chapter 4).
* {doc}`phillips_adaptive` — adaptive expectations and the Phelps problem (chapter 5).
* {doc}`phillips_misspecified` — equilibrium under optimal misspecified beliefs (chapter 6).
* {doc}`phillips_self_confirming` — self-confirming equilibria (chapter 7).
@@ -67,6 +72,47 @@ The remaining lectures in the suite build the models:
* {doc}`phillips_lost_conquest` — the same tools turned on the 2020s inflation and the Fed's slow response ({cite}`SargentWilliams2025`).
* {doc}`phillips_drifts_volatilities` — an empirical postscript that fits a drifting-coefficient, stochastic-volatility VAR to the data and asks whether the Great Inflation was bad policy or bad luck ({cite}`CogleySargent2005`).
+(phillips_notation)=
+### A note on notation
+
+The lectures follow the notation of the source each one is based on, and those sources do not
+agree with one another.
+
+Rather than impose a single scheme and diverge from the papers a reader may want to consult, we
+keep each lecture faithful to its source and record the translations here.
+
+| object | symbol | where |
+|---|---|---|
+| inflation | $y$ | {doc}`phillips_credibility` – {doc}`phillips_self_confirming` |
+| | $\pi$ | {doc}`phillips_escaping_nash` onward |
+| public's expected inflation | $x$ | {doc}`phillips_credibility`, {doc}`phillips_adaptive` |
+| government's systematic inflation | $x$ | {doc}`phillips_escaping_nash`, {doc}`phillips_priors` |
+| natural rate of unemployment | $U^*$ | {doc}`phillips_credibility` – {doc}`phillips_learning` |
+| | $u$ | {doc}`phillips_escaping_nash`, {doc}`phillips_priors` |
+| Phillips-curve slope | $\theta$ | {doc}`phillips_credibility` – {doc}`phillips_escaping_nash` |
+| government's beliefs | $\gamma$ | {doc}`phillips_adaptive` – {doc}`phillips_priors` |
+| | $\theta$ | {doc}`phillips_lost_conquest`, {doc}`phillips_drifts_volatilities` |
+| discount factor | $\delta$ | {doc}`phillips_credible_policies` – {doc}`phillips_learning` |
+| | $\beta$ | {doc}`phillips_lost_conquest`, {doc}`phillips_drifts_volatilities` |
+| learning gain | $\lambda$, $g_t$ | {doc}`phillips_adaptive`, {doc}`phillips_learning` |
+| | $\varepsilon$ | {doc}`phillips_escaping_nash`, {doc}`phillips_priors` |
+
+Three collisions are worth flagging in advance, because a reader carrying symbols forward will
+otherwise be caught by them.
+
+The letter $x$ changes sides: in {doc}`phillips_credibility` it is what the *public* expects,
+while from {doc}`phillips_escaping_nash` onward it is what the *government* sets.
+
+The two coincide in equilibrium, which is exactly what makes the switch easy to miss.
+
+The letter $\theta$ is the slope of the Phillips curve in the early lectures and the government's
+whole belief vector in the last two.
+
+And $\lambda$ works four shifts: the Cagan–Friedman adaptation parameter in
+{doc}`phillips_adaptive`, a forgetting factor in {doc}`phillips_learning`, a prior-tightening
+parameter in {doc}`phillips_priors`, and a measured persistence root in
+{doc}`phillips_lost_conquest`.
+
Let's start with some imports:
```{code-cell} ipython3
@@ -79,9 +125,15 @@ from statsmodels.tsa.filters.bk_filter import bkfilter
```
```{note}
-The figures in the next two sections reproduce the ones in chapters 1 and 2 of {cite}`Sargent1999`, which were drawn from data available in the late 1990s.
-We download the underlying series from [FRED](https://fred.stlouisfed.org/) and restrict attention to the same historical window.
-The section {ref}`phillips_after_1999` then carries the most enlightening of these figures through to the present and asks what the additional quarter-century of data means for the two stories.
+The figures in the next two sections reproduce the ones in chapters 1 and 2 of
+{cite}`Sargent1999`, which were drawn from data available in the late 1990s.
+
+We download the underlying series from [FRED](https://fred.stlouisfed.org/) and restrict
+attention to the same historical window.
+
+The section {ref}`phillips_after_1999` then carries the most enlightening of these figures
+through to the present and asks what the additional quarter-century of data means for the two
+stories.
```
## Facts
@@ -101,6 +153,12 @@ inflation_ma = inflation.rolling(13, center=True).mean()
```
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Monthly CPI inflation, 13-month centered moving average, 1948-1999
+ name: fig-ts-inflation
+---
fig, ax = plt.subplots(figsize=(9, 5))
ax.plot(inflation_ma, lw=1.2)
ax.axhline(0, color='k', lw=0.5)
@@ -136,6 +194,12 @@ data.head()
Figure 1.2 plots the two raw series together.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Monthly unemployment (white men 20+) and inflation rates
+ name: fig-ts-raw-series
+---
fig, ax = plt.subplots(figsize=(9, 5))
ax.plot(data.index, data['inflation'], 'C0', lw=1, label='inflation (CPI)')
ax.plot(data.index, data['unemployment'], 'C1:', lw=1.2,
@@ -159,6 +223,12 @@ bk.columns = ['inflation_cycle', 'unemployment_cycle']
```
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Business-cycle components of inflation and unemployment, Baxter-King bandpass filter
+ name: fig-ts-bandpass
+---
fig, ax = plt.subplots(figsize=(9, 5))
ax.plot(bk.index, bk['inflation_cycle'], 'C0', lw=1, label='inflation')
ax.plot(bk.index, bk['unemployment_cycle'], 'C1:', lw=1.2,
@@ -179,6 +249,12 @@ We can see the tradeoff more directly in a scatter plot for the subperiod that i
Figure 1.4 plots the raw series against each other, and Figure 1.5 the business-cycle components.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Inflation against unemployment, 1960-1982: raw series and business-cycle components"
+ name: fig-ts-scatter-6082
+---
sub = slice('1960', '1982')
fig, axes = plt.subplots(1, 2, figsize=(12, 5))
@@ -204,7 +280,14 @@ Focusing on the business-cycle components sharpens the apparent Phillips curve.
Figure 1.5 reveals **Phillips loops**: inflation and unemployment trace out counter-clockwise loops rather than a single stable curve, a signature of the shifting expectations that the natural-rate theory places at the center of the story.
```{note}
-The book adjusts for demographic change by choosing a single unemployment series. A broader definition of unemployment would inject additional low-frequency demographic components, which one might model with a unit-root process. The essay instead puts a unit root into the inflation-unemployment process from a different source: the *drifting beliefs* of a monetary authority cut loose from the discipline of Bretton Woods.
+The book adjusts for demographic change by choosing a single unemployment series.
+
+A broader definition of unemployment would inject additional low-frequency demographic
+components, which one might model with a unit-root process.
+
+The essay instead puts a unit root into the inflation-unemployment process from a different
+source: the *drifting beliefs* of a monetary authority cut loose from the discipline of Bretton
+Woods.
```
## Two interpretations
@@ -284,7 +367,7 @@ Within a **self-confirming equilibrium** (developed in {doc}`phillips_self_confi
Although the government's invariance assumption is wrong, it is not disappointed in outcomes, because those outcomes are statistically consistent with its beliefs.
-A self-confirming equilibrium is a rational expectations equilibrium with *fewer* free parameters than the models Lucas used — and precisely those lost parameters would be needed to represent regime changes.
+In a self-confirming equilibrium the government's beliefs are correct *along the equilibrium path*, so its forecasts satisfy the same cross-equation restrictions as a rational expectations equilibrium; but relative to the fully structural models Lucas used, the government's model carries *fewer* free parameters — and precisely those missing parameters would be needed to represent regime changes.
To admit regime changes and drifting coefficients, convergence to a self-confirming equilibrium must be *resisted*.
@@ -316,6 +399,8 @@ We begin, in {doc}`phillips_credibility`, by imposing rationality on *both* side
The one-period {cite}`KydlandPrescott1977` model delivers a pessimistic prediction — the high-inflation time-consistent (Nash) outcome — but a repeated-economy version of the theory of credible policy replaces that pessimism with *agnosticism*: so many outcomes become sustainable that the theory yields only weak predictions.
+{doc}`phillips_credible_policies` establishes this, computing the whole set of sustainable values with the recursive methods of {cite}`APS1990` and exhibiting three quite different equilibria that deliver an identical payoff.
+
That weakness is the first reason to hesitate before declaring the triumph of natural-rate theory.
We then turn back from the Lucas Critique and start again from the Phelps benchmark, but with one change: the government's model of the private sector is no longer arbitrary — it is *fit to historical data*.
@@ -328,7 +413,7 @@ Varying the details of that fitting problem generates the rest of the suite:
These adaptive models are a *disciplined* retreat from rational expectations, not an abandonment of it.
-They carry no free parameters governing expectations; period by period they impose the same cross-equation restrictions as a rational expectations model; and — because a self-confirming equilibrium is the attractor of their *mean dynamics* — they converge back to rational expectations under tranquil conditions, satisfying a desideratum of {cite}`Kreps1998`.
+They carry no free parameters governing expectations; period by period they impose the same cross-equation restrictions as a rational expectations model; and — because a self-confirming equilibrium is the attractor of their *mean dynamics* — under tranquil conditions they converge back to it, and hence, on the equilibrium path, to rational expectations, satisfying a desideratum of {cite}`Kreps1998`.
But, following {cite}`Sims1988`, our real interest is in the *recurrent* dynamics that adaptation adds.
@@ -404,6 +489,12 @@ Figure 1.1 showed inflation rising into the 1970s and falling under Volcker.
Extending it to the present adds three chapters the book could not see: the *Great Moderation* of low, stable inflation from the mid-1980s; a long spell near — and briefly below — zero after the 2008 financial crisis; and a sudden surge in 2021-2022 to the highest rate since 1981, followed by a rapid decline.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Inflation extended to the present, with the book's window shaded
+ name: fig-ts-inflation-long
+---
fig, ax = plt.subplots(figsize=(11, 5))
ax.plot(inflation_ma_full, lw=1)
ax.axhline(0, color='k', lw=0.5)
@@ -434,6 +525,12 @@ Figure 1.2 plotted the two series together for the post-war period.
Extending it shows the two most dramatic macroeconomic events of the new data: the COVID unemployment spike of 2020 — briefly the highest since the Great Depression — and the inflation surge that followed.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: Unemployment and inflation since 1990
+ name: fig-ts-recent
+---
recent = slice('1990', None)
fig, ax = plt.subplots(figsize=(11, 5))
@@ -462,6 +559,12 @@ The most striking post-1999 pattern is how *unstable* the inflation-unemployment
We split the sample into the book's acceleration era, the Great Moderation, and the post-2008 period, and plot inflation against unemployment in each.
```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: The inflation-unemployment scatter across three eras
+ name: fig-ts-three-eras
+---
scatter_data = pd.concat([inflation_yoy.rename('inflation'),
u_full.rename('unemployment')], axis=1).dropna()
@@ -481,8 +584,6 @@ plt.tight_layout()
plt.show()
```
-The three clouds could hardly look more different.
-
In 1960-1983 the points sprawl across a wide range of inflation rates — the era of shifting expectations and Phillips *loops*.
In 1984-2007 they collapse into a tight, low, nearly flat cloud — the Great Moderation, in which inflation barely responded to unemployment at all.
@@ -508,12 +609,20 @@ The surge is also a reminder that the monetary authority's *model* can still mis
The decade of near-zero inflation before 2020 — the apparently *flat* Phillips curve, with neither the "missing disinflation" of 2009-2013 nor the "missing inflation" of 2015-2019 fitting a stable curve — is precisely the kind of drifting empirical relationship whose changing slope and intercept the book's adaptive government tracks in real time.
```{note}
-A caveat the book itself would insist on: its mechanisms assume that the *fundamentals* — the true data-generating process — are stable, so that all the action comes from the government's evolving beliefs. The 2021-2022 episode involved genuine supply shocks (pandemic disruptions, energy prices), which lie outside that assumption. Disentangling shifting beliefs from shifting fundamentals is exactly the identification problem that makes this history so hard, and so interesting.
+A caveat the book itself would insist on: its mechanisms assume that the *fundamentals* — the
+true data-generating process — are stable, so that all the action comes from the government's
+evolving beliefs.
+
+The 2021-2022 episode involved genuine supply shocks (pandemic disruptions, energy prices),
+which lie outside that assumption.
+
+Disentangling shifting beliefs from shifting fundamentals is exactly the identification problem
+that makes this history so hard, and so interesting.
```
The tools built in the rest of this suite — self-confirming equilibria, drifting coefficients, and escape dynamics — remain a natural language for asking the question the new data pose: will a credible low-inflation equilibrium keep re-anchoring after each shock, or can a sequence of surprises still set beliefs drifting, as they did after 1965?
-The final lecture, {doc}`phillips_lost_conquest`, turns exactly these tools on the 2021-2022 surge, and asks why the Federal Reserve was so slow to respond.
+{doc}`phillips_lost_conquest` turns exactly these tools on the 2021-2022 surge, and asks why the Federal Reserve was so slow to respond, while the closing lecture {doc}`phillips_drifts_volatilities` asks the data directly whether the coefficients of post-war macroeconomic dynamics really drifted.
## Exercises
diff --git a/lectures/prospects_bounded_rationality.md b/lectures/prospects_bounded_rationality.md
new file mode 100644
index 000000000..d3648ecd4
--- /dev/null
+++ b/lectures/prospects_bounded_rationality.md
@@ -0,0 +1,683 @@
+---
+jupytext:
+ text_representation:
+ extension: .md
+ format_name: myst
+ format_version: 0.13
+ jupytext_version: 1.17.1
+kernelspec:
+ display_name: Python 3 (ipykernel)
+ language: python
+ name: python3
+---
+
+(prospects_bounded_rationality)=
+```{raw} jupyter
+
+```
+
+# 1993 Prospects for Bounded Rationality in Macroeconomics
+
+```{index} single: Bounded Rationality; Prospects
+```
+
+```{contents} Contents
+:depth: 2
+```
+
+## Overview
+
+This is the closing lecture of the series, and it is different in kind from the others.
+
+It carries no new model.
+
+Instead it gathers the judgments that {cite:t}`Sargent1993` recorded in 1993 — the opinions,
+reservations, and hopes with which he ended *Bounded Rationality in Macroeconomics* — and
+loops them back to the quest that opened the book, and that opened {doc}`bounded_rationality`,
+the first lecture here.
+
+That quest had a destination: a theory of **transition dynamics**, the out-of-equilibrium
+adjustment that the Eastern European reformers of 1989 had to manage with no map.
+
+The route was to expel the rational agents from our models and replace them with
+"artificially intelligent" agents who behave like **econometricians**, who gather data, form
+theories, estimate, and adapt.
+
+We can now ask how far that route carried us, by 1993, toward the destination.
+
+Sargent's own answer is a ledger, with entries on both sides.
+
+On the credit side: adaptive dynamics as a device for **selecting** among equilibria, and as
+a tool — evolutionary programming — for **computing** them.
+
+On the debit side: the original prize, a theory of transition dynamics, is largely unclaimed;
+and a striking asymmetry has emerged.
+
+The program set out to make the agents in our models behave more like econometricians.
+
+The econometricians, Sargent observes, have not returned the compliment.
+
+This lecture explains that asymmetry, gives it a concrete numerical face, and reads the ledger
+against the opening quest.
+
+The lecture then closes with a postscript on what became of the program after 1993.
+
+Let's start with some imports.
+
+```{code-cell} ipython3
+import numpy as np
+import matplotlib.pyplot as plt
+```
+
+## The quest, restated
+
+It is worth restating the argument of the first lecture, because everything below is measured
+against it.
+
+Rational expectations imposes two requirements: individual rationality, and mutual consistency
+of perceptions.
+
+The second is the demanding one, and — this is the crux — when a rational expectations model is
+taken to data, it imputes far more knowledge to the agents inside it than to the econometrician
+studying them.
+
+The agents evaluate their Euler equations using the *equilibrium* probability distributions,
+the very distributions the econometrician is still struggling to estimate.
+
+The bounded rationality program proposes to close that gap by demoting the agents to the
+econometrician's own level: they too must learn the distributions, from data, as they go.
+
+The hope was that this would deliver something rational expectations cannot: a description of
+the system *while it is still adjusting*, before beliefs and outcomes have settled into mutual
+consistency.
+
+That is what a theory of transition dynamics would be.
+
+We now take stock, following {cite:t}`Sargent1993`'s own accounting: first the debits, then the
+central asymmetry, then the credits.
+
+## The debit side: how much we must hard-wire
+
+The first reservation is about **arbitrariness**.
+
+Bounded rationality is most easily defined by what it is *not* — rational expectations — and
+that very malleability is a liability.
+
+Once we stop insisting that agents know the equilibrium, we must decide, case by case, exactly
+what they *do* know and how they learn it.
+
+Do they know their own utility and profit functions, or must they learn those too?
+
+Do they know calculus and dynamic programming, or only trial and error?
+
+Do they learn from their own experience alone, or from others'?
+
+To what class of approximating functions do we confine what they learn about?
+
+Every model in this series answered these questions by **hard-wiring**, by prompting the
+agents heavily, with an eye on the outcome we hoped they would reach.
+
+Bray's agents, in {doc}`bounded_rationality`'s adaptive-expectations model and in the least
+squares learning literature, know the correct supply curve and need only estimate one
+conditional expectation to plug into it.
+
+The Marcet–Sargent agents know dynamic programming, know their return function, and know the
+parametric form of the law of motion — they lack only its coefficients, which they update by
+vector autoregression.
+
+Even the classifier agents of {doc}`marimon_mcgrattan_sargent`, which are prompted far less —
+they are never told their utility functions, and recognize utility only when they experience
+it — are still told *when* to choose and *what* information to condition on, and the entire
+apparatus of their accounting system and genetic operators is designed by hand, with the
+Kiyotaki–Wright equilibrium in view.
+
+The second reservation is about **simplicity**.
+
+The learning tasks we set our agents are trivial next to those in a first econometrics course,
+let alone those that real firms and households are implicitly solving.
+
+We ask an agent to learn a single time-invariant decision rule, or a fixed collection of
+conditional expectations.
+
+We do not ask it to learn the parameters of a simultaneous-equations system, or to infer a
+mapping from policy regimes to distributions, the tasks that make econometrics hard.
+
+Put together, these reservations bear directly on the original prize.
+
+The environments into which we have cast our adaptive agents are far more stable and hospitable
+than the transitions we actually care about.
+
+Convergence-rate results are scarce, and tractability forces us to restrict the distribution of
+agents' beliefs severely.
+
+So the literature on adaptive processes, Sargent judged in 1993, falls well short of a secure
+foundation for a theory of real-time transition dynamics, the very thing the quest set out to
+find.
+
+He declines to end on that failure, though, and the baseball metaphor he reaches for is
+deliberately modest:
+
+> It would not be wise or fair to end this essay by dwelling on the failure of adaptive
+> methods so far to have 'hit a home run' by giving us a good theory of transition dynamics.
+> The problem of transition dynamics is difficult and long-standing. So maybe it should count
+> as a single, or at least a sacrifice fly, that these methods have sharpened our appreciation
+> of the problem.
+
+## Why the econometricians have not returned the compliment
+
+Here is the asymmetry that gives this lecture its theme.
+
+The bounded rationality program is, at bottom, a movement to make the agents in our models
+behave more like the econometricians who build and estimate those models.
+
+Imitation is the sincerest form of flattery.
+
+We might therefore have expected macroeconometricians to rush to fit these models to data.
+
+There was no rush.
+
+{cite:t}`Chung1990`'s estimation of the Sims policy-maker model — the application in
+{doc}`olg_adaptive_money` — was, Sargent noted, close to the *only* econometrically serious
+macroeconomic implementation of bounded rationality he knew of.
+
+Why the reluctance?
+
+The reasons are worth spelling out, because they are not about taste.
+
+The governing dictum among applied econometricians is Lucas's: *beware of theorists bearing
+free parameters*.
+
+Replacing a rational agent with a boundedly rational one **adds** parameters: parameters
+describing beliefs and how beliefs move.
+
+Take the simplest case, Bray's model.
+
+Relative to its rational expectations version, the adaptive version adds at least the initial
+belief and a parameter setting the gain sequence, and one might want more parameters to
+describe the gain's shape.
+
+That would already give Lucas's warning something to bite on.
+
+But there is a deeper problem, and it is the reason the added parameters are not just
+unwelcome but genuinely hard to estimate.
+
+Because the adaptive system **converges** to the rational expectations equilibrium, the extra
+parameters influence only the *transient*.
+
+The asymptotic distribution of the data contains no information about them.
+
+Let us see this directly.
+
+### The nuisance parameter, made concrete
+
+Bray's cobweb economy {cite:p}`Bray1982` sets the market price by
+
+$$
+p_t = a + b\, \beta_t + u_t,
+$$
+
+where $\beta_t$ is the price agents expect, formed by averaging past prices,
+
+$$
+\beta_t = \beta_{t-1} + \gamma_t\,(p_{t-1} - \beta_{t-1}),
+\qquad
+\gamma_t = \frac{1}{t + t_0},
+$$
+
+and $u_t$ is an IID shock.
+
+The constant $t_0$ is the weight the agents attach to the belief they start with, measured in
+observations: with $t_0 = 50$ they treat $\beta_0$ as though it summarized fifty prior prices, so
+that after $t$ periods $\beta_0$ still carries weight $t_0/(t + t_0)$.
+
+Some such weight is needed for the exercise to have any content.
+
+With $t_0 = 0$ the first gain is $\gamma_1 = 1$, so $\beta_1 = p_0$ exactly and the initial
+belief is erased after a single period — there would be no transient left for the
+econometrician to try to estimate.
+
+When $b < 1$ the belief converges to the rational expectations value $\beta^\star = a/(1-b)$,
+whatever value it started from.
+
+```{code-cell} ipython3
+a, b, sigma_u = 5.0, 0.7, 1.0
+t0 = 50 # weight on the initial belief
+β_star = a / (1 - b) # rational expectations belief
+
+def simulate(β0, u):
+ "Bray's cobweb under least squares learning, given a shock path u."
+ T = len(u)
+ β = np.empty(T)
+ p = np.empty(T)
+ β[0] = β0
+ for t in range(T):
+ p[t] = a + b * β[t] + u[t]
+ if t + 1 < T:
+ β[t + 1] = β[t] + (1 / (t + 1 + t0)) * (p[t] - β[t])
+ return p, β
+
+print(f"rational expectations belief β* = {β_star:.3f}")
+```
+
+The initial belief $\beta_0$ is the extra "bounded rationality" parameter.
+
+Watch three economies with very different initial beliefs forget where they started.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "The belief forgets its starting point"
+ name: fig-pbr-belief
+---
+rng = np.random.default_rng(0)
+u = rng.standard_normal(400)
+
+fig, ax = plt.subplots(figsize=(7.5, 4))
+for β0, colour in [(2.0, 'C0'), (16.667, 'C1'), (40.0, 'C2')]:
+ _, β = simulate(β0, u)
+ ax.plot(β, color=colour, lw=1.3, label=fr"$\beta_0 = {β0}$")
+ax.axhline(β_star, color='k', ls='--', lw=0.8, label=r"$\beta^\star$")
+ax.set_xlabel("$t$")
+ax.set_ylabel(r"belief $\beta_t$")
+ax.legend(frameon=False)
+plt.show()
+```
+
+All three converge to $\beta^\star$.
+
+Now put on the econometrician's hat.
+
+Given a sample of prices and the structural parameters $(a, b)$, the model implies a shock
+$\hat u_t = p_t - a - b\,\beta_t(\beta_0)$ for any candidate initial belief $\beta_0$, because
+the belief path is pinned down by $\beta_0$ and the observed prices.
+
+The sum of squared implied shocks measures how well a given $\beta_0$ fits the data.
+
+```{code-cell} ipython3
+def belief_path(β0, p):
+ "Belief sequence implied by an initial belief and an observed price path."
+ T = len(p)
+ β = np.empty(T)
+ β[0] = β0
+ for t in range(1, T):
+ β[t] = β[t - 1] + (1 / (t + t0)) * (p[t - 1] - β[t - 1])
+ return β
+
+def ssr(β0, p):
+ "Sum of squared implied shocks, as a function of the belief parameter β0."
+ β = belief_path(β0, p)
+ return np.sum((p - a - b * β) ** 2)
+```
+
+Whether the data can pin $\beta_0$ down is a question of how fast *information* about it
+accumulates as the sample grows.
+
+The natural way to read that off is the **excess** sum of squares — how much worse a wrong
+$\beta_0$ fits than the best one.
+
+For an ordinary, well-identified parameter every new observation adds to the penalty for being
+wrong, so the excess grows in proportion to $T$ and the confidence interval shrinks like
+$1/\sqrt{T}$.
+
+Watch what happens here.
+
+```{code-cell} ipython3
+---
+mystnb:
+ figure:
+ caption: "Information about the belief parameter stops accumulating"
+ name: fig-pbr-excess
+---
+grid = np.linspace(2, 40, 80)
+
+def excess_curve(T, seed=1):
+ "SSR(β₀) − min SSR across the β₀ grid, for a sample of length T."
+ u = np.random.default_rng(seed).standard_normal(T)
+ p, _ = simulate(β_star, u)
+ curve = np.array([ssr(b0, p) for b0 in grid])
+ return curve - curve.min()
+
+fig, ax = plt.subplots(figsize=(7.5, 4))
+for T, colour in zip((50, 500, 5000), ('C0', 'C1', 'C2')):
+ ax.plot(grid, excess_curve(T), color=colour, lw=1.5, label=f"$T = {T}$")
+ax.axvline(β_star, color='k', ls='--', lw=0.8)
+ax.set_xlabel(r"belief parameter $\beta_0$")
+ax.set_ylabel("excess sum of squares")
+ax.legend(frameon=False)
+plt.show()
+```
+
+The curves stack on top of one another instead of getting steeper.
+
+To see that this is not an artifact of the range plotted, compare the penalty for a wrong
+$\beta_0$ with the penalty for a wrong *slope* $b$, an ordinary structural parameter of the
+same model, estimated from the same data.
+
+```{code-cell} ipython3
+rows = []
+for T in (50, 500, 5_000, 50_000):
+ u = np.random.default_rng(1).standard_normal(T)
+ p, _ = simulate(β_star, u)
+ β = belief_path(β_star, p) # belief path at the true β0
+ wrong_β0 = ssr(2.0, p) - ssr(β_star, p)
+ wrong_slope = (np.sum((p - a - 0.75 * β) ** 2)
+ - np.sum((p - a - b * β) ** 2))
+ rows.append([T, wrong_β0, wrong_slope])
+
+for T, e_b0, e_b in rows:
+ print(f"T = {T:6d}: penalty for β₀ = 2 : {e_b0:10.1f} "
+ f"penalty for b = 0.75 : {e_b:12.1f}")
+```
+
+The two columns behave completely differently.
+
+The penalty for getting the slope wrong grows in proportion to the sample: a hundredfold more
+data makes a hundredfold stronger case against the wrong value, which is what identification
+looks like.
+
+The penalty for getting the initial belief wrong stops growing.
+
+Past a few thousand observations it is pinned at a constant, and every further observation is
+uninformative about $\beta_0$.
+
+The confidence interval for $\beta_0$ never shrinks; the parameter is not consistently
+estimable at all.
+
+This is the technical heart of the matter.
+
+The parameters that bounded rationality adds live entirely in the transient.
+
+A transient contributes a fixed, finite amount of information no matter how long we watch the
+economy afterwards, so those parameters become a **nuisance to estimate** — they enter the
+likelihood, and the data have only ever a bounded amount to say about them.
+
+And there is a final, decisive reason for the econometricians' cool response.
+
+Many applied macroeconometricians are in the market for methods that *reduce* the number of
+parameters needed to explain the data.
+
+A reduction is precisely what bounded rationality does not offer.
+
+It offers more parameters, most of them weakly identified.
+
+So the flattery ran one way.
+
+The theorists remade their agents in the econometrician's image; the econometricians, offered
+models full of extra, poorly-identified parameters, and warned by Lucas against exactly that,
+declined the gift.
+
+## The credit side: selection, computation, and a returned gift
+
+The ledger is not one-sided.
+
+Set against those debits, Sargent lists three genuine successes, and the last of them quietly
+undoes some of the asymmetry just described.
+
+**Equilibrium selection.**
+
+Where a rational expectations model has many equilibria, a system of adaptive agents often
+converges to a *particular* one, turning learning into a device for selecting among them.
+
+We saw it repeatedly: the low-inflation equilibrium chosen in {doc}`olg_adaptive_money`, the
+history-dependent exchange rate of {doc}`exchange_rate_learning`, the fundamental monetary
+equilibrium of {doc}`marimon_mcgrattan_sargent`.
+
+Sargent is candid that his affection for this use sits in some tension with his doubts about the
+dynamics that perform the selection:
+
+> I know that it is inconsistent to doubt the real-time dynamics but keep the equilibria
+> selected by them. I confess that my affection for the selection performed in the monetary
+> models described in Chapter 6 is partly driven by my prior conviction that the selected
+> equilibria seem sensible to me.
+
+**Evolutionary programming.**
+
+If a population of adaptive agents reliably converges to an equilibrium, we can run the
+population as a *method of computing* the equilibrium, especially in models too complicated to
+solve by hand.
+
+That is exactly what {doc}`marimon_mcgrattan_sargent` did with the five-good Kiyotaki–Wright
+economy, for which no analytical characterization was in hand.
+
+Sargent expected this use to be applied often.
+
+**New tools for the econometrician.**
+
+Here the asymmetry bends back on itself.
+
+The literatures on parallel and genetic algorithms have handed econometricians new
+computational gadgets — genetic algorithms and stochastic Gauss–Newton procedures among them —
+for solving their *own* estimation and optimization problems.
+
+McGrattan, for instance, used genetic algorithms to search for the neighborhood of a maximum
+before switching to a Newton method, a use noted back in {doc}`genetic_classifier`.
+
+So the econometricians took up the adaptive algorithms after all, not as *models* of the agents
+they study, but as *tools* in their own hands.
+
+The compliment was returned, but through a side door: the algorithms crossed over, the models
+did not.
+
+## Reading the ledger against the quest
+
+Line the two columns up against the destination we set out for.
+
+The **debit** column holds the prize itself.
+
+A theory of real-time transition dynamics — the map the Eastern European reformers lacked —
+remained, in 1993, largely unclaimed, blocked by arbitrariness, prompting, over-simple learning
+tasks, and a shortage of empirical traction.
+
+The **credit** column holds what the journey delivered along the way: a principled way to select
+among multiple equilibria, a practical way to compute equilibria that resist analysis, and a
+transfer of computational technique into econometrics itself.
+
+The organizing image of the whole book is the gap between the econometrician and the agents
+inside the model.
+
+Rational expectations closes that gap by lifting the agents up to a knowledge the econometrician
+lacks.
+
+Bounded rationality proposed to close it from the other side, by bringing the agents down to
+the econometrician's level and making them learn.
+
+By 1993 the program had not, on Sargent's own accounting, delivered the transition dynamics that
+motivated it, and the econometricians it sought to imitate had kept the models at arm's length —
+for the sound reason that those models add parameters the data cannot identify, when what the
+econometrician wants is fewer.
+
+But it had sharpened the questions, selected equilibria, computed them, and lent its tools to
+the very econometricians who declined its models.
+
+That was the state of the prospects for bounded rationality in macroeconomics, as they stood in
+1993.
+
+## Postscript: what became of the program
+
+Three decades is long enough to see which of the 1993 worries were permanent and which were
+about a field that had not yet found its footing.
+
+### The arbitrariness was disciplined
+
+The first reservation was that once we stop insisting agents know the equilibrium, nothing tells
+us what to put in its place.
+
+The discipline that emerged is **expectational stability**.
+
+Evans and Honkapohja {cite:p}`EvansHonkapohja2001` showed that whether a rational expectations
+equilibrium is learnable is governed by a condition on the map from perceived to actual laws of
+motion — the very $T$ map of {doc}`bounded_rationality` — and that the condition is largely
+*independent* of the details of the learning algorithm.
+
+That is exactly what was missing in 1993.
+
+Selection is no longer an artifact of whichever recursion the modeller happened to write down;
+a large class of reasonable learning rules select the same equilibria, and one can check which
+those are without simulating anything.
+
+The stability reversal of {doc}`olg_adaptive_money` is a case in point.
+
+It looked in 1993 like a fact about least squares, propped up by {cite:t}`BrunoFischer1990`
+having found the same thing with a different estimator.
+
+E-stability explains why the two agreed.
+
+### The transition dynamics arrived, in a narrower form than hoped
+
+The prize was a theory of out-of-equilibrium adjustment.
+
+What the program delivered instead was a theory of *departures from* equilibrium: escape
+dynamics.
+
+The sawtooth we simulated at the end of {doc}`olg_adaptive_money` was, in 1993, a numerical
+curiosity.
+
+{cite:t}`ChoWilliamsSargent2002` characterized it analytically with large-deviations theory,
+computing the most likely escape path and the rate at which escapes occur, and
+{cite:t}`Williams2019` extended the characterization considerably.
+
+The QuantEcon lectures {doc}`phillips_escaping_nash` and {doc}`phillips_priors` work through
+both.
+
+This is less than the original quest asked for.
+
+It describes recurrent excursions away from a self-confirming equilibrium, not the arrival of a
+market economy in a country that never had one.
+
+But it is a genuine theory of a system that does not settle down, derived rather than
+simulated, and in 1993 there was none.
+
+### The econometricians did return the compliment
+
+Here the 1993 assessment was simply overtaken.
+
+The obstacle, we saw above, was that the added parameters live in a vanishing transient.
+
+But that argument applies only to a learning scheme with a $1/t$ gain, which converges and then
+stops moving.
+
+*Constant-gain learning has no such transient.* Beliefs never settle; they keep moving forever,
+and their movement is part of the stationary distribution of the data.
+
+So the gain is identified the way an ordinary structural parameter is — from the whole sample,
+at the usual rate — and not, like $\beta_0$, from a bounded initial episode.
+
+Let us check that on the model we have been using.
+
+```{code-cell} ipython3
+def simulate_constant_gain(β0, gain, u):
+ "Bray's cobweb when agents discount old prices at a fixed rate."
+ T = len(u)
+ β, p = np.empty(T), np.empty(T)
+ β[0] = β0
+ for t in range(T):
+ p[t] = a + b * β[t] + u[t]
+ if t + 1 < T:
+ β[t + 1] = β[t] + gain * (p[t] - β[t])
+ return p, β
+
+def ssr_gain(g_hat, p, β0):
+ "Fit criterion for a candidate gain, given observed prices."
+ T = len(p)
+ β = np.empty(T)
+ β[0] = β0
+ for t in range(1, T):
+ β[t] = β[t - 1] + g_hat * (p[t - 1] - β[t - 1])
+ return np.sum((p - a - b * β) ** 2)
+
+gain_true = 0.05
+for T in (500, 5_000, 50_000):
+ penalties = []
+ for seed in range(20): # average out sampling noise
+ u = np.random.default_rng(seed).standard_normal(T)
+ p, _ = simulate_constant_gain(β_star, gain_true, u)
+ penalties.append(ssr_gain(0.08, p, β_star) - ssr_gain(gain_true, p, β_star))
+ print(f"T = {T:6d}: mean penalty for using gain 0.08 instead of 0.05 : "
+ f"{np.mean(penalties):9.1f}")
+```
+
+The penalty grows in proportion to the sample, exactly as it did for the slope $b$ and exactly as
+it did *not* for the initial belief.
+
+That is why the econometric work that eventually materialized uses constant gain.
+
+{cite:t}`SargentWilliamsZha2006` estimated a constant-gain learning model of the Federal
+Reserve on post-war U.S. data — imputing to the government inside the model a genuine recursive
+estimation procedure, and asking the data which gain it used.
+
+{cite:t}`SargentWilliams2005` studied how the government's prior about drifting coefficients
+shapes what it converges to, and {doc}`phillips_priors` develops that.
+
+{doc}`phillips_drifts_volatilities` fits a drifting-coefficient VAR to the same episode and
+asks whether it was bad policy or bad luck.
+
+So the flattery did run both ways in the end.
+
+It took a change in the learning technology — from a scheme that converges to one that never
+does — to make the models estimable, and that change was made for reasons of economics rather
+than econometrics.
+
+### A second retreat, made differently
+
+The program in this book keeps individual rationality and gives up mutual consistency: agents
+optimize, but against beliefs they are still estimating.
+
+A parallel literature retreats along the other axis.
+
+In the **robustness** work of Hansen and Sargent {cite:p}`HansenSargent2008`, agents do not
+estimate their model at all.
+
+They admit that they cannot know it, and optimize against the worst case among the models they
+cannot rule out.
+
+The two are complements rather than rivals, and both are answers to the question that opens
+{doc}`bounded_rationality`: what do we do about the knowledge that rational expectations
+imputes?
+
+One answer is that agents should learn what the econometrician is learning.
+
+The other is that they should behave well without ever learning it.
+
+### The algorithms kept crossing over
+
+The 1993 ledger noted that econometricians had adopted the adaptive algorithms as computational
+tools even while declining the models.
+
+That traffic increased, and reversed direction again.
+
+Holland's bucket brigade of {doc}`genetic_classifier` — pay part of your reward backward to
+whatever set you up — is *temporal-difference learning*, which became the organizing idea of
+modern reinforcement learning {cite:p}`Sutton_2018`.
+
+The classifier systems of {doc}`marimon_mcgrattan_sargent` are recognizable, in retrospect, as
+reinforcement learners with a hand-built function approximator; and the perceptrons of
+{doc}`genetic_classifier` became the deep networks of {doc}`back_prop`, which now serve as the
+function approximators.
+
+Sargent's artificially intelligent agents were not, it turned out, a metaphor borrowed from a
+neighboring field.
+
+They were an early instance of what that field went on to build.
+
+### Reading the ledger again
+
+The 1993 debits have not all been paid.
+
+The choices remain many, the learning tasks we set our agents remain simple next to the ones real
+firms solve, and no one has produced the theory of transition dynamics that the Eastern European
+reforms called for.
+
+But the entries have moved.
+
+Selection acquired a theory instead of a set of examples; non-convergence acquired an
+analytical characterization instead of a simulation; and the econometricians, offered a version
+of the models whose extra parameters the data could actually speak to, took them up.
+
+The gap between the econometrician and the agents inside the model is still there.
+
+It is narrower, and we now know a good deal about its width.