\documentclass[12pt,oneside,openany]{book}
\usepackage[T1]{fontenc}
\usepackage[utf8]{inputenc}
\usepackage{amsmath,amssymb,amsthm,mathtools,mathrsfs}
\usepackage{booktabs,array}
\usepackage[numbers,sort&compress]{natbib}
\usepackage{imakeidx}% required by \indexsetup in shared/trilogy-style.tex
\usepackage{environ}
\PassOptionsToPackage{hypertexnames=false}{hyperref}
%% trilogy-style.tex -- the shared typographic design of the three volumes.
%%
%% Loaded by each notes/Volume_*/source/main.tex with \input{../../shared/trilogy-style}
%% (builds run in source/, so the relative path resolves), after the class, fontenc and
%% inputenc, the AMS packages, natbib or biblatex, and imakeidx, and before the volume's own
%% theorem declarations and macros. Volume-specific material (title, theorem names, local
%% notes, short running heads) stays in the volume's main.tex.
%%
%% Principles: readability first, then understated elegance. The volumes are read mostly
%% on screen, so the page is set for that: 12pt type on a full-width letter page with 1in
%% side margins, one-sided, with the same quiet head on every page. Black text only; one
%% readable face with matching mathematics; headings distinguished by size, weight and
%% space rather than colour or rules; statements in the standard amsthm manner; classic
%% title pages that read as one series. Each main.tex uses \documentclass[12pt,oneside,
%% openany]{book}.
\makeatletter
%% ---------------------------------------------------------------------------
%% Type: Palatino for text and mathematics (mathpazo, with true small capitals).
%% Its larger x-height and sturdier strokes read better than Latin Modern, on screen
%% and in dense formulas, and the math italic and symbols are drawn to match the text.
%% Latin Modern supplies the sans-serif and typewriter faces (URLs). bm comes after
%% the math fonts so that \bm finds bold Palatino.
%% ---------------------------------------------------------------------------
\usepackage{lmodern}
\usepackage[sc]{mathpazo}
\usepackage{bm}
\linespread{1.1}
\usepackage{microtype}
% Font expansion limited to 1.5% (default 2%): steadier colour, and it avoids the
% hairline overfull lines that maximal shrink plus protrusion can leave.
\microtypesetup{stretch=15,shrink=15}
%% ---------------------------------------------------------------------------
%% Page, set for reading on screen: US letter, 1in side margins (a 6.5in line, about
%% 85 characters of Palatino 12pt), and slightly smaller top and bottom margins with
%% the head and foot inside them, so a page fitted to the window shows large type.
%% ---------------------------------------------------------------------------
\usepackage{geometry}
\geometry{letterpaper,left=1in,right=1in,top=0.75in,bottom=0.75in,includeheadfoot,
headheight=15pt,headsep=16pt,footskip=26pt}
\setlength{\parindent}{1.5em}
\setlength{\parskip}{0pt}
\setlength{\emergencystretch}{3em}
% A paragraph may set a slightly loose line rather than push a word into the margin.
\tolerance=1000
\clubpenalty=3000 \widowpenalty=3000 \displaywidowpenalty=1500
\raggedbottom
\allowdisplaybreaks[2]
%% Lists: compact, indented like a paragraph.
\usepackage{enumitem}
\setlist{topsep=4pt plus 2pt minus 1pt,itemsep=2pt plus 1pt,parsep=0pt}
\setlist[itemize]{leftmargin=1.5em}
\setlist[enumerate]{leftmargin=2em}
\usepackage[newparttoc]{titlesec}
\usepackage{titletoc,fancyhdr}
%% Letterspaced small capitals, for labels such as "Chapter 3" and the series name.
\newcommand{\trilogy@spacedsc}[1]{{\scshape\textls[90]{\MakeLowercase{#1}}}}
%% Headings are ragged right and unhyphenated; a title that needs two lines is broken
%% into lines of similar length rather than leaving a single word on the second line.
\newcommand{\trilogy@headpar}{\rightskip=0pt plus 0.5\hsize\relax
\parfillskip=0pt plus 0.5\hsize\relax\hyphenpenalty=10000\relax\exhyphenpenalty=10000\relax}
%% ---------------------------------------------------------------------------
%% Headings (titlesec): a regular-weight chapter title under a spaced small-capitals
%% label; bold section titles; plain part pages; no colour and no rules.
%% ---------------------------------------------------------------------------
\titleformat{\part}[display]
{\normalfont\centering}
{\large\trilogy@spacedsc{\partname\ \thepart}}
{22pt}
{\fontsize{22}{28}\selectfont}
\assignpagestyle{\part}{empty}
\titleformat{\chapter}[display]
{\normalfont\trilogy@headpar}
{\trilogy@spacedsc{\chaptertitlename\ \thechapter}}
{12pt}
{\fontsize{20}{25}\selectfont}
\titleformat{name=\chapter,numberless}[display]
{\normalfont\trilogy@headpar}
{}
{0pt}
{\fontsize{20}{25}\selectfont\trilogy@starmark}
\titlespacing*{\chapter}{0pt}{12pt}{30pt}
% An unnumbered chapter (front matter, contents, references, index) sets both running heads.
\newcommand{\trilogy@starmark}[1]{#1\markboth{#1}{#1}}
\titleformat{\section}
{\normalfont\fontsize{13.5}{17}\selectfont\bfseries\trilogy@headpar}{\thesection}{0.75em}{}
\titlespacing*{\section}{0pt}{22pt plus 4pt minus 2pt}{9pt plus 2pt}
\titleformat{\subsection}
{\normalfont\normalsize\bfseries\trilogy@headpar}{\thesubsection}{0.75em}{}
\titlespacing*{\subsection}{0pt}{15pt plus 3pt minus 2pt}{6pt plus 1pt}
\titleformat{\paragraph}[runin]
{\normalfont\normalsize\bfseries}{}{0pt}{}
\titlespacing*{\paragraph}{0pt}{9pt plus 2pt minus 1pt}{0.6em}
\setcounter{tocdepth}{1}
\setcounter{secnumdepth}{1}
%% ---------------------------------------------------------------------------
%% Running heads (one-sided, the same on every page): the chapter number and title at
%% the left, and at the right the section number and title followed by the page number,
%% all in small italic except the numbers, with no rule. When chapter and section titles
%% do not both fit on the line, the right head shows only the section number (ยง1.2). Chapter-
%% opening pages carry only a centred page number at the foot; part pages carry nothing.
%%
%% A long title can be given a short form in the volume's main.tex:
%% \shorthead{
}{}
%% A chapter head too wide for the line falls back to "Chapter N" (with a warning).
%% ---------------------------------------------------------------------------
\newcommand{\shorthead}[2]{\expandafter\def\csname trilogy@sh@\detokenize{#1}\endcsname{#2}}
\newcommand{\trilogy@lookup}[1]{%
\ifcsname trilogy@sh@\detokenize{#1}\endcsname
\expandafter\let\expandafter\trilogy@headtext\csname trilogy@sh@\detokenize{#1}\endcsname
\else
\def\trilogy@headtext{#1}%
\fi}
\newcommand{\trilogy@headfont}{\normalfont\small\itshape}
% A mark is \trilogy@hd{number}{title}; the number is empty for unnumbered chapters.
\DeclareRobustCommand{\trilogy@hd}[2]{%
\ifx\relax#1\relax\else{\upshape#1}\hspace{0.8em}\fi#2}
\newsavebox{\trilogy@chapbox}
\newsavebox{\trilogy@secbox}
\newsavebox{\trilogy@cachedchapbox}
\newsavebox{\trilogy@cachedsecbox}
\newif\iftrilogy@headtoolong
\newif\iftrilogy@cachedheadtoolong
\let\trilogy@cachedheadkey\relax
% Compute the two boxes only when their marks or dimensions change.
\newcommand{\trilogy@computeheads}{%
\trilogy@headtoolongfalse
\sbox\trilogy@chapbox{\trilogy@headfont\nouppercase{\leftmark}}%
\sbox\trilogy@secbox{}%
\protected@edef\trilogy@l{\leftmark}%
\protected@edef\trilogy@r{\rightmark}%
\ifx\trilogy@r\@empty\else\ifx\trilogy@r\trilogy@l\else
\sbox\trilogy@secbox{\trilogy@headfont\nouppercase{\rightmark}}%
\fi\fi
\dimen@=\wd\trilogy@chapbox\advance\dimen@ 4em\relax
\ifdim\dimen@>\headwidth
\sbox\trilogy@chapbox{\trilogy@headfont\@chapapp\ \thechapter}%
\trilogy@headtoolongtrue
\fi
\dimen@=\wd\trilogy@chapbox\advance\dimen@\wd\trilogy@secbox\advance\dimen@ 8em\relax
\ifdim\dimen@>\headwidth
\sbox\trilogy@secbox{\trilogy@headfont
\expandafter\let\csname trilogy@hd \endcsname\trilogy@hdnum\nouppercase{\rightmark}}%
\fi}
% Both fancyhdr fields, and successive pages with the same marks, reuse these
% boxes. The cache is global because fancyhdr evaluates its fields in groups;
% the page number and page-specific warnings remain outside the cached boxes.
\newcommand{\trilogy@heads}[1]{%
\protected@edef\trilogy@headkey{%
{\leftmark}{\rightmark}{\the\headwidth}%
{\fontname\font}{\the\dimexpr1em\relax}{\thechapter}{\@chapapp}}%
\ifx\trilogy@headkey\trilogy@cachedheadkey\else
\trilogy@computeheads
\global\setbox\trilogy@cachedchapbox=\copy\trilogy@chapbox
\global\setbox\trilogy@cachedsecbox=\copy\trilogy@secbox
\global\let\trilogy@cachedheadkey\trilogy@headkey
\iftrilogy@headtoolong
\global\trilogy@cachedheadtoolongtrue
\else
\global\trilogy@cachedheadtoolongfalse
\fi
\fi
\setbox\trilogy@chapbox=\copy\trilogy@cachedchapbox
\setbox\trilogy@secbox=\copy\trilogy@cachedsecbox
\iftrilogy@cachedheadtoolong
\ifx\relax#1\relax\else\PackageWarningNoLine{trilogy-style}{Running head too long on
page \thepage; give the chapter a \string\shorthead}\fi
\fi}
% The section number alone, for a right head with no room for the section title. (A mark
% holds the robust command's inner macro "\trilogy@hd ", which is what is replaced above.)
\newcommand{\trilogy@hdnum}[2]{\ifx\relax#1\relax\else\S\,{\upshape#1}\fi}
\pagestyle{fancy}
\fancyhf{}
\fancyhead[L]{\trilogy@heads{warn}\usebox\trilogy@chapbox}
\fancyhead[R]{\trilogy@heads{}%
\ifdim\wd\trilogy@secbox>\z@\usebox\trilogy@secbox\hspace{2em}\fi
\normalfont\small\thepage}
\renewcommand{\headrulewidth}{0pt}
\renewcommand{\footrulewidth}{0pt}
\fancypagestyle{plain}{\fancyhf{}\fancyfoot[C]{\normalfont\small\thepage}%
\renewcommand{\headrulewidth}{0pt}\renewcommand{\footrulewidth}{0pt}}
\renewcommand{\chaptermark}[1]{\trilogy@lookup{#1}%
\markboth{\trilogy@hd{\ifnum\c@secnumdepth>\m@ne\if@mainmatter\thechapter\fi\fi}{\trilogy@headtext}}{}}
\renewcommand{\sectionmark}[1]{\trilogy@lookup{#1}%
\markright{\trilogy@hd{\ifnum\c@secnumdepth>\z@\thesection\fi}{\trilogy@headtext}}}
%% ---------------------------------------------------------------------------
%% Table of contents (titletoc): parts in spaced small capitals, chapters in roman with
%% hanging numbers, sections indented and smaller with sparse leaders.
%% ---------------------------------------------------------------------------
\contentsmargin{2.2em}
\titlecontents{part}[0pt]
{\addvspace{16pt plus 2pt}\normalfont}
{\trilogy@spacedsc{\partname\ \thecontentslabel}\hspace{1em}}
{}
{\hfill\contentspage}
[\addvspace{4pt}]
\titlecontents{chapter}[2.2em]
{\addvspace{7pt plus 1pt}\normalfont}
{\contentslabel{2.2em}}
{\hspace*{-2.2em}}
{\hfill\contentspage}
\titlecontents{section}[4.8em]
{\normalfont\small}
{\contentslabel{2.6em}}
{\hspace*{-2.6em}}
{\titlerule*[0.9em]{.}\contentspage}
%% ---------------------------------------------------------------------------
%% Index and bibliography: the index in two columns of small type; the bibliography
%% is "References" in every volume, listed in the contents.
%% ---------------------------------------------------------------------------
% The last index page is balanced by multicol, which by default accepts a balanced column up
% to 12pt (a line) too long, so that one added entry can push a line into the bottom margin.
% With no overflow allowed, multicol fills that page normally and balances the rest on the
% next page instead; ragged columns keep balanced columns at their natural height.
\indexsetup{othercode={\small\raggedcolumns\maxbalancingoverflow=\z@}}
\@ifpackageloaded{natbib}{%
\renewcommand{\bibsection}{\chapter*{References}%
\addcontentsline{toc}{chapter}{References}}%
}{}
%% ---------------------------------------------------------------------------
%% Statements, examples and notes (amsthm). The three standard styles are redefined
%% with one generous spacing, so \theoremstyle{plain|definition|remark} in a volume's
%% main.tex picks up the design:
%% plain bold label, italic body (Statement, Theorem, Lemma, ...)
%% definition bold label, upright body (Definition, Example, Technical walkthrough)
%% remark italic label, upright body (Caution, Seminar question)
%% ---------------------------------------------------------------------------
\newlength{\trilogyenvskip}
\setlength{\trilogyenvskip}{10pt plus 3pt minus 2pt}
\newtheoremstyle{plain}{\trilogyenvskip}{\trilogyenvskip}
{\itshape}{}{\bfseries}{.}{0.5em}{}
\newtheoremstyle{definition}{\trilogyenvskip}{\trilogyenvskip}
{\normalfont}{}{\bfseries}{.}{0.5em}{}
\newtheoremstyle{remark}{\trilogyenvskip}{\trilogyenvskip}
{\normalfont}{}{\itshape}{.}{0.5em}{}
%% Run-in notes ("Used in:", "Sources.", ...): an italic label and ordinary text,
%% a little space above. \trilogynote{}{}
\newcommand{\trilogynote}[2]{\par\addvspace{5pt plus 1pt}\noindent{\itshape #1}\enspace #2\par}
%% "Used in:" line after a statement: \usedin{FRSB Lemma 2.5; PF Lemma 2.1}
\newcommand{\usedin}[1]{\trilogynote{Used in:}{#1}}
%% Technical walkthrough: a numbered, multi-paragraph guided calculation,
%% \begin{walkthrough}[] ... \end{walkthrough}. Definition style;
%% it ends with an open square at the right margin, like a proof, so the reader sees
%% where it stops. A volume may \renewcommand{\walkthroughname}{...}.
\newcommand{\walkthroughname}{Technical walkthrough}
\newcommand{\walkthroughend}{\ensuremath{\square}}
%% Set-off statement, for a theorem quoted from one of the six papers (Vol II's "Main
%% result"): definition style, bold label and upright text, with the whole statement
%% indented 1.5em on both sides like the asides, instead of a box. Displays stay centred.
%% Declare with \theoremstyle{trilogysetoff}\newtheorem{}{}[chapter].
\newtheoremstyle{trilogysetoff}{\trilogyenvskip}{\trilogyenvskip}
{\normalfont\leftskip=1.5em\rightskip=1.5em\relax}{}{\bfseries}{.}{0.5em}{}
\theoremstyle{definition}
\newtheorem{trilogywalk}{\walkthroughname}[chapter]
\newenvironment{walkthrough}[1][]
{\ifx\relax#1\relax\begin{trilogywalk}\else\begin{trilogywalk}[#1]\fi
\pushQED{\qed}\renewcommand{\qedsymbol}{\walkthroughend}}
{\popQED\end{trilogywalk}}
\theoremstyle{plain}
%% Set-off asides, \begin{sayit} ... \end{sayit} ("How to say it") and
%% \begin{fiveminute} ... \end{fiveminute} ("Five-minute explanation"): a spaced
%% small-capitals label and a body indented on both sides; no box, no rule, no tint.
%% Inside sayit, \tophysicist and \tomathematician start the two voices.
\newenvironment{trilogyaside}[1]
{\par\addvspace{11pt plus 3pt minus 2pt}%
\list{}{\leftmargin=1.5em\rightmargin=1.5em\topsep=0pt\partopsep=0pt\parsep=0pt
\listparindent=1.5em\itemsep=0pt}%
\item\relax{\trilogy@spacedsc{#1}}\par\nopagebreak\vspace{3pt}\noindent\ignorespaces}
{\endlist\par\addvspace{11pt plus 3pt minus 2pt}}
\newenvironment{sayit}{\begin{trilogyaside}{How to say it}}{\end{trilogyaside}}
\newenvironment{fiveminute}{\begin{trilogyaside}{Five-minute explanation}}{\end{trilogyaside}}
\newcommand{\tophysicist}{\par\addvspace{2pt}\noindent{\itshape To a physicist.}\enspace\ignorespaces}
\newcommand{\tomathematician}{\par\addvspace{4pt}\noindent{\itshape To a mathematician.}\enspace\ignorespaces}
%% Keeping a heading with what follows it: \trilogyneedspace{} starts a new page at
%% once unless at least of the current page remains. The test is made on the spot,
%% so it also works before a longtable (a breakpoint left for later would be taken under
%% longtable's output routine, which then repeats the table head above the heading).
\newcommand{\trilogyneedspace}[1]{\par\penalty-100\begingroup
\setlength{\dimen@}{#1}\dimen@ii\pagegoal\advance\dimen@ii-\pagetotal
\ifdim\dimen@>\dimen@ii\ifdim\dimen@ii>\z@\vfil\fi\break\fi\endgroup}
%% Captions: small type (11pt at the 12pt base), "Table A.1." with a full stop.
\usepackage[font=small,labelsep=period,skip=8pt]{caption}
%% Footnotes in \small rather than \footnotesize, for legibility on screen.
\usepackage{etoolbox}
\patchcmd{\@footnotetext}{\footnotesize}{\small}{}{}
%% ---------------------------------------------------------------------------
%% Wide reference tables. \widetablesetup, issued inside a group just before a
%% longtable, lets that table run \trilogy@overhang into each margin, centred on the
%% page: within the group \linewidth is the measure plus twice the overhang, so
%% p{\linewidth} columns scale with it. With the 6.5in screen measure the
%% overhang is zero; the command is kept so that tables using it need no change.
%% ---------------------------------------------------------------------------
\newcommand{\trilogy@overhang}{0pt}
\newcommand{\widetablesetup}{%
\setlength{\LTleft}{-\trilogy@overhang plus 1fill}%
\setlength{\LTright}{-\trilogy@overhang plus 1fill}%
\setlength{\linewidth}{\dimexpr\textwidth+2\dimexpr\trilogy@overhang\relax\relax}}
%% ---------------------------------------------------------------------------
%% Links: black and unboxed; they still work. No author in the PDF metadata.
%% The PDF opens on the document alone (no side panel), at full page width,
%% with the document title (not the file name) in the window bar.
%% ---------------------------------------------------------------------------
\usepackage[hidelinks,bookmarksnumbered=true]{hyperref}
\usepackage{bookmark}
\hypersetup{pdfauthor={},pdfpagemode=UseNone,pdfdisplaydoctitle=true,
pdfstartview={FitH \hypercalcbp{\paperheight}}}
%% ---------------------------------------------------------------------------
%% Title page. \trilogytitlepage{}{}{}
%% Title and subtitle may contain \\ for a chosen line break.
%% ---------------------------------------------------------------------------
\newcommand{\trilogyseries}{Notes on Spin Glasses}
\newcommand{\trilogytitlepage}[3]{%
\hypersetup{pageanchor=false}%
\begin{titlepage}
\centering\normalfont
% Fixed positions, so the three title pages line up: the series line about 2.9in
% from the top of the paper, the title about 4.3in, the block a little above centre.
\vspace*{1.7in}
{\large\trilogy@spacedsc{\trilogyseries}\par}
\vspace{12pt}
{\normalsize\itshape Volume #1\par}
\vspace{1.05in}
{\fontsize{28}{35}\selectfont #2\par}
\vspace{24pt}
{\Large\itshape #3\par}
\vfill
\end{titlepage}%
\hypersetup{pageanchor=true}}
\makeatother
\usepackage[nameinlink,capitalize,noabbrev]{cleveref}
\usepackage{aliascnt}
\hypersetup{pdftitle={The Parisi Formula for the Edwards--Anderson Model in the High-Dimensional Limit},pdfauthor={P. M. Aronow and Patrick Lopatto}}
% The paper is a single part. Its top-level units are sections: they are set with the
% design's chapter headings, labelled "Section", and listed as such in the contents; their
% subsections are the design's numbered sections.
\renewcommand{\chaptername}{Section}
\numberwithin{equation}{chapter}
\theoremstyle{plain}
\newtheorem{theorem}{Theorem}[chapter]
\newaliascnt{proposition}{theorem}
\newtheorem{proposition}[proposition]{Proposition}
\aliascntresetthe{proposition}
\newaliascnt{lemma}{theorem}
\newtheorem{lemma}[lemma]{Lemma}
\aliascntresetthe{lemma}
\newaliascnt{corollary}{theorem}
\newtheorem{corollary}[corollary]{Corollary}
\aliascntresetthe{corollary}
\theoremstyle{definition}
\newaliascnt{definition}{theorem}
\newtheorem{definition}[definition]{Definition}
\aliascntresetthe{definition}
\theoremstyle{remark}
\newaliascnt{remark}{theorem}
\newtheorem{remark}[remark]{Remark}
\aliascntresetthe{remark}
\crefname{theorem}{Theorem}{Theorems}
\crefname{proposition}{Proposition}{Propositions}
\crefname{lemma}{Lemma}{Lemmas}
\crefname{corollary}{Corollary}{Corollaries}
\crefname{definition}{Definition}{Definitions}
\crefname{remark}{Remark}{Remarks}
\crefname{chapter}{Section}{Sections}
\crefname{section}{Section}{Sections}
\crefname{subsection}{Section}{Sections}
\newcommand{\crefrangeconjunction}{--}
% Serial comma in lists of three or more references.
\newcommand{\creflastconjunction}{, and~}
\newcommand{\creflastgroupconjunction}{, and~}
% A centered block of small type on a narrower measure, with an optional spaced
% small-capitals label: the authors' note and the abstract.
\makeatletter
\newcommand{\npp@noteblock}[2]{%
\ifx\relax#1\relax\else{\centering\trilogy@spacedsc{#1}\par}\vspace{6pt}\fi
{\centering
\begin{minipage}{0.84\textwidth}\small\normalfont\setlength{\parindent}{1.5em}%
\noindent\ignorespaces#2\end{minipage}\par}}
% The authors' note and the abstract each have a page of their own, after the title page,
% set a little above the middle of the page and with no head or foot.
\NewEnviron{authorsnote}{%
\clearpage\thispagestyle{empty}%
\vspace*{\stretch{2}}%
\npp@noteblock{Authors' note}{\BODY}%
\vspace*{\stretch{3}}%
\clearpage}
\NewEnviron{abstractpage}{%
\clearpage\thispagestyle{empty}%
\vspace*{\stretch{2}}%
\npp@noteblock{Abstract}{\BODY}%
\vspace*{\stretch{3}}%
\clearpage}
% Title page in the design of \trilogytitlepage: the title at the same position, and the
% authors, one to a line, in the lower half of the page.
% \npptitlepage{}{}
\newcommand{\npptitlepage}[2]{%
\hypersetup{pageanchor=false}%
\begin{titlepage}
\centering\normalfont
\vspace*{1.7in}
{\large\strut\par}
\vspace{1.05in}
{\fontsize{28}{35}\selectfont #1\par}
\vspace{1.1in}
{\Large\def\\{\par\vspace{10pt}}#2\par}
\vfill
\end{titlepage}%
\hypersetup{pageanchor=true}}
\makeatother
\newcommand{\R}{\mathbb R}
\newcommand{\Z}{\mathbb Z}
\newcommand{\N}{\mathbb N}
\newcommand{\E}{\mathbb E}
\renewcommand{\P}{\mathbb P}
\newcommand{\T}{\mathbb T}
\newcommand{\Var}{\operatorname{Var}}
\newcommand{\Cov}{\operatorname{Cov}}
\newcommand{\sgn}{\operatorname{sgn}}
\newcommand{\supp}{\operatorname{supp}}
\newcommand{\tr}{\operatorname{tr}}
\newcommand{\dd}{\,\mathrm d}
\newcommand{\Pstar}{\mathsf P_{*}}
\newcommand{\PSK}{\mathcal P}
\newcommand{\Par}{\mathscr P}
\newcommand{\ent}{\operatorname{ent}}
\newcommand{\KL}[2]{\operatorname{KL}(#1\,\Vert\,#2)}
\newcommand{\cG}{\mathcal G}
\begin{document}
\frontmatter
\npptitlepage{The Parisi Formula for the\\ Edwards--Anderson Model\\ in the High-Dimensional Limit}{P.\ M.\ Aronow\\ Patrick Lopatto}
\begin{authorsnote}
With the exception of this note, the entire document is machine written. We
have taken care to supervise this writing to ensure that the proofs meet a
minimum standard of readability and that previous literature is appropriately
cited. However, the resulting document remains lacking in exposition and
overall coherence. We are nonetheless releasing this manuscript on Hexagon to
disseminate the result as quickly as possible, because we believe it may be of
interest to the mathematics and physics communities. We welcome suggestions
concerning mathematical corrections, omitted citations, and attribution of
ideas.
P.L.\ was partially supported by NSF grant DMS-2450004.
\end{authorsnote}
\begin{abstractpage}
We study the Edwards--Anderson spin glass on the torus $(\Z/L\Z)^d$, with independent Gaussian couplings of variance $1/(2d)$ between nearest neighbors. We prove that, as $d\to\infty$, its free energy converges to that of the Sherrington--Kirkpatrick model, given by the Parisi formula, at every temperature and in every uniform external field, uniformly in the side length $L$. The same holds on the hypercube $\{0,1\}^d$, and the ground-state energies converge to the Sherrington--Kirkpatrick ground-state energy. The upper bound holds on every regular graph of large degree. Its proof follows Mourrat's Hamilton--Jacobi approach for an enriched free energy, and it controls the negative part of the spectrum of the adjacency matrix with Panchenko's synchronization of overlaps, in a finitary form due to Mourrat. For the lower bound, we construct a vector of magnetizations by message passing, and we show that the pressure is at least its Thouless--Anderson--Palmer free energy using a sublinear relative-entropy bound for the associated planted model.
\end{abstractpage}
\tableofcontents
\mainmatter
% ---------- 01-intro.tex
\chapter{Introduction}\label{sec:intro}
\section{Background}\label{sec:intro-background}
The Edwards--Anderson (EA) model~\cite{EdwardsAnderson1975} is the basic model of a short-range spin glass. Ising spins $\sigma_x\in\{-1,1\}$ sit at the sites of a lattice, and each pair of neighboring spins interacts through an independent centered random coupling, so that some interactions favor alignment and others favor opposition. Its mean-field counterpart is the Sherrington--Kirkpatrick (SK) model~\cite{SherringtonKirkpatrick1975}, in which every pair of spins interacts. For the SK model, Parisi's theory of replica symmetry breaking~\cite{Parisi1979,Parisi1980Sequence} has to a large extent been made rigorous. The free energy is given by the Parisi formula~\cite{Guerra2003,Talagrand2006}, its minimizer is unique~\cite{AuffingerChen2015unique}, and Panchenko proved that overlap arrays satisfying the Ghirlanda--Guerra identities are ultrametric~\cite{Panchenko2013ultrametricity,PanchenkoBook}. For the EA model in a fixed dimension, by contrast, the existence of a spin-glass phase at positive temperature remains open. At zero temperature, Chatterjee has proved several signatures of glassy behavior, including disorder chaos and low-energy macroscopic excitations~\cite{Chatterjee2026EAGroundState}. There are competing physical descriptions of the low-temperature states: the replica-symmetry-breaking picture~\cite{mezard1987spin}, the droplet picture~\cite{FisherHuse1988}, and intermediate pictures such as the trivial--nontrivial picture~\cite{KrzakalaMartin2000,PalassiniYoung2000}; see~\cite{NewmanStein2013} for a discussion. The pictures disagree in particular about a uniform external field: replica symmetry breaking predicts a phase transition at a de Almeida--Thouless line~\cite{deAlmeidaThouless1978,GeorgesMezardYedidia1990}, whereas the droplet picture predicts that there is no transition in any positive field.
A natural way to relate the two models is to let the number of neighbors of each spin grow. On the torus $(\Z/L\Z)^d$ with nearest-neighbor couplings of variance $1/(2d)$, the local field $\sum_{y\sim x}J_{xy}\sigma_y$ acting on a spin, for a fixed configuration, is a sum of $2d$ small independent terms, as in the SK model. The variance of the Hamiltonian per site is $1/2$, which agrees with that of the SK model up to its finite-size correction. Mean-field theory is expected to describe the model increasingly well as $d$ grows. In zero field, the critical behavior is predicted to be that of mean-field theory above the upper critical dimension six~\cite{HarrisLubenskyChen1976}, and the low-temperature phase of the EA model on $\Z^d$ has been studied through an expansion in powers of $1/d$ around mean-field theory~\cite{GeorgesMezardYedidia1990}. The EA model on diluted hypercubes of large dimension has also been studied numerically as a mean-field spin glass~\cite{FernandezMartinMayorParisiSeoane2010}. In particular, the free energy per site of the EA model is expected to converge to that of the SK model as $d\to\infty$, at every temperature and in every field.
Rigorous results on this limit have so far required high temperature, interactions of diverging range, or graphs that are random or locally tree-like. For $\beta\le1$ in zero field, the SK free energy equals the annealed value $\log2+\beta^2/4$~\cite{AizenmanLebowitzRuelle1987}. The same limit holds on any regular graph of diverging degree. For $\beta<1$, this graph extension follows from the second-moment method and the Gaussian concentration of the free energy. On a $D$-regular graph with $n$ vertices, the ratio of the second moment of the partition function to the square of its mean is at most $\det(I-\beta^2S_+)^{-1/2}$, where $S_+$ is the positive part of the adjacency matrix divided by $D$. Since $\tr S_+\le n/\sqrt D$, this ratio is $e^{o(n)}$ as $D\to\infty$. The case $\beta=1$ follows by monotonicity in $\beta$. In every field, Hachem recently proved that for $\beta^2<\log2$ the free energy of the SK model with a sparse doubly stochastic variance profile, which includes the EA model on regular graphs of diverging degree, converges to the replica-symmetric SK value~\cite[Corollary~3]{Hachem2026sparse}. Fr\"ohlich and Zegarlinski~\cite{FrohlichZegarlinski1987SK} proved high-temperature mean-field limits of short-range spin glasses as the interaction range tends to infinity. In the Kac limit, the dimension is fixed, the strength of the couplings decays to zero over a distance $\gamma^{-1}$, and $\gamma\to0$ after the thermodynamic limit. For positive-semidefinite interaction kernels, Franz and Toninelli~\cite{FranzToninelli2004,FranzToninelli2004JPA} proved that the free energy converges to the SK free energy. One of the two inequalities, proved earlier by Guerra and Toninelli~\cite{GuerraToninelli2003Kac}, requires the interaction kernel to be positive semidefinite. For sparse random graphs, Dembo, Montanari, and Sen~\cite{DemboMontanariSen2017} showed that the maximum cut of random regular and Erd\H{o}s--R\'enyi graphs of large average degree is governed by the SK ground-state energy. Their comparison uses the randomness of the graph. It was extended to other optimization problems on sparse random hypergraphs by Sen~\cite{Sen2018hypergraphs}, and to inhomogeneous random graphs, whose large-degree limit is an inhomogeneous Potts spin glass, by Jagannath, Ko, and Sen~\cite{JagannathKoSen2018}. El Cheairi and Gamarnik showed that algorithms given by low-degree polynomials with a certain tree structure perform equally well on the SK model and on sparse Erd\H{o}s--R\'enyi graphs of large average degree~\cite{ElCheairiGamarnik2024}. Approximating message passing for the SK model by such polynomials, they obtained nearly maximal cuts on these graphs, assuming that the support of the Parisi measure is an interval $[0,q_*]$ for all large $\beta$. On deterministic regular graphs of large degree and large girth, El Alaoui, Montanari, and Sellke~\cite{ElAlaouiMontanariSellke2023local} constructed cuts of the same size by local algorithms, under a similar assumption at zero temperature.
% CHECK: Hachem2026sparse Corollary 3 verified against arXiv:2604.25535v1 (Assumptions 1-2 with t=beta^2, C_row=1).
None of these methods applies to a deterministic lattice at low temperature. The lattice is not locally tree-like: it has many short cycles. The methods that prove the Parisi formula, namely Guerra's interpolation~\cite{Guerra2003}, the Aizenman--Sims--Starr scheme~\cite{AizenmanSimsStarr2003}, and Panchenko's analysis of the overlap distribution~\cite{PanchenkoBook}, use the fact that the covariance of the SK Hamiltonian is a function of the overlap of two configurations. On the lattice, the covariance of $H(\sigma)$ and $H(\tau)$ is instead a function of the fraction of edges $xy$ with $\sigma_x\sigma_y=\tau_x\tau_y$. Further, the nearest-neighbor lattice with an even side length is bipartite, and flipping all spins on one sublattice reverses the sign of the Hamiltonian. Its normalized adjacency matrix then has the eigenvalue $-1$, and the comparison arguments that require this matrix to be close to positive semidefinite do not apply.
The goal of this paper is to prove that the free energy of the EA model on the torus $(\Z/L\Z)^d$ converges to the SK free energy as $d\to\infty$, at every temperature and in every uniform field, uniformly in the side length $L$. We prove the same statement for the hypercube $\{0,1\}^d$.
\section{Main results}\label{sec:intro-results}
\emph{The model.} Throughout, $G=(V,E)$ is a finite simple $D$-regular graph with $n=|V|$ vertices and $D\ge1$. It need not be connected. We write $\cG_D$ for the set of all such graphs. Given $G$, let $(g_{xy})_{xy\in E}$ be independent standard Gaussian variables and $J_{xy}=g_{xy}/\sqrt D$. For $\beta\ge0$ and $h\in\R$, the Hamiltonian, the pressure, and the ground-state energy per site are
\begin{equation}\label{eq:intro-model}
\begin{gathered}
H_G(\sigma)=\sum_{xy\in E}J_{xy}\sigma_x\sigma_y,\qquad
p_G(\beta,h)=\frac1n\E\log\sum_{\sigma\in\{-1,1\}^V}\exp\Big(\beta H_G(\sigma)+h\sum_{x\in V}\sigma_x\Big),\\
e(G;h)=\frac1n\E\max_{\sigma\in\{-1,1\}^V}\Big(H_G(\sigma)+h\sum_{x\in V}\sigma_x\Big).
\end{gathered}
\end{equation}
As in the literature on the SK model, the field enters the pressure through the Gibbs exponent and is not multiplied by $\beta$, so that $p_G(\beta,\beta h)/\beta\to e(G;h)$ as $\beta\to\infty$. The normalization gives $\E H_G(\sigma)^2=|E|/D=n/2$ for every $\sigma$, and $\sum_{y\sim x}\E J_{xy}^2=1$ for every vertex $x$. Our main examples are the following.
\begin{itemize}
\item The torus $\T^d_L=(\Z/L\Z)^d$, in which two vertices are adjacent if they differ by $\pm1$ in exactly one coordinate, is simple and $2d$-regular for $L\ge3$, with $n=L^d$. On it, $H_G$ is the EA Hamiltonian with couplings $g_{xy}/\sqrt{2d}$. We write $p_{d,L}=p_{\T^d_L}$.
\item The hypercube $Q_d=\{0,1\}^d$, in which two vertices are adjacent if they differ in exactly one coordinate, is $d$-regular with $n=2^d$.
\end{itemize}
No relation between $n$ and $D$ is assumed. For example, $D=2d$ is proportional to $\log n$ on $\T^d_3$, and $D=\log_2n$ on $Q_d$.
\emph{The SK model.} The SK Hamiltonian on $N$ spins is $H_N^{\mathrm{SK}}(\sigma)=N^{-1/2}\sum_{1\le i0$ at which $\PSK(\cdot,h)$ is differentiable, $\E\langle Q\rangle$ on $\T^d_L$ converges, uniformly in $L$, to the limit of $\E\langle R^2\rangle$ in the SK model. Since $\PSK(\cdot,h)$ is convex, this holds for all but countably many $\beta$. For the EA model in a fixed dimension, Contucci and Giardin\`a proved, in temperature average, Ghirlanda--Guerra identities for the link overlap~\cite{ContucciGiardina2007GG}, and Contucci, Mingione, and Starr proved distributional identities after suitable generic Gaussian perturbations whose covariances are powers of the link overlap~\cite{ContucciMingioneStarr2013}. By Panchenko's theorem~\cite{Panchenko2013ultrametricity}, these distributional Ghirlanda--Guerra identities yield ultrametricity of the limiting link-overlap arrays of the generically perturbed models.
\end{remark}
\begin{remark}\label{rem:intro-inputs}
The proof of the lower bound uses a recent theorem on the support of the Parisi measure in a positive field~\cite{Lopatto2026ExternalField}, which we recall in \cref{thm:pre-supporth}: for $h>0$ the support is an interval $[q_-,q_+]$ with $q_->0$. The case $h=0$ follows from the case $h>0$ by continuity in $h$. In zero field, Parisi predicted that the support is an interval $[0,q_+]$ for $\beta>1$ (full replica symmetry breaking); this was proved by Zhou for $\beta$ close to $1$~\cite{Zhou2026FRSB} and for every $\beta>1$ in~\cite{Lopatto2026SKFRSB}. At zero temperature, the Stieltjes measure associated with the canonical Parisi order parameter has infinitely many points in its support~\cite{AuffingerChenZeng2020}, and its support is $[0,1)$~\cite{Chen2026ZeroTempFRSB}. We do not use these results. The upper bound uses no information on the support of the Parisi measure.
\end{remark}
The upper bound in \cref{thm:main,thm:hypercube} holds for every regular graph of large degree: \cref{prop:upper} states that
\begin{equation}\label{eq:intro-upper}
\limsup_{D\to\infty}\sup_{G\in\cG_D}p_G(\beta,h)\le\PSK(\beta,h)
\end{equation}
for every $\beta\ge0$ and $h\in\R$. The supremum is over all finite simple $D$-regular graphs, with no relation between $n$ and $D$ and no assumption of transitivity or of a spectral condition. The bound is attained, for example, by the complete graphs $K_{D+1}$. The lower bound uses the structure of tori and hypercubes, through a homogenization statement for the output of a message-passing algorithm (\cref{sec:amp}). We do not know whether the lower bound holds for every sequence of regular graphs of diverging degree.
\begin{remark}\label{rem:intro-bipartite}
The complete bipartite graph $K_{D,D}$ is $D$-regular, and the model on it is the bipartite SK model with two species of $D$ spins each. For it, the upper bound \eqref{eq:intro-upper} is attained in every field: for every $\beta\ge0$ and $h\in\R$,
\[
\lim_{D\to\infty}p_{K_{D,D}}(\beta,h)=\PSK(\beta,h).
\]
The lower bound follows from a Gaussian interpolation with the SK model on $2D$ spins, since the normalized adjacency matrix of $K_{D,D}$ is at most the projection onto the constant vectors; we give the proof in \cref{sec:geo-main}. In the limit $D\to\infty$, this lower bound is a special case of the lower bound of Bates and Sohn for balanced multispecies models~\cite[Theorem~1.3]{BatesSohn2025balanced}, which holds in every field~\cite[Remark~1.8]{BatesSohn2025balanced}. For the bipartite SK model in zero field, a Gaussian interpolation with independent SK models on the two species, which gives a lower bound of this kind, was used by Barra, Genovese, and Guerra~\cite[Section~3.1]{BarraGenoveseGuerra2011}, and for models with exchangeable species this comparison was proved by Issa~\cite[Theorem~38]{Issa2024permutation}. In zero field, the limit of the free energy of balanced multispecies models, which include this one, was identified in~\cite{ChenIssaMourrat2026,Ho2026}; see \cref{sec:intro-related}.
% CHECK: verified only in arXiv versions: BatesSohn2025balanced Theorem 1.3 and Remark 1.8 (arXiv:2507.06522v2);
% Issa2024permutation Theorem 38 checked in ALEA 23 (2026), p. 794.
\end{remark}
To our knowledge, for interactions of fixed range, no deterministic family of lattices was previously known on which the free energy of the EA model at low temperature, or its ground-state energy, converges to the SK value. Our results concern the limit in which the dimension tends to infinity, taken after or jointly with the thermodynamic limit; they do not decide between the competing pictures of the EA model in a fixed dimension.
\section{Ideas of the proofs}\label{sec:intro-ideas}
The upper and lower bounds use different methods. We describe them in turn.
\emph{The upper bound.} The interpolation identity recalled in \cref{sec:intro-related} gives the upper bound with an error of at most $\beta^2\epsilon/4$ when $\lambda_{\min}(S)\ge-\epsilon$, where $S=A_G/D$. On bipartite graphs, $S$ has the eigenvalue $-1$, so this error does not vanish. We follow instead the Hamilton--Jacobi approach to the Parisi formula developed by Mourrat and coauthors~\cite{Mourrat2022Wasserstein,MourratPanchenko2020extending,DominguezMourrat2024}, and in particular the strategy of Mourrat's upper bounds for models with nonconvex interactions~\cite{Mourrat2021nonconvex,Mourrat2020vector}. There, an enriched free energy is shown to be an approximate viscosity supersolution of a Hamilton--Jacobi equation, with Ghirlanda--Guerra identities and synchronization enforced at the points where a test function touches it. Fix a finite Ruelle cascade. We give each vertex $x$ a Gaussian field indexed by the leaves of the cascade, whose covariance is determined by a nondecreasing path $q_x$, and we let the couplings be independent across the first-level branches of the cascade. Let $F(t,q)$ be minus the free energy of the resulting model at inverse temperature $\beta\sqrt t$, plus $\beta^2t/4$. At $q=0$ and $t=1$, it is at most $\beta^2/4-p_G(\beta,h)$, and at $t=0$ it is an explicit function of the paths. Its time derivative equals a quadratic Hamiltonian, evaluated at the gradient of $F$ in $q$, plus a term of the form $\beta^2(4n)^{-1}\sum_jw_j\tr(SC_j)$, where $C_j$ is the covariance matrix of $\sigma^1\odot\sigma^2$ given that the two replicas meet at depth $j$ of the cascade. For the SK model the analogous term is nonnegative, and Guerra's bound follows. Here $S$ has negative eigenvalues, and the term has no sign.
We show instead that it is small. For every $\eta>0$, we have $|\tr(SC_j)|\le\eta\tr C_j+(4\eta)^{-1}\tr(S^2C_j)$, with $\tr C_j\le n$. Thus it suffices to control $\tr(S^2C_j)$, a sum of conditional variances of the row overlaps $(S(\sigma^1\odot\sigma^2))_x$, $x\in V$. A sparsification theorem of Cohen, Nelson, and Woodruff~\cite{CohenNelsonWoodruff2016} reduces this sum to a set of $O(n/D)$ rows. For these rows, small Gaussian perturbations of the Hamiltonian enforce approximate Ghirlanda--Guerra identities for the pair formed by the overlap in the cascade and the row overlap. A finitary form of Panchenko's synchronization theorem~\cite{Panchenko2015multispecies}, due to Mourrat~\cite{Mourrat2021nonconvex}, then bounds the conditional variance of each of these row overlaps, given the overlap in the cascade, by a multiple of the mesh of the cascade. The perturbations change the free energy by $O(\mathsf s^2/D)$ for a perturbation of strength $\mathsf s$, and the identities require $\mathsf s\to\infty$; this is where we use that $D\to\infty$. The independence of the couplings across first-level branches makes the fluctuations of the free energy of order $1/n$. This allows us to enforce the identities at every point where a test function touches a deterministic regularization of $F$, and we find that this regularization is, up to small errors, a viscosity supersolution of a Hamilton--Jacobi equation on the cone of $V$-indexed nondecreasing paths, with the quadratic Hamiltonian $\beta^2(4n)^{-1}\sum_jw_j\,p_j\cdot Sp_j$. A comparison principle bounds it below by a subsolution built from the Hopf--Lax formula for the SK model. That this function is a subsolution of the equation with the indefinite matrix $S$ follows from the inequality between the arithmetic and the harmonic mean, which uses only that $S$ has nonnegative entries and unit row sums. Its initial value lies below that of $F$ because the free energy of a finite cascade is convex in the square roots of its variances, which we derive in \cref{sec:up-cascades} from complete monotonicity. Finally, the value of the Hopf--Lax formula at time one is given by the Parisi formula, as observed by Mourrat for $h=0$~\cite{Mourrat2022Wasserstein}. A comparison of this kind with a one-species Hamilton--Jacobi equation was used by Chen, Issa, and Mourrat~\cite[Appendix~B]{ChenIssaMourrat2026} for balanced multispecies models in zero field; see also~\cite[Section~7]{Issa2024permutation}. There the initial condition is convex in the paths themselves~\cite{ChenIssaMourrat2026,Ho2026}. This convexity fails in a strong field~\cite[Section~6.1]{Mourrat2021nonconvex}, whereas the convexity in the square roots holds in every field.
% CHECK: ChenIssaMourrat2026 Appendix B verified in arXiv:2606.16636v2; Mourrat2021nonconvex Section 6.1 in arXiv:2004.01679v8.
\emph{The lower bound.} For $m\in[-1,1]^V$ and $\kappa_x=1-m_x^2$, let
\[
\Theta_G(m)=\beta H_G(m)+h\sum_{x\in V}m_x+\sum_{x\in V}\ent(m_x)+\frac{\beta^2}4\langle\kappa,S\kappa\rangle,
\]
where $S=A_G/D$. This is the Thouless--Anderson--Palmer (TAP) free energy~\cite{ThoulessAndersonPalmer1977} of the graph, that is, the expansion to second order in the couplings of the free energy at fixed magnetizations~\cite{Plefka1982,GeorgesYedidia1991expand}, with each squared coupling replaced by its mean $1/D$. Jensen's inequality for the product measure on $\{-1,1\}^V$ with means $m$ gives $\log Z_G\ge\Theta_G(m)-(\beta^2/4)\langle\kappa,S\kappa\rangle$ for every $m$, where $Z_G$ is the partition function. The last term of $\Theta_G(m)$, the Onsager correction, is not visible to this argument, and it is of the same order as the other terms. The proof of the lower bound has three steps.
First, we construct $m$ by approximate message passing (AMP), following the algorithms of Montanari~\cite{Montanari2021} and Sellke~\cite{Sellke2021field}, whose nonlinearities are given by the Parisi partial differential equation (PDE). Assuming that the support of the Parisi measure is an interval $[0,q_*]$ for all large $\beta$, Montanari also showed that at low temperature his algorithm constructs approximate solutions of the TAP equations. The output $m$ is computed from finitely many products of the coupling matrix $W$ with vectors, each a function of the earlier products. On tori and hypercubes, Hachem's state evolution for sparse matrices~\cite{Hachem2024} shows that $n^{-1}\E\,\Theta_G(m)$ is close to the right side of an identity for $\PSK(\beta,h)$ in terms of the Parisi martingale (\cref{lem:pre-parisiTAP}), and that the empirical law of the coordinates of $m$ is close to the law $\mu_+$ of $\tanh X_{q_+}$, where $X$ is the Parisi diffusion and $q_+$ is the largest point of the support of the Parisi measure. Only this step uses the theorem on the support of the Parisi measure, and only for $h>0$; the case $h=0$ follows by continuity in $h$.
Second, we condition on the products computed by the algorithm. Conditioning on the iterates of message passing goes back to Bolthausen~\cite{Bolthausen2014}, and it was used to compute the free energy of the SK model at high temperature in~\cite{Bolthausen2018Morita,BrenneckeYau2022}. Given these products, the vector of couplings is the sum of its conditional mean and a Gaussian vector projected onto a subspace of bounded codimension per vertex. We expand the partition function around $m$, writing $\sigma=m+v$. The unobserved part of the disorder enters through the inner product of a Gaussian vector with the edge vector $a(v)_{xy}=D^{-1/2}v_xv_y$. Averaging the exponential over the Gaussian vector produces a variance term from which the Onsager correction is extracted, whereas the logarithm of this average differs from the average of the logarithm by a relative entropy. This relative entropy is controlled by $\mathcal D(m)$, the relative entropy of the law of $\beta a(v^*)+Z$, where $v^*=\sigma^*-m$ for a random configuration $\sigma^*$ with independent coordinates of means $m$, with respect to the law of the noise $Z$. Using Talagrand's transport inequality for the Gaussian measure~\cite{Talagrand1996transport}, we show that the loss in the pressure is at most $4\beta(n^{-1}\E\,\mathcal D(m))^{1/2}$ (\cref{thm:band}). Thus the lower bound reduces to the sublinear relative-entropy estimate $\E\,\mathcal D(m)=o(n)$ for the planted edge model. In~\cite{Bolthausen2018Morita,BrenneckeYau2022}, the corresponding error is bounded by a conditional second-moment computation at high temperature; the planted model and the transport inequality take its place here. For spherical models, Huang and Sellke~\cite{HuangSellke2023constructive} proved the lower bound in the Parisi formula by constructing an ultrametric tree of pure states, each with approximately the same free energy as the whole model. They build the tree with an optimization algorithm and a truncated second-moment argument.
Third, we prove this relative-entropy estimate (\cref{thm:nd}). We interpolate between the edge observations and independent scalar Gaussian observations of the coordinates of $v^*$, and we compare the relative entropy along the interpolation with a Hopf--Lax formula, as in the Hamilton--Jacobi approach to the inference of low-rank matrices~\cite{Mourrat2021HJ,Mourrat2020matrix,DominguezMourrat2024}. The Hopf--Lax formula involves the relative entropy $\psi_\mu(r)$ of a scalar Gaussian channel with signal-to-noise ratio $r$, averaged over the law $\mu$ of the coordinates of $m$, and it shows that $n^{-1}\mathcal D(m)$ is asymptotically at most $\sup_{0\le s\le1}[\psi_\mu(\beta^2s)-\beta^2s^2/4]$. The comparison requires the empirical averages of $\psi_{m_x}$ to be close to $\psi_\mu$, uniformly over a set of signal-to-noise profiles; on tori and hypercubes, this follows from the finite range of the algorithm and the fact that all but an exponentially small fraction of the eigenvalues of $S$ are close to zero. It uses that $S\preceq I$, but not that $S$ is positive semidefinite. Finally, for $\mu=\mu_+$, the supremum vanishes. For $0\le s\le1-q_+$, the Doob-transform representation of the Parisi diffusion above $q_+$ identifies $\psi_{\mu_+}(\beta^2s)-\beta^2s^2/4$ with a difference of values of the first variation of the Parisi functional at its minimizer, computed by Jagannath and Tobasco~\cite{JagannathTobasco2017} and by Chen~\cite{Chen2017Variational}. The optimality of the Parisi measure shows that this difference is nonpositive. For $s\ge1-q_+$, the bound $\Gamma_{\mu_+}'(s)\le\beta^2(1-q_+-s)\le0$ extends the inequality to the remaining range (\cref{prop:nd-scalar}). For the SK model, the function $s\mapsto2\psi_\mu(\beta^2s)-\beta^2s^2/2$ is the function $\Gamma_\mu$ introduced by Chen, Panchenko, and Subag~\cite{CPSGeneralizedTAP}. With $q=\int m^2\mu(\mathrm dm)<1$, they showed that it is nonpositive on $[0,1-q]$ if and only if the variational formula for their generalized TAP correction at the law of $|m|$ is minimized by the point mass at zero, and that the correction is then the classical one~\cite[Proposition~13]{CPSGeneralizedTAP}.
The lower bound obtained in this way holds for tori of a fixed side length and for hypercubes. In \cref{sec:proofs}, a Gaussian interpolation that adds or removes edges shows that the torus pressure changes by at most $\beta^2/(4L)$ when the side length is changed from $L$ to infinity, uniformly in $d$, which gives the uniformity in $L$. This interpolation was used by Contucci and Graffi~\cite{ContucciGraffi2004monotonicity,ContucciGraffi2004surface} to study the thermodynamic limit and the surface pressure of the EA model.
\section{Related work}\label{sec:intro-related}
\emph{Mean-field comparisons.} Gaussian interpolation and comparison arguments are classical~\cite{GuerraToninelli2002Thermodynamic,Guerra2003}. For the SK model, they give the identity
\[
p^{\mathrm{SK}}_n(\beta,h)-p_G(\beta,h)=\frac{\beta^2}4\int_0^1\E\big\langle Q-R^2\big\rangle_\theta\dd\theta
\]
along the interpolation $\sqrt{1-\theta}\,H_G+\sqrt\theta\,H_n^{\mathrm{SK}}$ between a graph Hamiltonian and an independent SK Hamiltonian on the same vertex set. We have $Q-R^2=n^{-1}\langle u-R\mathbf 1,S(u-R\mathbf 1)\rangle$. If $\lambda_{\min}(S)\ge-\epsilon$, then $Q-R^2\ge-\epsilon(1-R^2)\ge-\epsilon$, and the identity gives $p_G(\beta,h)\le p^{\mathrm{SK}}_n(\beta,h)+\beta^2\epsilon/4$. For every $G\in\cG_D$, the matrix $S$ is nonzero with zero trace, so it cannot be positive semidefinite. Positive-semidefinite interaction kernels give the upper bound without this error for Kac models, as in Guerra and Toninelli~\cite{GuerraToninelli2003Kac}. If the eigenvalues of $S$ on the orthogonal complement of $\mathbf 1$ tend to zero, then $Q-R^2$ tends to zero uniformly, and the identity gives both bounds. Neither condition holds on tori and hypercubes. Their matrices $S$ have eigenvalues close to $1$ on this complement, and on bipartite graphs $S$ has the eigenvalue $-1$ and $Q-R^2$ takes negative values. \Cref{prop:upper} has no spectral hypothesis.
\emph{Kac limits and mean-field approximations.} High-temperature results for diverging-range mean-field limits were obtained earlier by Fr\"ohlich and Zegarlinski~\cite{FrohlichZegarlinski1987SK}. In the Kac limit, the lower bound of Franz and Toninelli~\cite{FranzToninelli2004,FranzToninelli2004JPA} interpolates between the Kac model and independent SK models on boxes whose side is large compared with the lattice spacing and small compared with the range of the interaction. It uses that each spin interacts with many spins through a kernel that varies slowly on the scale of the boxes. For nearest-neighbor interactions there are no such boxes, and our lower bound uses message passing instead. For Ising and Potts models with an interaction matrix $A$ whose entries may have either sign, Basak and Mukherjee proved, under a boundedness condition on $A$, that the mean-field approximation of the free energy is asymptotically exact when $\tr(A^2)=o(n)$~\cite[Theorem~1.1]{BasakMukherjee2017universality}. In particular, on regular graphs of diverging degree $D$, the free energy of the ferromagnetic Ising model with couplings $1/D$ converges to that of the Curie--Weiss model~\cite[Theorem~2.1]{BasakMukherjee2017universality}. For the EA model, $\tr((\beta W)^2)$ is of order $n$, and the mean-field approximation misses the Onsager correction.
% CHECK: BasakMukherjee2017universality Theorems 1.1 and 2.1 checked in the published paper.
\emph{The Parisi formula.} The upper bound in the Parisi formula was proved by Guerra~\cite{Guerra2003} and the lower bound by Talagrand~\cite{Talagrand2006}. Panchenko proved ultrametricity from the Ghirlanda--Guerra identities~\cite{GhirlandaGuerra1998,Panchenko2013ultrametricity,PanchenkoBook}. For multispecies models with a positive-definite interaction matrix, Barra, Contucci, Mingione, and Tantari~\cite{BarraContucciMingioneTantari2015} proved the upper bound in the Parisi formula, and Panchenko proved the lower bound by showing that the overlaps of different species are synchronized~\cite{Panchenko2015multispecies}. Synchronization was extended to Potts and vector spins in~\cite{Panchenko2018Potts,Panchenko2018vector}. Connections between the free energy of spin glasses and Hamilton--Jacobi equations were pointed out by Guerra and coauthors; see~\cite{BarraDiBiasioGuerra2010}. Mourrat and coauthors developed an approach to the Parisi formula and its extensions through Hamilton--Jacobi equations~\cite{Mourrat2021HJ,Mourrat2022Wasserstein,MourratPanchenko2020extending,Mourrat2020vector,DominguezMourrat2024}, and Chen and Xia studied these equations on convex cones~\cite{ChenXia2022cones}. Our upper bound combines these tools. The rows of $S$ play the role of species, but their number is of order $n/D$ even after sparsification, so we do not use a multispecies formula; each row overlap is synchronized separately with the overlap in a Ruelle cascade. The lower bound uses the variational representation of the Parisi functional of Auffinger and Chen~\cite{AuffingerChen2015unique} and the characterization of the Parisi measure through the first variation of the Parisi functional, obtained by Jagannath and Tobasco~\cite{JagannathTobasco2017} and independently by Chen~\cite{Chen2017Variational}.
\emph{Multispecies models with a nonconvex covariance.} For such models, which include the bipartite SK model~\cite{BarraGenoveseGuerra2011}, the Hamilton--Jacobi approach gives an upper bound on the free energy~\cite{Mourrat2021nonconvex,Mourrat2020vector}. Chen and Mourrat showed that, if the limit of the free energy of a vector spin glass with nonconvex interactions exists, it is a critical value of a functional of Parisi type~\cite{ChenMourrat2025nonconvex}, and Chen extended this to multispecies models~\cite{Chen2024multispecies}. For the bipartite spherical SK model, the limit of the free energy was computed by Auffinger and Chen at high temperature~\cite{AuffingerChen2014bipartite} and by Baik and Lee in zero field, at every noncritical temperature~\cite{BaikLee2017bipartite}; the critical value follows by continuity. Subag computed it for pure multispecies spherical models through a TAP representation, assuming that certain free energies converge~\cite{Subag2021multispeciesII}. For Ising spins, in zero field the limit of the free energy was recently identified by Chen, Issa, and Mourrat~\cite{ChenIssaMourrat2026}. For balanced models, such as the bipartite SK model with species of equal sizes, this limit is given by a one-species formula. The corresponding lower bound is due to Bates and Sohn~\cite{BatesSohn2025balanced}, and for models with exchangeable species to Issa~\cite{Issa2024permutation}. The matching upper bound was proved in~\cite{ChenIssaMourrat2026,Ho2026}, and the lower bound of Bates and Sohn holds in every field~\cite[Remark~1.8]{BatesSohn2025balanced}. The model on the complete bipartite graph $K_{D,D}$ is a bipartite SK model with two species of equal size. \Cref{prop:upper} gives the upper bound for it in every field, and together with the interpolation identity above it gives the limit of its free energy in every field (\cref{rem:intro-bipartite}).
\emph{Message passing and the TAP free energy.} AMP was introduced by Donoho, Maleki, and Montanari~\cite{DonohoMalekiMontanari2009}. For the SK model, an iteration with Gaussian state evolution was introduced by Bolthausen~\cite{Bolthausen2014}, and general AMP algorithms were analyzed in~\cite{BayatiMontanari2011,JavanmardMontanari2013,BayatiLelargeMontanari2015}. Hachem extended polynomial AMP to sparse variance profiles~\cite{Hachem2024}. Montanari~\cite{Montanari2021} showed that AMP with nonlinearities given by the Parisi PDE approximately maximizes the SK Hamiltonian if, for all large $\beta$, the support of the Parisi measure is an interval $[0,q_*]$, and Sellke~\cite{Sellke2021field} extended the algorithm to nonzero external fields under a corresponding no-overlap-gap assumption. El Alaoui, Montanari, and Sellke~\cite{ElAlaouiMontanariSellke2021} extended these algorithms to mixed $p$-spin models and studied the energy that they reach through a stochastic control problem; we use this formulation in \cref{sec:amp}. El Alaoui, Montanari, and Sellke also combined AMP with stochastic localization to sample from the SK Gibbs measure at high temperature~\cite{ElAlaouiMontanariSellke2022sampling}. Fan, Li, and Sen~\cite{FanLiSen2022} proved the TAP equations for spin glasses with orthogonally invariant couplings at high temperature, through a conditional second-moment analysis of the free energy restricted to a thin band around the output of AMP. The functional $\beta H(m)+h\sum_xm_x+\sum_x\ent(m_x)+(\beta^2/4)\sum_{x,y}S_{xy}(1-m_x^2)(1-m_y^2)$, whose value at the output of our algorithm we bound below, is the TAP free energy of the graph. For mixed $p$-spin models with Ising spins, Chen and Panchenko proved that the free energy is asymptotically the maximum of the TAP free energy over magnetizations whose self-overlap lies to the right of the support of the Parisi measure~\cite[Theorem~1]{ChenPanchenko2018TAP}, and this representation was extended by Chen, Panchenko, and Subag~\cite{CPSGeneralizedTAP,CPSII}; see also~\cite{AuffingerJagannath2019TAP,FanMeiMontanari2021TAP}. For spherical models, a TAP representation of the free energy was proved by Subag~\cite{Subag2018FreeEnergyLandscapes}.
% CHECK: ChenPanchenko2018TAP Theorem 1 verified against arXiv:1709.03468v2 only.
\section{Organization}\label{sec:intro-organization}
\Cref{sec:prelim} collects notation and the facts about the SK model and the Parisi formula that we use. \Cref{sec:up-cascades,sec:up-upper} prove the upper bound \eqref{eq:intro-upper} for all regular graphs. \Cref{sec:amp,sec:band,sec:nondetection} prove the lower bound for tori of fixed side length and for hypercubes: \cref{sec:amp} constructs the centers by message passing, \cref{sec:band} reduces the lower bound to a sublinear relative-entropy bound for a planted model, and \cref{sec:nondetection} proves this statement. \Cref{sec:proofs} removes the restriction to a fixed side length and proves \cref{thm:main,thm:hypercube,cor:ground}.
% ---------- 02-prelim.tex
\chapter{Preliminaries}\label{sec:prelim}
This section fixes notation and collects the inputs used in the rest of the paper. \Cref{sec:pre-model} defines the model and records its elementary properties. \Cref{sec:pre-tools} lists classical facts about Gaussian vectors. \Cref{sec:pre-SK} recalls the Sherrington--Kirkpatrick model, the Parisi formula, and the properties of the Parisi PDE that we use. It ends with \cref{lem:pre-parisiTAP}, which writes $\PSK(\beta,h)$ in terms of the Parisi martingale; this identity is the target of the lower bound.
\section{The model}\label{sec:pre-model}
For $D\ge1$ let $\cG_D$ be the set of finite simple $D$-regular graphs $G=(V,E)$, with any number $n=|V|$ of vertices, connected or not. Vertices are denoted by $x,y,z$ and edges by $e$ or $xy$. For $G\in\cG_D$ we write $S=A_G/D$ for the normalized adjacency matrix. It is symmetric, has nonnegative entries and zero diagonal, and satisfies
\begin{equation}\label{eq:pre-S}
S\mathbf 1=\mathbf 1,\qquad \|S\|\le1,\qquad \tr S^2=\frac nD,\qquad \max_{x,y\in V}S_{xy}=\frac1D,
\end{equation}
where $\|\cdot\|$ is the operator norm. The bound on the norm follows from Jensen's inequality, since $\sum_x(\sum_yS_{xy}w_y)^2\le\sum_{x,y}S_{xy}w_y^2=\sum_yw_y^2$, and the trace identity follows by summing the squares of the entries.
Let $(g_e)_{e\in E}$ be independent standard Gaussian variables. The couplings are $J_{xy}=g_{xy}/\sqrt D$, and the coupling matrix $W$ is the symmetric $V\times V$ matrix with $W_{xy}=J_{xy}$ if $xy\in E$ and $W_{xy}=0$ otherwise; in particular, $W_{xx}=0$. We evaluate the Hamiltonian also at real vectors. The Hamiltonian, the pressure at inverse temperature $\beta\ge0$ and field $h\in\R$, and the ground-state energy are
\begin{equation}\label{eq:pre-model}
\begin{gathered}
H_G(w)=\sum_{xy\in E}J_{xy}w_xw_y=\frac12\langle w,Ww\rangle\qquad(w\in\R^V),\\
p_G(\beta,h)=\frac1n\E\log\sum_{\sigma\in\{-1,1\}^V}\exp\Big(\beta H_G(\sigma)+h\sum_{x\in V}\sigma_x\Big),\\
e(G;h)=\frac1n\E\max_{\sigma\in\{-1,1\}^V}\Big(H_G(\sigma)+h\sum_{x\in V}\sigma_x\Big).
\end{gathered}
\end{equation}
The field is not multiplied by $\beta$. For all $\sigma$, we have $\E H_G(\sigma)^2=|E|/D=n/2$.
The torus $\T^d_L=(\Z/L\Z)^d$ with $L\ge3$, in which two vertices are adjacent if they differ by $\pm1$ in exactly one coordinate, belongs to $\cG_{2d}$ and has $n=L^d$ vertices. We write $p_{d,L}$ for its pressure. The hypercube $Q_d=\{0,1\}^d$, in which two vertices are adjacent if they differ in exactly one coordinate, belongs to $\cG_d$ and has $n=2^d$ vertices.
For fixed disorder, $\langle\cdot\rangle$ denotes the average under the Gibbs measure, which gives $\sigma$ weight proportional to $\exp(\beta H_G(\sigma)+h\sum_x\sigma_x)$. We use it for one or several independent replicas $\sigma^1,\sigma^2,\dots$ under the same disorder. We write $u=\sigma^1\odot\sigma^2$ for the coordinatewise product of two replicas, and
\[
R=\frac1n\sum_{x\in V}u_x,\qquad Q=\frac1{|E|}\sum_{xy\in E}u_xu_y=\frac1n\langle u,Su\rangle
\]
for the site overlap and the link overlap. The second expression for $Q$ uses $|E|=nD/2$.
The following notation is used in the lower bound. For $m\in[-1,1]$ we write
\[
\ent(m)=-\frac{1+m}2\log\frac{1+m}2-\frac{1-m}2\log\frac{1-m}2\in[0,\log2],
\]
with $0\log0=0$, for the entropy of a $\{-1,1\}$-valued random variable with mean $m$. For $m\in[-1,1]^V$ we write $\pi_m$ for the product law on $\{-1,1\}^V$ with means $m$, and $\kappa_x=1-m_x^2$ for its variances. If $m_x=\tanh y$, then
\begin{equation}\label{eq:pre-ent-tanh}
\ent(m_x)=\log2\cosh y-y\tanh y.
\end{equation}
\begin{lemma}\label{lem:pre-basic}
Let $G\in\cG_D$.
\begin{enumerate}[label=(\alph*)]
\item The function $(\beta,h)\mapsto p_G(\beta,h)$ is convex on $\R^2$ and even in $h$, and $p_G(0,h)=\log2\cosh h$.
\item For $\beta\ge0$ and $h\in\R$,
\[
\partial_\beta p_G(\beta,h)=\frac\beta2\big(1-\E\langle Q\rangle\big)\in\Big[0,\frac\beta2\Big],\qquad
\partial_hp_G(\beta,h)=\frac1n\sum_{x\in V}\E\langle\sigma_x\rangle\in[-1,1].
\]
\item For $\beta\ge0$ and $h\in\R$, we have $\log2\cosh h\le p_G(\beta,h)\le\log2\cosh h+\beta^2/4$.
\item For $B\ge0$, $\beta,\beta'\in[0,B]$, and $h,h'\in\R$, we have $|p_G(\beta,h)-p_G(\beta',h')|\le B|\beta-\beta'|/2+|h-h'|$.
\end{enumerate}
\end{lemma}
\begin{proof}
For fixed disorder, the function $(\beta,h)\mapsto\log\sum_\sigma\exp(\beta H_G(\sigma)+h\sum_x\sigma_x)$ is convex, since it is a log-sum-exp of affine functions, and its partial derivatives are $\langle H_G\rangle$ and $\sum_x\langle\sigma_x\rangle$. Since these are bounded by $\max_\sigma|H_G(\sigma)|$ and $n$, which are integrable, we may differentiate under the expectation. The substitution $\sigma\mapsto-\sigma$ shows that $p_G$ is even in $h$, and at $\beta=0$ the sum factorizes and equals $(2\cosh h)^n$. This proves (a) and the formula for $\partial_hp_G$.
For the $\beta$-derivative, Gaussian integration by parts (\ref{F:ibp} below) in $g_{xy}$ gives
\[
\E\big[g_{xy}\langle\sigma_x\sigma_y\rangle\big]=\E\,\partial_{g_{xy}}\langle\sigma_x\sigma_y\rangle=\frac\beta{\sqrt D}\E\big(1-\langle\sigma_x\sigma_y\rangle^2\big).
\]
Since $\langle\sigma_x\sigma_y\rangle^2=\langle u_xu_y\rangle$ for two replicas, summing over the $|E|=nD/2$ edges, we obtain
\[
\partial_\beta p_G=\frac1n\sum_{xy\in E}\frac1{\sqrt D}\E\big[g_{xy}\langle\sigma_x\sigma_y\rangle\big]=\frac{\beta}{nD}\sum_{xy\in E}\big(1-\E\langle u_xu_y\rangle\big)=\frac\beta2\big(1-\E\langle Q\rangle\big).
\]
Since $\langle Q\rangle=|E|^{-1}\sum_{xy\in E}\langle\sigma_x\sigma_y\rangle^2\in[0,1]$, this proves (b). Part (c) follows by integrating the bounds of (b) in $\beta$ from $0$ and using (a), and (d) follows from (b).
\end{proof}
\section{Classical tools}\label{sec:pre-tools}
For a random variable $Y$ we write $\|Y\|_p=(\E|Y|^p)^{1/p}$. A function on $\R^k$ has \emph{polynomial growth} if it is bounded by $C(1+|w|^p)$ for some $C,p$. We use the following classical facts.
\begin{enumerate}[label=(F\arabic*),leftmargin=*]
\item\label{F:ibp} \emph{Gaussian integration by parts.} Let $(U_1,\dots,U_k)$ be a centered Gaussian vector, with degenerate covariance allowed, and let $\Psi\colon\R^k\to\R$ be $C^1$, with $\Psi$ and $\nabla\Psi$ of polynomial growth. Then for all $j\le k$,
\[
\E\big[U_j\Psi(U_1,\dots,U_k)\big]=\sum_{\ell=1}^k\Cov(U_j,U_\ell)\,\E\,\partial_\ell\Psi(U_1,\dots,U_k).
\]
For one standard Gaussian variable this is Stein's identity~\cite{Stein1981}, and the general case follows by writing $U_\ell=a_\ell U_j+V_\ell$ with $a_\ell=\Cov(U_\ell,U_j)/\Var U_j$ and $(V_\ell)$ independent of $U_j$, and applying Stein's identity conditionally on $(V_\ell)$; if $\Var U_j=0$, both sides vanish.
\item\label{F:moments} \emph{Method of moments.} A Gaussian law on $\R^k$, degenerate or not, is determined by its moments. If random vectors $Z_\ell$ satisfy $\E\psi(Z_\ell)\to\E\psi(Z)$ for every polynomial $\psi$, with $Z$ Gaussian, then $Z_\ell\to Z$ in law. In dimension one this is~\cite[Theorems~30.1 and~30.2]{Billingsley1995}, and the general case follows by the Cram\'er--Wold device~\cite[Theorem~29.4]{Billingsley1995}.
\item\label{F:density} \emph{Polynomial density.} For every Gaussian measure $\gamma$ on $\R^k$, including a degenerate one, and every $1\le p<\infty$, polynomials are dense in $L^p(\gamma)$. It suffices to work on the affine support of $\gamma$, where $\gamma$ is nondegenerate. If density failed there, duality would give a nonzero $f\in L^{p'}(\gamma)$, with $1/p+1/p'=1$, such that $\int fP\dd\gamma=0$ for every polynomial $P$. Since $\int|f(w)|e^{c|w|}\gamma(\mathrm dw)<\infty$ for all $c>0$ by H\"older's inequality and the Gaussian exponential moments, we may integrate the exponential series term by term, and we obtain $\int f(w)e^{i\langle t,w\rangle}\gamma(\mathrm dw)=0$ for every $t$. Since $\int f\dd\gamma=0$, the positive and negative parts of $f\gamma$ have equal mass and equal Fourier transforms. If this mass is positive, normalize the two parts and apply Fourier uniqueness for probability measures~\cite[Section~29]{Billingsley1995}; if it is zero, both parts vanish. In either case $f\gamma=0$, a contradiction.
\item\label{F:concentration} \emph{Gaussian concentration.} Let $\Psi\colon\R^k\to\R$ be $L$-Lipschitz for the Euclidean norm and let $g$ be a standard Gaussian vector in $\R^k$. Then $\Var\Psi(g)\le L^2$, and $\P(|\Psi(g)-\E\Psi(g)|\ge t)\le2e^{-t^2/(2L^2)}$ for every $t\ge0$; see~\cite[Chapter~3 and Section~5.4]{BoucheronLugosiMassart2013}. In particular, $\|\Psi(g)-\E\Psi(g)\|_p\le C_pL$ for every $p\ge1$, where $C_p$ depends only on $p$.
\item\label{F:transport} \emph{Transport inequality.} Let $\gamma_k$ be the standard Gaussian measure on $\R^k$. For every probability measure $\rho$ on $\R^k$, the quadratic Wasserstein distance satisfies $W_2(\rho,\gamma_k)^2\le2\KL{\rho}{\gamma_k}$~\cite{Talagrand1996transport}.
\item\label{F:gaussian-max} \emph{Maximum of the Hamiltonian.} Let $G\in\cG_D$. Since $H_G$ has no diagonal terms, it is affine in each coordinate separately, and $\sup_{w\in[-1,1]^V}|H_G(w)|=\max_{\sigma}|H_G(\sigma)|$. For every $p\ge1$ there exists $C_p$, depending only on $p$, such that $\E\max_\sigma|H_G(\sigma)|^p\le C_pn^p$. Indeed, since $\max_\sigma|H_G(\sigma)|$ is the maximum of the $2^{n+1}$ centered Gaussian variables $\pm H_G(\sigma)$, each of variance $n/2$, its mean is at most $(2\cdot(n/2)\log2^{n+1})^{1/2}\le2n$. Further, it is a $(n/2)^{1/2}$-Lipschitz function of $(g_e)$, and \ref{F:concentration} bounds its fluctuations.
\end{enumerate}
\section{The Sherrington--Kirkpatrick model and the Parisi formula}\label{sec:pre-SK}
\emph{The constants.} Let $H_N^{\mathrm{SK}}(\sigma)=N^{-1/2}\sum_{1\le i0$,
\begin{equation}\label{eq:pre-SK-sandwich}
\beta m_N(h)\le\frac1N\E\log\sum_{\sigma}e^{\beta(H_N^{\mathrm{SK}}(\sigma)+h\sum_i\sigma_i)}\le\beta m_N(h)+\log2.
\end{equation}
By the same argument, we have $\beta e(G;h)\le p_G(\beta,\beta h)\le\beta e(G;h)+\log2$ for every $G\in\cG_D$. Letting $N\to\infty$ in \eqref{eq:pre-SK-sandwich}, we obtain $\limsup_Nm_N(h)\le\PSK(\beta,\beta h)/\beta\le\liminf_Nm_N(h)+(\log2)/\beta$ for every $\beta>0$, and letting $\beta\to\infty$ shows that $m_N(h)$ converges. We write $\Pstar(h)$ for its limit and $\Pstar=\Pstar(0)$. Then $\Pstar(h)=\lim_{\beta\to\infty}\PSK(\beta,\beta h)/\beta$, and for every $\beta>0$,
\begin{equation}\label{eq:pre-PSK-ge-Pstar}
\beta\Pstar(h)\le\PSK(\beta,\beta h)\le\beta\Pstar(h)+\log2.
\end{equation}
\emph{The Parisi formula.} For a probability measure $\mu$ on $[0,1]$ we write $\mu(t)=\mu([0,t])$, and let $\Phi_\mu$ be the solution of the Parisi PDE
\begin{equation}\label{eq:pre-parisi-pde}
\partial_t\Phi+\frac{\beta^2}2\Big(\partial_x^2\Phi+\mu(t)\,(\partial_x\Phi)^2\Big)=0\quad\text{on }[0,1)\times\R,\qquad \Phi(1,x)=\log2\cosh x .
\end{equation}
For atomic $\mu$, Auffinger and Chen~\cite[(4)--(5)]{AuffingerChen2015unique} define $\Phi_\mu$ by successive Cole--Hopf transforms; it is smooth and satisfies \eqref{eq:pre-parisi-pde} classically between consecutive atoms. They extend $\mu\mapsto\Phi_\mu$ to all $\mu$ by continuity, using a Lipschitz estimate of Guerra~\cite{Guerra2003}, and this is the function we use. The PDE does not involve the field. The Parisi functional is
\[
\Par_{\beta,h}(\mu)=\Phi_\mu(0,h)-\frac{\beta^2}2\int_0^1t\,\mu(t)\dd t.
\]
Parisi predicted the formula \eqref{eq:pre-parisi-formula} below~\cite{Parisi1979,Parisi1980Sequence}; Guerra proved the upper bound~\cite{Guerra2003}, and Talagrand proved the formula~\cite{Talagrand2006}. By Talagrand's theorem and the uniqueness theorem of Auffinger and Chen~\cite[Corollary~1]{AuffingerChen2015unique}, for every $\beta>0$ and $h\in\R$,
\begin{equation}\label{eq:pre-parisi-formula}
\PSK(\beta,h)=\min_\mu\Par_{\beta,h}(\mu)=\Par_{\beta,h}(\nu_{\beta,h})
\end{equation}
for a unique minimizer $\nu_{\beta,h}$, the Parisi measure. Since $\Phi_\mu(t,\cdot)$ is even, $\Par_{\beta,h}=\Par_{\beta,-h}$ and $\nu_{\beta,-h}=\nu_{\beta,h}$. We write $q_-=\min\supp\nu_{\beta,h}$ and $q_+=\max\supp\nu_{\beta,h}$.
Most of the cited papers use the terminal condition $\log\cosh x$ and add $\log2$ to the functional, with covariance $N\xi(R)$ and $\xi(s)=\beta^2s^2/2$. In~\cite[Sections~1.1 and~2]{Lopatto2026ExternalField} the Gibbs weight is $\exp(\beta H_N^{\mathrm{SK}}(\sigma)+h\sum_i\sigma_i)$, and the field parameter is the same as ours. The time variable there is $\beta^2t$: in the notation of that paper, $\Phi(t,x)=\log2+U(\beta^2t,x)$, and the Parisi diffusion below is a Brownian time change of the diffusion used there. That paper therefore describes our $\PSK(\beta,h)$ and $\nu_{\beta,h}$, and we use the following theorem from it~\cite[Theorem~1.1]{Lopatto2026ExternalField}. We need it only for $h>0$; the lower bound in zero field follows by continuity in $h$ (\cref{prop:geo-fixed}).
\begin{theorem}\label{thm:pre-supporth}
For every $\beta>0$ and $h>0$, there exist $00$ and $h\ge0$, and write $\nu=\nu_{\beta,h}$ and $\Phi=\Phi_\nu$. Let
\[
v(t,x)=\beta^2\nu(t)\,\partial_x\Phi(t,x),\qquad a(t,x)=\beta\,\partial_x^2\Phi(t,x),
\]
let $B$ be a standard Brownian motion, let $X$ solve
\begin{equation}\label{eq:pre-diffusion}
\mathrm dX_t=v(t,X_t)\dd t+\beta\dd B_t,\qquad X_0=h,
\end{equation}
and let $M_t=\partial_x\Phi(t,X_t)$. The solution $X$ exists and is pathwise unique because $v$ is bounded, measurable, and Lipschitz in $x$ by \ref{P:reg} below. This diffusion appears in~\cite[Theorem~3]{AuffingerChen2015unique}, and it is called the Auffinger--Chen stochastic differential equation (SDE) in~\cite{JagannathTobasco2017}. We call $X$ the Parisi diffusion and $M$ the Parisi martingale. Since $\nu(t)=0$ for $t0$ and $h\ge0$. Then $\Psi(q)=\min_{x\in[0,1]}\Psi(x)$ for every $q\in\supp\nu_{\beta,h}$.
\end{theorem}
The statement in~\cite[Proposition~1.1]{JagannathTobasco2017} is that $\nu(\{x:\Psi(x)=\min\Psi\})=1$; since $\Psi$ is continuous, this set is closed, and as a closed set of full measure it contains $\supp\nu$. The normalizations agree. In~\cite{JagannathTobasco2017}, the model is $\xi=\beta^2\xi_0$, and the SK model corresponds to $\xi_0(t)=t^2/2$. The Parisi PDE there is \eqref{eq:pre-parisi-pde} with terminal condition $\log\cosh x$, and its solution $u_\mu$ is the unique weak solution in the sense of~\cite[Section~8.1]{JagannathTobasco2017}, where the results of~\cite{JagannathTobasco2016dynamic} on the Parisi PDE are recalled. For atomic $\mu$, the function $\Phi_\mu-\log2$ is a weak solution, since it is continuous, has a bounded spatial derivative, and solves the PDE classically between consecutive atoms; it equals $u_\mu$ by the uniqueness in~\cite[Proposition~8.1]{JagannathTobasco2017}. Both $\mu\mapsto u_\mu$ and $\mu\mapsto\Phi_\mu$ are continuous under weak convergence, by~\cite[Proposition~8.2]{JagannathTobasco2017} and \ref{P:reg}; for the first, weak convergence gives convergence of distribution functions at all but countably many points, which implies convergence in the metric $d(\mu,\mu')=\int_0^1|\mu([0,s])-\mu'([0,s])|\dd s$ used there. Approximating $\mu$ weakly by atomic measures, we obtain $u_\mu=\Phi_\mu-\log2$ for all $\mu$. Then the Parisi functional of~\cite{JagannathTobasco2017} is $\Par_{\beta,h}-\log2$, the SDE~\cite[(1.1.2)]{JagannathTobasco2017} is \eqref{eq:pre-diffusion}, the field is the same, and the function $G_\nu$ of~\cite[(1.1.1)]{JagannathTobasco2017} is $\Psi$.
The conditions $\E M_t^2=t$ and $\Sigma(t)\le1$ in the next corollary are the self-consistency conditions of~\cite[Proposition~1.1]{JagannathTobasco2017} and~\cite[Proposition~1]{Chen2017Variational}. If $\nu$ has one atom, they are the conditions of de Almeida and Thouless~\cite{deAlmeidaThouless1978}; see~\cite[Remarks~1.2 and~1.4]{JagannathTobasco2017} and~\cite{Toninelli2002AT}. In zero field, they appear earlier in~\cite[Theorem~5]{AuffingerChen2015properties}, where the expectations are written under a Girsanov change of measure. We include the short derivation from \cref{thm:pre-optimality}; the bound $q_+<1$ is~\cite[Lemma~3.8]{JagannathTobasco2017}, with the same proof.
\begin{corollary}\label{cor:pre-optimality}
Let $\beta>0$ and $h\ge0$. For every $t\in\supp\nu\cap(0,1)$, we have $\E M_t^2=t$ and $\Sigma(t)\le1$. Further, $q_+<1$ and $\E M_{q_+}^2=q_+$.
\end{corollary}
\begin{proof}
If $t\in\supp\nu\cap(0,1)$, then $t$ is an interior minimum point of $\Psi$ by \cref{thm:pre-optimality}, so $\Psi'(t)=0$ and $\Psi''(t)\ge0$, and \eqref{eq:pre-Psi-prime} gives the first claim. If $q_+=1$, then the left derivative of $\Psi$ at $1$ is nonpositive, so $\E M_1^2\ge1$ by \eqref{eq:pre-Psi-prime}; however, $M_1=\partial_x\Phi(1,X_1)=\tanh X_1$ and $|\tanh X_1|<1$ almost surely, a contradiction. Thus $q_+<1$. If $q_+>0$, the first claim gives $\E M_{q_+}^2=q_+$. If $q_+=0$, then the right derivative of $\Psi$ at $0$ is nonnegative, which gives $\E M_0^2\le0$ and $\E M_{q_+}^2=0=q_+$.
\end{proof}
We also need the regularity of $\partial_x\Phi$ and $\partial_x^2\Phi$ in time, which follows from \ref{P:reg}.
\begin{lemma}\label{lem:pre-time-reg}
For every probability measure $\mu$ on $[0,1]$, all $s,t\in[0,1]$, and $x\in\R$,
\[
|\partial_x\Phi_\mu(t,x)-\partial_x\Phi_\mu(s,x)|\le3\beta^2|t-s|,\qquad
|\partial_x^2\Phi_\mu(t,x)-\partial_x^2\Phi_\mu(s,x)|\le4\sqrt3\,\beta\,|t-s|^{1/2}.
\]
\end{lemma}
\begin{proof}
We write $\Phi=\Phi_\mu$. First let $\mu$ be atomic. Between consecutive atoms, $\Phi$ is smooth and satisfies \eqref{eq:pre-parisi-pde} classically, with $\mu(t)$ constant. Differentiating in $x$, we obtain
\[
\partial_t\partial_x\Phi=-\frac{\beta^2}2\big(\partial_x^3\Phi+2\mu(t)\,\partial_x\Phi\,\partial_x^2\Phi\big),
\]
which is at most $\beta^2(4+2)/2=3\beta^2$ in absolute value by \ref{P:reg}. Since $\partial_x\Phi$ is continuous on $[0,1]\times\R$, the first bound holds on $[0,1]$. For the second, fix $s,t$ and let $f=\partial_x\Phi(t,\cdot)-\partial_x\Phi(s,\cdot)$. Then $\|f\|_\infty\le3\beta^2|t-s|$, and $\|f''\|_\infty\le8$ by \ref{P:reg}. If $\|f''\|_\infty=0$, then $f$ is bounded and affine, hence constant, and the claim is immediate. Otherwise, Taylor's formula at $x\pm\epsilon$ gives $2\epsilon|f'(x)|\le2\|f\|_\infty+\epsilon^2\|f''\|_\infty$, and the choice $\epsilon=(2\|f\|_\infty/\|f''\|_\infty)^{1/2}$ gives Landau's inequality $\|f'\|_\infty^2\le2\|f\|_\infty\|f''\|_\infty\le48\beta^2|t-s|$. This is the second bound, since $f'=\partial_x^2\Phi(t,\cdot)-\partial_x^2\Phi(s,\cdot)$. For general $\mu$, we approximate $\mu$ weakly by atomic measures; both bounds pass to the limit by the uniform convergence in \ref{P:reg}.
\end{proof}
\emph{The law of the Parisi center.} We write $\mu_+$ for the law of $M_{q_+}$, and call it the law of the Parisi center. We also set
\[
\mathcal E(\beta,h)=\int_0^{q_+}\E\,a(t,X_t)\dd t .
\]
The next lemma collects the properties of the Parisi martingale used in the lower bound. Part (c) writes $\PSK(\beta,h)$ as the sum of an energy, a field term, an entropy, and a correction, and it identifies the value that the lower bound must reach. For $h>0$ and a measurable $f\colon\R\to[-1,1]$, we define $\phi_f(t)=\E f(Y)f(Y')$ for $t\in[0,q_-]$, where $(Y,Y')$ is a centered Gaussian pair with variances $q_-$ and covariance $t$. The right side of (c) has the form of the TAP free energy of Thouless, Anderson, and Palmer~\cite{ThoulessAndersonPalmer1977} at a magnetization whose coordinates have law $\mu_+$ and whose energy per site is $\mathcal E(\beta,h)$. That the free energy of mixed $p$-spin models equals the maximum of the TAP free energy over magnetizations with self-overlap close to $q_+$ was proved by Chen and Panchenko~\cite[Theorem~1]{ChenPanchenko2018TAP}, and a computation close to our proof of (c) appears in the proof of~\cite[Theorem~7]{CPSII}, without external field. After the change of variables $s=\beta^2t$, the inequality $\phi_f(t)>t$ in (a) is the analogue of the property of the root map in the first phase of Sellke's algorithm~\cite[Lemma~3.2]{Sellke2021field}. There the root map is defined through the Parisi PDE at zero temperature. If $q_-=q_+$, then $f(y)=\tanh(h+\beta y)$, and this inequality follows from~\cite[Lemma~2.2 and Corollary~2.3]{Bolthausen2014}. Our proof uses the same convexity argument as~\cite{Bolthausen2014,Sellke2021field}. The proof of (b) follows~\cite[Lemma~3.7]{Montanari2021}. Since $a(q_+,x)=\beta(1-\tanh^2x)$, the inequality $\E a(q_+,X_{q_+})^2\le1$ used there is Plefka's condition~\cite{Plefka1982} for the law of $M_{q_+}$; see~\cite[Proposition~14]{CPSGeneralizedTAP}.
% CHECK: numbering verified only in arXiv versions: ChenPanchenko2018TAP Theorem 1 (arXiv:1709.03468v2),
% CPSII Theorem 7 (arXiv:1903.01030v1), Sellke2021field Lemma 3.2 (arXiv:2105.03506v6),
% Bolthausen2014 Lemma 2.2 and Corollary 2.3 (arXiv:1201.2891v1), Montanari2021 Lemma 3.7 (arXiv:1812.10897v2),
% CPSGeneralizedTAP Proposition 14 (arXiv:1812.05066v3).
\begin{lemma}\label{lem:pre-parisiTAP}
Let $\beta>0$ and $h>0$.
\begin{enumerate}[label=(\alph*)]
\item We have $\E M_t^2=t$ for $t\in[q_-,q_+]$. If $q_-t$ for every $t\in[0,q_-)$.
\item We have $M_{q_+}=\tanh X_{q_+}$ and $\beta^2(1-q_+)^2\le1$.
\item We have
\begin{equation}\label{eq:pre-TAP}
\PSK(\beta,h)=\beta\,\mathcal E(\beta,h)+h\,\E M_{q_+}+\E\ent(M_{q_+})+\frac{\beta^2}4(1-q_+)^2 .
\end{equation}
\end{enumerate}
\end{lemma}
\begin{proof}
(a) By \cref{thm:pre-supporth}, $[q_-,q_+]=\supp\nu\subset(0,1)$, and \cref{cor:pre-optimality} implies that $\E M_t^2=t$ for $t\in[q_-,q_+]$. If $q_-q_--(q_--t)=t$.
(b) By \ref{P:colehopf}, we have $\partial_x\Phi(q_+,x)=\tanh x$ and $a(q_+,x)=\beta(1-\tanh^2x)$. Then $M_{q_+}=\tanh X_{q_+}$, and (a) gives $\E\tanh^2X_{q_+}=q_+$. Hence $\beta(1-q_+)=\E a(q_+,X_{q_+})$. Since $q_+\in(0,1)$, Jensen's inequality and \cref{cor:pre-optimality} at $q_+$ give $\beta^2(1-q_+)^2\le\E a(q_+,X_{q_+})^2\le1$.
(c) We compute each term of $\Par_{\beta,h}(\nu)$, which equals $\PSK(\beta,h)$ by \eqref{eq:pre-parisi-formula}.
\begin{itemize}
\item Since $\nu(r)=0$ for $r0$ and $\sum_jw_j=1$. For $x,y\in\R^{K+1}$ we write
$\langle x,y\rangle_w=\sum_{j=0}^Kw_jx_jy_j$, $\|x\|_{1,w}=\sum_jw_j|x_j|$, and
$\|x\|_{2,w}=\langle x,x\rangle_w^{1/2}$. Since $\sum_jw_j=1$, we have
$\|x\|_{1,w}\le\|x\|_{2,w}$. The ordered cone is
\[
\upC_K=\{q\in\R^{K+1}:0\le q_0\le q_1\le\dots\le q_K\},
\]
and its dual cone is
$\upC_K^*=\{z\in\R^{K+1}:\langle z,q\rangle_w\ge0\text{ for all }q\in\upC_K\}$.
Every $q\in\upC_K$ can be written as
$q=\sum_{\ell=0}^K(q_\ell-q_{\ell-1})\mathbf 1_{[\ell,K]}$ with $q_{-1}=0$, where
$\mathbf 1_{[\ell,K]}$ is the vector whose coordinates with index at least $\ell$
are equal to one and whose other coordinates vanish. Since the coefficients
are nonnegative, we have
\begin{equation}\label{eq:up-dualcone}
\upC_K^*=\Big\{z\in\R^{K+1}:\sum_{j=\ell}^Kw_jz_j\ge0\text{ for }0\le\ell\le K\Big\}.
\end{equation}
We write $z\succeq_*z'$ if $z-z'\in\upC_K^*$. A function $u\colon\upC_K\to\R$ is
called \emph{differentiable on $\upC_K$} if there exists a continuous map
$Du\colon\upC_K\to\R^{K+1}$ such that
$u(q')=u(q)+\langle Du(q),q'-q\rangle_w+o(|q'-q|)$ as $q'\to q$ within $\upC_K$;
we call $Du$ the weighted gradient of $u$. Its coordinates are
$(Du)_j=w_j^{-1}\partial u/\partial q_j$. If $u$ is differentiable on $\upC_K$,
then for $q,q'\in\upC_K$ the segment from $q$ to $q'$ lies in $\upC_K$ and
\begin{equation}\label{eq:up-segment}
u(q')-u(q)=\int_0^1\langle Du(q+\theta(q'-q)),q'-q\rangle_w\dd\theta .
\end{equation}
We write $Z,Z_0,Z_1,\dots$ for standard Gaussian variables and, for $b\ge0$ and a
continuous function $g$ of at most exponential growth,
$\gamma_b*g(x)=\E g(x+\sqrt b\,Z)$. If $g$ is smooth and all its derivatives
have at most exponential growth, then $b\mapsto\gamma_b*g(x)$ is smooth on
$[0,\infty)$, with one-sided derivatives at $b=0$, and
\begin{equation}\label{eq:up-heat}
\partial_b(\gamma_b*g)=\frac12\gamma_b*g''=\frac12\partial_x^2(\gamma_b*g).
\end{equation}
\section{Cascade recursions}\label{sec:up-recursion}
We consider recursions in which, at each level, a Gaussian increment of the
field is integrated together with an auxiliary random element. Let
$(\Omega_\ell,\mathscr A_\ell,\mathsf P_\ell)$, $1\le\ell\le K$, be probability
spaces, let $\omega=(\omega_1,\dots,\omega_K)$ have independent coordinates with
$\omega_\ell\sim\mathsf P_\ell$, and let $Z_1,\dots,Z_K$ be independent standard
Gaussian variables, independent of $\omega$. We write
$\omega_{\le k}=(\omega_1,\dots,\omega_k)$. Let
$L\colon\R\times\Omega_1\times\dots\times\Omega_K\to\R$ be measurable and such that
$L(\cdot,\omega)$ is smooth for each $\omega$, with
\begin{equation}\label{eq:up-terminal}
|\partial_xL|\le1,\qquad\partial_x^2L=1-(\partial_xL)^2,\qquad
\E\exp\big(c|L(0,\omega)|\big)<\infty\ \text{ for every }c>0 .
\end{equation}
The two main examples are $L(x)=\log2\cosh x$, without auxiliary randomness, and
$L(x,\omega)=\log\sum_{\sigma}\exp(x\sigma_{x_0}+A(\sigma,\omega))$, where $\sigma$
ranges over a finite set of spin configurations and $x_0$ is a fixed site; then
$\partial_xL=\langle\sigma_{x_0}\rangle$ and
$\partial_x^2L=1-\langle\sigma_{x_0}\rangle^2$ for the corresponding Gibbs
average. The second relation in \eqref{eq:up-terminal} implies, by induction,
that every derivative of $L(\cdot,\omega)$ of order $r\ge1$ is a polynomial in
$\partial_xL$, so it is bounded by a constant depending only on $r$.
Let $\Delta_1,\dots,\Delta_K\ge0$. We set $L_K=L$ and, for $1\le\ell\le K$,
\begin{equation}\label{eq:up-rec}
L_{\ell-1}(x,\omega_{\le\ell-1})=\frac1{\zeta_{\ell-1}}\log\E\Big[\exp\Big(\zeta_{\ell-1}
L_\ell\big(x+\sqrt{\Delta_\ell}\,Z_\ell,\omega_{\le\ell-1},\omega_\ell\big)\Big)\Big],
\end{equation}
where the expectation is over $(Z_\ell,\omega_\ell)$ only. The function $L_0$ is
then a deterministic function of $x$. All recursions and identities involving
auxiliary randomness are understood for almost every auxiliary prefix; on
exceptional null sets, we choose arbitrary versions of the recursive functions
and transition kernels. For $0\le k0$. Since
$|L_\ell(x,\cdot)-L_\ell(0,\cdot)|\le|x|$, differentiation under the expectation in
\eqref{eq:up-rec} is justified by dominated convergence, and
\begin{equation}\label{eq:up-firstsecond}
\partial_xL_{\ell-1}=\upE_{\ell-1}[\partial_xL_\ell],\qquad
\partial_x^2L_{\ell-1}=\upE_{\ell-1}[\partial_x^2L_\ell]
+\zeta_{\ell-1}\Big(\upE_{\ell-1}\big[(\partial_xL_\ell)^2\big]-(\partial_xL_{\ell-1})^2\Big),
\end{equation}
where $\upE_{\ell-1}$ denotes the expectation of the step from level $\ell-1$
to level $\ell$ in \eqref{eq:up-tilt}. Higher derivatives of $L_{\ell-1}$ are
polynomials, with coefficients bounded in terms of the order, in tilted
moments of derivatives of $L_\ell$, and they are bounded by constants depending
only on the order and on $K$. Let $Y=|L_\ell(0,\cdot)|+\sqrt{\Delta_\ell}|Z_\ell|$. By
\eqref{eq:up-rec} and Jensen's inequality,
$|L_{\ell-1}(0,\cdot)|\le\zeta_{\ell-1}^{-1}\log\E_\ell e^{\zeta_{\ell-1}Y}$, where
$\E_\ell$ integrates $(Z_\ell,\omega_\ell)$. For $c\ge\zeta_{\ell-1}$, using Jensen's
inequality again, we obtain $\E\exp(c|L_{\ell-1}(0,\cdot)|)\le\E e^{cY}<\infty$.
By downward induction, this proves the first two assertions in (a) and the
martingale identity. Since $M_k=\upE[M_j\,|\,X_k,\omega_{\le k}]$ for $k\le j$, the
monotonicity of $j\mapsto\Lambda_{k,j}$ follows from the conditional form of
Jensen's inequality, and $|M_j|\le1$ implies that $\Lambda_{k,K}\le1$.
We prove (b) by downward induction. For $k=K$ it is the second relation in
\eqref{eq:up-terminal}, since $\zeta_K=1$. Suppose that \eqref{eq:up-curvature}
holds at level $k+1$. By the Markov property,
$\upE_k[\Lambda_{k+1,j}]=\Lambda_{k,j}$ for $j\ge k+1$, and inserting this into
\eqref{eq:up-firstsecond} we obtain
\[
\partial_x^2L_k=1-\zeta_{k+1}\Lambda_{k,k+1}-\sum_{j\ge k+2}w_j\Lambda_{k,j}
+\zeta_k\Lambda_{k,k+1}-\zeta_k\Lambda_{k,k}.
\]
Since $\zeta_{k+1}-\zeta_k=w_{k+1}$, this is \eqref{eq:up-curvature} at level $k$.
For (c), let first $\ell=k+1$. By \eqref{eq:up-heat}, applied for fixed
$(\omega_{\le k+1})$ and then averaged over $\omega_{k+1}$, we have
\[
\partial_{\Delta_{k+1}}L_k=\frac{1}{2\zeta_k}\,
\frac{\partial_x^2\exp(\zeta_kL_k)}{\exp(\zeta_kL_k)}
=\frac12\big(\partial_x^2L_k+\zeta_k(\partial_xL_k)^2\big),
\]
and \eqref{eq:up-curvature} turns the right side into
$(1-\sum_{j\ge k+1}w_j\Lambda_{k,j})/2$. If $\ell>k+1$, then
$\partial_{\Delta_\ell}L_k=\upE_k[\partial_{\Delta_\ell}L_{k+1}]$, because $\Delta_\ell$
enters \eqref{eq:up-rec} at level $k$ only through $L_{k+1}$. By induction on
$\ell-k$ and the identity $\upE_k[\Lambda_{k+1,j}]=\Lambda_{k,j}$, we obtain
\eqref{eq:up-heatderiv}.
\end{proof}
\section{The scalar transform}\label{sec:up-transform}
We now specialize \cref{lem:up-recursion} to $L(x)=\log2\cosh x$, without
auxiliary randomness. For $v\in\upC_K$ we take $\Delta_\ell=v_\ell-v_{\ell-1}$
for $1\le\ell\le K$, write $\phi_k=L_k$, and let the chain start from
$X_0=h+\sqrt{v_0}\,Z_0$. The \emph{compensated scalar transform} is
\begin{equation}\label{eq:up-Psi}
\Psi_h(v)=\frac{v_K}2-\E\,\phi_0\big(h+\sqrt{v_0}\,Z_0\big),
\qquad
\psi_h(q)=\Psi_h(2q)\qquad(v,q\in\upC_K).
\end{equation}
A level of $\psi_h$ with endpoint $q_j$ has Gaussian variance $2q_j$. For $h=0$
and up to the additive constant $\log2$, the function $\psi_h$ is the initial
condition of the Hamilton--Jacobi equation for the Sherrington--Kirkpatrick model
in \cite{Mourrat2022Wasserstein}; see also \cite[Lemma~6.4]{DominguezMourrat2024}. We
also write
\begin{equation}\label{eq:up-T}
T_j(h)=T_j(h;v)=\E\,\Lambda_{0,j}\big(h+\sqrt{v_0}\,Z_0\big)=\upE\big[\phi_j'(X_j)^2\big]
\qquad(0\le j\le K).
\end{equation}
\begin{lemma}\label{lem:up-psi}
For every $h\in\R$, the function $\Psi_h$ is differentiable on $\upC_K$, and
\begin{equation}\label{eq:up-Psigrad}
\partial_{v_j}\Psi_h(v)=\frac{w_j}2\,T_j(h;v)\qquad(0\le j\le K),\qquad
0\le T_0\le T_1\le\dots\le T_K\le1 .
\end{equation}
The function $\psi_h$ is differentiable on $\upC_K$ with
$D\psi_h(q)=(T_j(h;2q))_{j\le K}\in\upC_K\cap[0,1]^{K+1}$. In particular,
$|\psi_h(q)-\psi_h(q')|\le\|q-q'\|_{1,w}$ for $q,q'\in\upC_K$, and
$\psi_h(q)\le\psi_h(q')$ if moreover $q\le q'$ coordinatewise.
\end{lemma}
\begin{proof}
Iterating \eqref{eq:up-heat} through the finite recursion gives all mixed
derivatives of $\phi_0$ in $x$ and in the durations $\Delta_\ell\in[0,\infty)$.
On compact sets of durations, the differentiated integrands have a common
Gaussian-integrable exponential bound, by \cref{lem:up-recursion}(a).
Dominated convergence therefore gives joint continuity of these derivatives,
with one-sided derivatives when a duration is zero. Applying the same argument
to the outer convolution $\gamma_{v_0}*\phi_0$ shows that $\Psi_h$ is smooth in
$(v_0,\Delta_1,\dots,\Delta_K)$ on $[0,\infty)^{K+1}$ in this sense, and
differentiable on $\upC_K$. For $1\le j0$, then $1/g$ is completely monotone.
\item A function $f$ is completely monotone if and only if
$f(s)=\int_{[0,\infty)}e^{-\lambda s}\mu(\mathrm d\lambda)$ for a finite positive
measure $\mu$ with $\int\lambda^k\mu(\mathrm d\lambda)<\infty$ for every $k$.
\item If $G$ is completely monotone in $x^2$ and $b\ge0$, then $\gamma_b*G$ is
completely monotone in $x^2$, and $b\mapsto\gamma_b*G(0)$ is completely monotone
on $[0,\infty)$.
\end{enumerate}
\end{lemma}
\begin{proof}
Part (i) follows from the Leibniz rule and differentiation under the integral
sign. For (ii), let $H=e^{-g}$. Then $H'=-g'H$ and, by the Leibniz rule,
\[
(-1)^{k+1}H^{(k+1)}=\sum_{i=0}^k\binom ki\big[(-1)^ig'^{(i)}\big]\big[(-1)^{k-i}H^{(k-i)}\big],
\]
which is nonnegative by induction on $k$. If $g>0$ and $H=1/g$, then
$H'=-g'H^2$, and the same induction, applied to $H$ and to $H^2$ simultaneously,
shows that $(-1)^kH^{(k)}\ge0$. Part (iii) is Bernstein's theorem
\cite[Theorem~1.4]{SchillingSongVondracek2012}% Verified in the publisher preview of the second edition.
; the measure is finite because $f(0)<\infty$, and its moments are finite because
$f$ is smooth at $0$. For (iv), we use (iii) to write $\widetilde G(s)=\int e^{-\lambda s}\mu(\mathrm d\lambda)$.
For $\lambda\ge0$ and $b\ge0$ we have
\[
\E\exp\big(-\lambda(x+\sqrt bZ)^2\big)=(1+2b\lambda)^{-1/2}\exp\Big(-\frac{\lambda x^2}{1+2b\lambda}\Big).
\]
Integrating against $\mu$ represents $\gamma_b*G$ as a mixture of the functions
$x\mapsto e^{-\lambda'x^2}$ with $\lambda'=\lambda/(1+2b\lambda)\ge0$, with finite
positive weights; differentiation under the integral is justified because
$\lambda'\le\lambda$ and $\mu$ has finite moments. At $x=0$ we obtain
$\int(1+2b\lambda)^{-1/2}\mu(\mathrm d\lambda)$, and $b\mapsto(1+2b\lambda)^{-1/2}$
is completely monotone.
\end{proof}
The derivatives that we must control satisfy linear parabolic equations in the
variable $s=x^2$, which degenerate at $s=0$. We use the following maximum
principle.
\begin{lemma}\label{lem:up-max}
Let $d>0$ and let $z\colon[0,d]\times[0,\infty)\to\R$ be continuous, such that
$\partial_bz$, $\partial_sz$, and $\partial_s^2z$ exist and are continuous on
$(0,d]\times[0,\infty)$, with one-sided derivatives at $s=0$. Suppose that
$|z(b,s)|\le C_ze^{\sqrt s}$, that $z(0,\cdot)\ge0$, and that
\[
\partial_bz\ge2s\,\partial_s^2z+\beta(b,s)\,\partial_sz+V(b,s)\,z
\qquad\text{on }(0,d]\times[0,\infty),
\]
where $0\le\beta(b,s)\le C_\beta(1+\sqrt s)$ and $|V(b,s)|\le C_V$. Then $z\ge0$.
\end{lemma}
\begin{proof}
Let $\Phi(s)=\cosh(2\sqrt s)=\sum_{k\ge0}4^ks^k/(2k)!$, which is smooth on
$[0,\infty)$. We have $\Phi'(s)=\sinh(2\sqrt s)/\sqrt s\ge0$ and
$2s\Phi''(s)=2\cosh(2\sqrt s)-\sinh(2\sqrt s)/\sqrt s$. Using
$\sinh y\le y\cosh y$ and $\sinh y\le\cosh y$ for $y\ge0$, we find
\[
2s\Phi''+\beta\Phi'\le2\Phi+C_\beta\Big(\frac{\sinh(2\sqrt s)}{\sqrt s}+\sinh(2\sqrt s)\Big)
\le(2+3C_\beta)\Phi .
\]
Set $C_\Phi=2+3C_\beta$, fix $\Lambda>C_V$, and for $\epsilon>0$ let
$z_\epsilon(b,s)=e^{-\Lambda b}z(b,s)+\epsilon e^{C_\Phi b}\Phi(s)$. Since
$\Phi(s)\ge e^{2\sqrt s}/2$, the function $z_\epsilon$ tends to $+\infty$ as
$s\to\infty$, uniformly in $b\in[0,d]$, so it attains its minimum on
$[0,d]\times[0,\infty)$ at some point $(b_1,s_1)$. Suppose that this minimum is
negative. Then $b_1>0$, because $z_\epsilon(0,\cdot)>0$. At $(b_1,s_1)$ we have
$\partial_bz_\epsilon\le0$; moreover
$2s\partial_s^2z_\epsilon+\beta\partial_sz_\epsilon\ge0$, since either $s_1>0$, in
which case $\partial_sz_\epsilon=0\le\partial_s^2z_\epsilon$, or $s_1=0$, in which
case $\partial_sz_\epsilon\ge0$ and the second-order term vanishes. On the other
hand, by the assumed inequality, we have
\[
\partial_bz_\epsilon\ge\big(2s\partial_s^2z_\epsilon+\beta\partial_sz_\epsilon\big)
+\epsilon e^{C_\Phi b}\big(C_\Phi\Phi-2s\Phi''-\beta\Phi'\big)+(V-\Lambda)e^{-\Lambda b}z .
\]
At $(b_1,s_1)$, the first two terms are nonnegative and
$e^{-\Lambda b}z=z_\epsilon-\epsilon e^{C_\Phi b}\Phi<0$, so the last term is
positive. Then $\partial_bz_\epsilon>0$ there, a contradiction. We conclude that
$z_\epsilon\ge0$ for all $\epsilon>0$, and $z\ge0$.
\end{proof}
We now show that one step of the recursion \eqref{eq:up-rec}, without auxiliary
randomness, preserves the relevant complete monotonicity properties.
\begin{lemma}\label{lem:up-step}
Let $00$; we prove the claim for $b\in[0,d]$. The function $\phi(b,\cdot)$ is
even and smooth, and \eqref{eq:up-firstsecond}, with $\zeta_{\ell-1}$ replaced by
$m$, shows that $|\partial_x\phi|\le1$, $\partial_x^2\phi\ge0$, and that the
derivatives of order $r\ge1$ are bounded uniformly in $b\in[0,d]$. Similarly,
$\Lambda^\iota(b,\cdot)$ takes values in $[0,1]$ and has bounded derivatives, since it
is a tilted average of $\Lambda^\iota_{\rm in}$. Let $u=\partial_x\phi$ and
$\psi=\partial_x^2\phi$. By \eqref{eq:up-heat}, for $b>0$,
\begin{equation}\label{eq:up-burgers}
\partial_b\phi=\frac12\big(\partial_x^2\phi+m(\partial_x\phi)^2\big),\qquad
\partial_bu=\frac12\partial_x^2u+mu\partial_xu,\qquad
\partial_b\psi=\frac12\partial_x^2\psi+mu\partial_x\psi+m\psi^2,
\end{equation}
and $2\partial_b\Lambda^\iota=\partial_x^2\Lambda^\iota+2mu\partial_x\Lambda^\iota$.
For $c\in[0,m]$ we set $H_c=e^{c\phi}\psi$, $G_c=e^{c\phi}(1-u^2)$, and
$\Theta^\iota_c=e^{c\phi}(1-\Lambda^\iota)$. Using \eqref{eq:up-burgers}, we compute
\begin{equation}\label{eq:up-HGK}
\begin{aligned}
\partial_bH_c&=\frac12\partial_x^2H_c+(m-c)u\,\partial_xH_c
+\big(m\psi-\frac12c(m-c)u^2\big)H_c,\\
\partial_bG_c&=\frac12\partial_x^2G_c+(m-c)u\,\partial_xG_c
-\frac12c(m-c)u^2G_c+H_c\psi,\\
\partial_b\Theta^\iota_c&=\frac12\partial_x^2\Theta^\iota_c+(m-c)u\,\partial_x\Theta^\iota_c
-\frac12c(m-c)u^2\Theta^\iota_c .
\end{aligned}
\end{equation}
We pass to the variable $s=x^2$. Let $f(b,s)=\psi(b,\sqrt s)$ and
$b_*(b,s)=\sqrt s\,u(b,\sqrt s)$. For an even function $F(b,\cdot)$ we have
$\partial_x^2F=2\partial_s\widetilde F+4s\,\partial_s^2\widetilde F$ and
$\gamma u\,\partial_xF=2\gamma b_*\partial_s\widetilde F$, where $\widetilde F(b,s)=F(b,\sqrt s)$.
Since $u(b,0)=0$, we have $u(b,x)=x\int_0^1\psi(b,x\theta)\dd\theta$, which implies
\begin{equation}\label{eq:up-bstar}
\partial_sb_*=\frac12\Big[f(s)+\int_0^1f(s\theta^2)\dd\theta\Big],\qquad
\partial_s\big(u(\sqrt s)^2\big)=f(s)\int_0^1f(s\theta^2)\dd\theta .
\end{equation}
Since $\phi(b,\cdot)$ is even and convex, $u(b,x)$ has the sign of $x$, and
$0\le b_*\le\sqrt s$ because $|u|\le1$. Since $\psi$ is nonnegative and bounded, we see from
\eqref{eq:up-bstar} that $\partial_sb_*$ is nonnegative and bounded.
All functions below are smooth in $(b,s)\in(0,d]\times[0,\infty)$ and continuous
on $[0,d]\times[0,\infty)$ together with their $s$-derivatives, by
\eqref{eq:up-Dop} and the smoothness of the heat semigroup. Since
$\phi(b,x)\le\phi(b,0)+|x|$ and $\phi(b,0)$ is bounded on $[0,d]$, every $x$-derivative
of $H_c$, $G_c$, and $\Theta_c^\iota$ is bounded by $Ce^{c|x|}\le Ce^{|x|}$, and by
\eqref{eq:up-Dop} every $s$-derivative of their transforms is bounded by
$Ce^{\sqrt s}$. The functions $f$, $b_*$, and $u^2$ have bounded derivatives.
\emph{Step 1: $f(b,\cdot)$ is completely monotone.} Set
$\varpi_k=(-1)^k\partial_s^kf$ and, for $j\ge1$,
$B_j=(-1)^{j-1}\partial_s^jb_*$. By \eqref{eq:up-bstar},
\[
B_j=\frac12\Big[\varpi_{j-1}(s)+\int_0^1\theta^{2j-2}\varpi_{j-1}(s\theta^2)\dd\theta\Big].
\]
In the variable $s$, the last equation in \eqref{eq:up-burgers} reads
$\partial_bf=2s\partial_s^2f+(1+2mb_*)\partial_sf+mf^2$. Differentiating $k\ge1$
times in $s$, we obtain
\begin{equation}\label{eq:up-fk}
\begin{aligned}
\partial_b\varpi_k={}&2s\,\partial_s^2\varpi_k+(2k+1+2mb_*)\partial_s\varpi_k
+\big(2mk\,\partial_sb_*+2mf\big)\varpi_k\\
&+m\sum_{i=1}^{k-1}\binom ki\varpi_i\varpi_{k-i}
+2m\sum_{i=2}^k\binom kiB_i\varpi_{k-i+1}.
\end{aligned}
\end{equation}
We have $\varpi_0=f\ge0$ because $\phi(b,\cdot)$ is convex. Suppose that
$\varpi_0,\dots,\varpi_{k-1}\ge0$ on $[0,d]\times[0,\infty)$. Then
$B_2,\dots,B_k\ge0$, and the two sums in \eqref{eq:up-fk} are nonnegative,
because they involve only $\varpi_1,\dots,\varpi_{k-1}$ and $B_2,\dots,B_k$.
Further, $\varpi_k(0,\cdot)\ge0$ by \eqref{eq:up-stephyp}. \Cref{lem:up-max},
applied with $\beta=2k+1+2mb_*$ and $V=2mk\partial_sb_*+2mf$, implies that
$\varpi_k\ge0$. By induction, $f(b,\cdot)$ is completely monotone. By
\eqref{eq:up-bstar} and \cref{lem:up-cmfacts}(i), $(-1)^{j-1}\partial_s^jb_*\ge0$ and
$(-1)^{j-1}\partial_s^j(u^2)\ge0$ for all $j\ge1$.
\emph{Step 2: a general induction.} Let $F$ be one of $H_c$, $G_c$, $\Theta^\iota_c$,
with $c\in[0,m]$ and $\gamma=m-c\ge0$. By \eqref{eq:up-HGK}, $\widetilde F$
satisfies
\[
\partial_b\widetilde F=2s\,\partial_s^2\widetilde F+(1+2\gamma b_*)\partial_s\widetilde F
+P\widetilde F+J,
\]
where $P=mf-c\gamma u^2/2$ and $J=0$ for $H_c$, $P=-c\gamma u^2/2$ and
$J=\widetilde{H_c}f$ for $G_c$, and $P=-c\gamma u^2/2$ and $J=0$ for $\Theta_c^\iota$.
By Step 1, $(-1)^jP^{(j)}\ge0$ for every $j\ge1$, where derivatives are in $s$.
Let $z_k=(-1)^k\partial_s^k\widetilde F$. Differentiating $k$ times, we find
\[
\begin{aligned}
\partial_bz_k={}&2s\,\partial_s^2z_k+(2k+1+2\gamma b_*)\partial_sz_k
+\big(2\gamma k\,\partial_sb_*+P\big)z_k\\
&+2\gamma\sum_{i=2}^k\binom kiB_iz_{k-i+1}
+\sum_{i=1}^k\binom ki(-1)^iP^{(i)}z_{k-i}+(-1)^kJ^{(k)} .
\end{aligned}
\]
If $J$ is completely monotone and $z_0,\dots,z_{k-1}\ge0$, then all source terms
are nonnegative, $z_k(0,\cdot)\ge0$ by \eqref{eq:up-stephyp}, and
\cref{lem:up-max} gives $z_k\ge0$. The base case $z_0=\widetilde F\ge0$ holds
because $\psi\ge0$, $|u|\le1$, and $\Lambda^\iota\le1$.
We apply Step 2 first to $H_c$, with $J=0$, which shows that $H_c(b,\cdot)$ is
completely monotone in $x^2$. Then $J=\widetilde{H_c}f$ is completely monotone,
and Step 2 applies to $G_c$. Finally it applies to $\Theta_c^\iota$. Together with
Step 1, this proves the lemma.
\end{proof}
\begin{lemma}\label{lem:up-cm}
Let $v\in\upC_K$, and let $\phi_k$ and $\Lambda_{k,j}$ be as in
\cref{sec:up-transform}. For every $0\le k\le j\le K$ and $c\in[0,\zeta_k]$, the functions
\[
\phi_k'',\qquad e^{c\phi_k}\phi_k'',\qquad e^{c\phi_k}\big(1-\Lambda_{k,j}\big)
\]
are completely monotone in $x^2$. Further, $\phi_k'$ is odd and
nondecreasing, $\Lambda_{k,j}$ is even and nondecreasing in $|x|$, and for every
$j$ there exists a finite positive measure $\nu_j$ on $[0,\infty)$, with finite
moments and total mass $1-T_j(0)$, such that
\begin{equation}\label{eq:up-Tlaplace}
1-T_j(h)=\int_{[0,\infty)}e^{-\lambda h^2}\nu_j(\mathrm d\lambda)\qquad(h\in\R).
\end{equation}
\end{lemma}
\begin{proof}
We prove the complete-monotonicity claims by downward induction, carrying also
that $\phi_k$ and $\Lambda_{k,j}$ are smooth and even, that their derivatives
of every positive order are bounded, and that $0\le\Lambda_{k,j}\le1$.
For $k=K$ these additional properties hold, and we have
$\phi_K=\log2\cosh x$, $\phi_K''=1-(\phi_K')^2=\operatorname{sech}^2x$,
$\Lambda_{K,K}=\tanh^2x$, and $\zeta_K=1$. The three functions are then
$\operatorname{sech}^2$ and $2^c\operatorname{sech}^{2-c}$ with $2-c\ge1$. By the
product formula $\cosh x=\prod_{i\ge1}(1+4x^2/((2i-1)^2\pi^2))$
\cite[(4.36.2)]{DLMF},% Equation 4.36.2 checked directly on NIST DLMF.
\[
\partial_s\log\cosh\sqrt s=\sum_{i\ge1}\frac{4}{(2i-1)^2\pi^2+4s},
\]
where the series of derivatives of every order converges uniformly on
$[0,\infty)$. Since each summand is completely monotone, $\log\cosh\sqrt s$ is a
Bernstein function, and $\operatorname{sech}^p\sqrt s=e^{-p\log\cosh\sqrt s}$ is
completely monotone for all $p>0$ by \cref{lem:up-cmfacts}(ii).
Suppose that the strengthened claim holds at level $k+1\le K$. We apply \cref{lem:up-step}
with $m=\zeta_k$, $m'=\zeta_{k+1}$, $\phi_{\rm in}=\phi_{k+1}$, the family
$\Lambda^j_{\rm in}=\Lambda_{k+1,j}$ for $k+1\le j\le K$, and $b=\Delta_{k+1}$. The
hypotheses hold by the induction hypothesis and \cref{lem:up-recursion}(a): the
function $\phi_{k+1}$ is even, because $\log2\cosh$ is even and the Gaussian
increments are symmetric; it is convex because $\phi_{k+1}''$, being completely
monotone in $x^2$, is nonnegative; and $(\phi'_{k+1})^2=\Lambda_{k+1,k+1}$. By the definition of the tilted
chain, $\phi(\Delta_{k+1},\cdot)=\phi_k$ and
$\Lambda^j(\Delta_{k+1},\cdot)=\Lambda_{k,j}$ for $j\ge k+1$, and
$\Lambda_{k,k}=(\phi_k')^2$. \Cref{lem:up-step} preserves the additional
regularity and range bounds, which also hold for $(\phi_k')^2$, and implies
the claim at level $k$.
Since $\phi_k$ is even, $\phi_k'$ is odd, and it is nondecreasing
because $\phi_k''\ge0$. Since $1-\Lambda_{k,j}$ is completely monotone in $x^2$,
it is nonincreasing in $|x|$. Finally, $1-\Lambda_{0,j}$ is completely monotone
in $x^2$, so $1-T_j=\gamma_{v_0}*(1-\Lambda_{0,j})$ is completely monotone in $h^2$
by \cref{lem:up-cmfacts}(iv), and \eqref{eq:up-Tlaplace} follows from
\cref{lem:up-cmfacts}(iii).
\end{proof}
The monotonicity of $\phi_k'$ and of $\Lambda_{k,j}$ in $|x|$ also follows from
\cite[Lemma~2]{Panchenko2005question}. The complete monotonicity is the additional
information used in \cref{lem:up-floor,prop:up-sqrt}.
\section{Square-root convexity}\label{sec:up-sqrt}
The next estimate bounds the curvature of $T_j$ at $h=0$ in terms of the total
variance $v_j$. We write $T_j''$ for the second derivative in $h$.
\begin{lemma}\label{lem:up-floor}
For every $v\in\upC_K$ and $0\le j\le K$, we have
$v_jT_j''(0)\le2T_j(0)$.
\end{lemma}
\begin{proof}
Fix $j$. We follow the response $\Lambda_{k,j}$ as $k$ decreases from $j$ to $0$,
and then through the outer Gaussian average. Let $0\le kk+1}w_i=1$,
\begin{align*}
\mathsf D''&=\zeta_k\mathsf D\big(\phi_{k+1}''+\zeta_k(\phi_{k+1}')^2\big)\\
&=\zeta_k^2\mathsf D+\zeta_k(\zeta_{k+1}-\zeta_k)\mathsf D(1-\Lambda_{k+1,k+1})
+\zeta_k\sum_{i>k+1}w_i\mathsf D(1-\Lambda_{k+1,i}).
\end{align*}
By \eqref{eq:up-heat},
\[
B'=\frac12\gamma_b*\mathsf D''(0)=\frac{\zeta_k^2}2B+\frac{\zeta_k}2\sum_{i\ge k+1}c_iN_i,
\qquad c_i\ge0 .
\]
The function
$\widetilde B=e^{-\zeta_k^2b/2}B$ is then positive with completely monotone derivative.
By \cref{lem:up-cmfacts}(ii), $1/\widetilde B$ is completely monotone, and
$1-g=N e^{-\zeta_k^2b/2}/\widetilde B$ is completely monotone by
\cref{lem:up-cmfacts}(i). It follows that $g$ is nondecreasing and concave on
$[0,\Delta_{k+1}]$. For the outer Gaussian average the same holds with
$g(b)=\gamma_b*\Lambda_{0,j}(0)$, $b\in[0,v_0]$, directly from
\cref{lem:up-cmfacts}(iv).
Since $\partial_x\phi(b,0)=0$, we have $2g'(b)=\partial_x^2\Lambda(b,0)$ by the
transport equation for $\Lambda$. Suppose that at the start of a step
the response satisfies $b_0g'(0)\le g(0)$ for some $b_0\ge0$. By concavity,
$bg'(b)\le g(b)-g(0)$ and $g'(b)\le g'(0)$. Using these bounds, we obtain
\[
(b_0+b)g'(b)\le g(b)-g(0)+b_0g'(0)\le g(b).
\]
At level $j$ the response is $\Lambda_{j,j}=(\phi_j')^2$, which vanishes at $0$,
so the inequality holds with $b_0=0$. At the junction of two consecutive steps,
the values of $g$ and $g'$ are $\Lambda_{k,j}(0)$ and $\Lambda_{k,j}''(0)/2$
for both steps. Applying the previous display through the steps of durations
$\Delta_j,\dots,\Delta_1$ and then through the outer average of variance $v_0$,
whose total is $v_j$, we obtain
$v_jT_j''(0)\le2T_j(0)$.
\end{proof}
\begin{proposition}\label{prop:up-sqrt}
For every $h\in\R$, the map $s\mapsto\Psi_h(s_0^2,\dots,s_K^2)$ is convex on
$\{s\in\R^{K+1}:0\le s_0\le s_1\le\dots\le s_K\}$. The same holds for
$s\mapsto\psi_h(s_0^2,\dots,s_K^2)$.
\end{proposition}
\begin{proof}
The second statement follows from the first, since
$\psi_h(s^2)=\Psi_h((\sqrt2s)^2)$. Let
$\mathcal O=\{s:00$. By
\eqref{eq:up-rowsum},
\[
(\mathsf M\varrho)_i=\frac{w_iT_i}{s_i}+2w_is_i\sum_jJ_{ij}=\frac{w_i}{s_i}\big(T_i+v_iT_i''(h)\big)\ge0 .
\]
For $y\in\R^{K+1}$ and $\xi_i=\varrho_iy_i$, using the symmetry of $\mathsf M$, we have
\[
\xi^{\mathsf T}\mathsf M\xi=\sum_{i1/3$.
The same threshold is stated in \cite[Section~1]{ChenIssaMourrat2026}.
% Threshold: CIM (arXiv:2606.16636v1, p. 2) state max(p,1-p) > (3+sqrt 3)/6 for p delta_1 + (1-p) delta_{-1},
% which is (2p-1)^2 > 1/3. The formula was rechecked by hand and numerically (third pass).
% Fourth pass: the same statement is on p. 2 of arXiv:2606.16636v2 (29 Sep 2026); Propositions 2.2, 2.5,
% Section 2 and Appendix B have the same numbers in v1 and v2.
% Section 6.1 verified in arXiv:2004.01679v8; arXiv:2606.16636 cites Section 6 of the published version for the same point.
For spherical spins, the analogous initial condition is convex in the paths in
every field \cite[Lemma~2.9]{ChenMourrat2026spherical}.
The proof in \cite[Section~2]{ChenIssaMourrat2026} has the same structure as the
proof of \cref{prop:up-sqrt}: the Hessian of $\psi_0$ has nonpositive off-diagonal
entries, which is deduced from \cite[Theorem~2]{Panchenko2005question}, and
nonnegative row sums. For $h\ne0$, the row sums of the Hessian of $\Psi_h$ are
proportional to $T_i''(h)$ by \eqref{eq:up-rowsum}, and they may be negative. In
the proof of \cref{prop:up-sqrt}, the change of variables $v=s^2$ and
\cref{lem:up-floor} compensate for this.
\section{The Hopf--Lax formula and the Parisi functional}\label{sec:up-hopf}
Fix $\beta>0$ and $h\in\R$. For $t>0$ and $y\in\upC_K$ we define
\begin{equation}\label{eq:up-hopfU}
\mathsf U(t,y)=\sup_{y'\in\upC_K}\Big\{\psi_h(y')-\frac{\|y'-y\|_{2,w}^2}{\beta^2t}\Big\},
\qquad \mathsf U(0,y)=\psi_h(y).
\end{equation}
This is the Hopf--Lax formula for the equation
$\partial_t\mathsf U-(\beta^2/4)\|D\mathsf U\|_{2,w}^2=0$ with initial condition
$\psi_h$; see \cite[Theorem~3.8]{DominguezMourrat2024} for the corresponding
statement on $\R^{K+1}$ and \cite[Proposition~6.2]{ChenXia2022cones} for cones. We use
only the elementary properties in the next lemma.
\begin{lemma}\label{lem:up-hopf}
The following hold.
\begin{enumerate}[label=(\alph*)]
\item For $t>0$ and $y\in\upC_K$ the supremum in \eqref{eq:up-hopfU} is attained,
and every maximizer $y'$ satisfies
\begin{equation}\label{eq:up-hopfP}
\mathsf P=\frac{2(y'-y)}{\beta^2t}=D\psi_h(y')\in\upC_K\cap[0,1]^{K+1}.
\end{equation}
\item For $t\ge0$ and $y,\tilde y\in\upC_K$, we have
$|\mathsf U(t,y)-\mathsf U(t,\tilde y)|\le\|y-\tilde y\|_{1,w}$.
\item For $0\le t\le t'$ and $y\in\upC_K$, we have
$0\le\mathsf U(t',y)-\mathsf U(t,y)\le\beta^2(t'-t)/2$.
\end{enumerate}
In particular, $\mathsf U$ is continuous on $[0,\infty)\times\upC_K$.
\end{lemma}
\begin{proof}
(a) By \cref{lem:up-psi}, $\psi_h(y')\le\psi_h(y)+\|y'-y\|_{2,w}$, so the function
maximized in \eqref{eq:up-hopfU} is continuous on the closed set $\upC_K$ and
tends to $-\infty$ as $|y'|\to\infty$. Let $y'$ be a maximizer and
$\mathsf P=D\psi_h(y')$. For $y''\in\upC_K$, the segment from $y'$ to $y''$ lies in
$\upC_K$, and the right derivative at $y'$ of the maximized function in the
direction $y''-y'$ is nonpositive:
\[
\Big\langle\mathsf P-\frac{2(y'-y)}{\beta^2t},\,y''-y'\Big\rangle_w\le0 .
\]
Since $\mathsf P\in\upC_K$ by \cref{lem:up-psi}, we may take $y''=y+\beta^2t\mathsf P/2\in\upC_K$.
With this choice, the left side equals
$(\beta^2t/2)\|\mathsf P-2(y'-y)/(\beta^2t)\|_{2,w}^2$, which must vanish. This
proves \eqref{eq:up-hopfP}.
(b) For $t=0$ this is \cref{lem:up-psi}. Let $t>0$, let $y'$ be a maximizer
for $y$, and write $y'=y+\beta^2t\mathsf P/2$ as in (a). Since
$\mathsf P\in\upC_K$, the point $\tilde y+\beta^2t\mathsf P/2$ lies in $\upC_K$.
Taking this point in the supremum, we obtain
\begin{align*}
\mathsf U(t,\tilde y)&\ge\psi_h\Big(\tilde y+\frac{\beta^2t}2\mathsf P\Big)-\frac{\beta^2t}4\|\mathsf P\|_{2,w}^2\\
&\ge\psi_h(y')-\|y-\tilde y\|_{1,w}-\frac{\beta^2t}4\|\mathsf P\|_{2,w}^2
=\mathsf U(t,y)-\|y-\tilde y\|_{1,w}.
\end{align*}
Exchanging $y$ and $\tilde y$ proves (b).
(c) If $t>0$ and $y'$ is a maximizer at time $t$, then
$\mathsf U(t',y)\ge\psi_h(y')-\|y'-y\|_{2,w}^2/(\beta^2t')\ge\mathsf U(t,y)$; the case
$t=0$ is similar with $y'=y$. Let $y''=y+\beta^2t'\mathsf P''/2$ be a maximizer
at time $t'$, with $\mathsf P''\in\upC_K\cap[0,1]^{K+1}$. Since the point
$y+\beta^2t\mathsf P''/2$ lies in $\upC_K$, we have
\begin{align*}
\mathsf U(t',y)-\mathsf U(t,y)&\le\psi_h\Big(y+\frac{\beta^2t'}2\mathsf P''\Big)
-\psi_h\Big(y+\frac{\beta^2t}2\mathsf P''\Big)-\frac{\beta^2(t'-t)}4\|\mathsf P''\|^2_{2,w}\\
&\le\frac{\beta^2(t'-t)}2\|\mathsf P''\|_{1,w},
\end{align*}
which is at most $\beta^2(t'-t)/2$.
\end{proof}
The next lemma compares $\mathsf U(1,0)$ with the Parisi formula. Its identity
\eqref{eq:up-parisi-id} is a finite-dimensional form of the representation of the
Parisi formula by a Hopf--Lax formula, due to Mourrat \cite{Mourrat2022Wasserstein}
for $h=0$; see \cite[Theorem~6.7]{DominguezMourrat2024}.
% Mourrat2022Wasserstein, Theorem 1.1 and Proposition 4.1 (arXiv:1906.08471v2), assume (1.1): uniform measure on
% {-1,1}^N or on the sphere, so no external field; a general reference measure is only Conjecture 2.5 there.
% DominguezMourrat2024, Theorem 6.7 (arXiv:2311.08976), also uses the uniform measure, see (6.120). Fourth pass.
We use the Parisi equation \eqref{eq:pre-parisi-pde}, the Parisi functional
$\Par_{\beta,h}$, and the formula \eqref{eq:pre-parisi-formula} for $\PSK(\beta,h)$,
and we write
$\mu_r=\sum_{j=0}^Kw_j\delta_{r_j}$ for $r\in\upC_K\cap[0,1]^{K+1}$.
\begin{lemma}\label{lem:up-parisi}
Let $\beta>0$ and $h\in\R$. For every $r\in\upC_K\cap[0,1]^{K+1}$,
\begin{equation}\label{eq:up-parisi-id}
\psi_h\Big(\frac{\beta^2}2r\Big)-\frac{\beta^2}4\|r\|_{2,w}^2=\frac{\beta^2}4-\Par_{\beta,h}(\mu_r),
\end{equation}
and $\mathsf U(1,0)\ge\beta^2/4-\Par_{\beta,h}(\mu_r)$. Further, the map $\mu\mapsto\Par_{\beta,h}(\mu)$ is continuous for weak convergence of probability measures on $[0,1]$.
\end{lemma}
\begin{proof}
We first prove \eqref{eq:up-parisi-id}. Let $\Phi=\Phi_{\mu_r}$ solve the
Parisi equation \eqref{eq:pre-parisi-pde}. Since $\mu_r$ is atomic, $\Phi$ is
obtained by successive Cole--Hopf transforms \cite{AuffingerChen2015unique}:
on $[r_K,1]$ we have $\mu_r([0,t])=1$ and $\Phi(t,x)=\log2\cosh x+\beta^2(1-t)/2$;
on an interval $[r_j,r_{j+1})$ of positive length we have
$\mu_r([0,t])=\zeta_j$ and
$\Phi(r_j,x)=\zeta_j^{-1}\log\E\exp(\zeta_j\Phi(r_{j+1},x+\beta\sqrt{r_{j+1}-r_j}\,Z))$;
and $\Phi(0,x)=\E\Phi(r_0,x+\beta\sqrt{r_0}\,Z)$. Let $\phi_j$ be the recursion of
\cref{sec:up-transform} for the endpoints $v=\beta^2r$, so that
$\Delta_{j+1}=\beta^2(r_{j+1}-r_j)$. Since constants commute with the Cole--Hopf
transforms, we have $\Phi(r_j,\cdot)=\phi_j+\beta^2(1-r_K)/2$ for every $j$ by
downward induction, including the case of repeated endpoints. Then
$\Phi(0,h)=\E\phi_0(h+\beta\sqrt{r_0}\,Z)+\beta^2(1-r_K)/2$. If $X\sim\mu_r$,
then
\[
\int_0^1t\,\mu_r([0,t])\dd t=\E\int_X^1t\dd t=\frac{1-\|r\|_{2,w}^2}2 .
\]
We conclude that
\[
\Par_{\beta,h}(\mu_r)=\E\phi_0\big(h+\beta\sqrt{r_0}\,Z\big)+\frac{\beta^2}4-\frac{\beta^2}2r_K
+\frac{\beta^2}4\|r\|_{2,w}^2 .
\]
Since $\psi_h(\beta^2r/2)=\Psi_h(\beta^2r)=\beta^2r_K/2-\E\phi_0(h+\beta\sqrt{r_0}\,Z)$,
this is \eqref{eq:up-parisi-id}. Taking $y'=\beta^2r/2$
in \eqref{eq:up-hopfU} with $t=1$ and $y=0$, for which
$\|y'\|_{2,w}^2/\beta^2=\beta^2\|r\|^2_{2,w}/4$, we obtain the lower bound on $\mathsf U(1,0)$.
Finally, $\Phi_\mu(0,h)$ is continuous in $\mu$ for weak convergence by \ref{P:reg}, and so is
$\int_0^1t\mu([0,t])\dd t=(1-\int x^2\mu(\mathrm dx))/2$. This proves the continuity of $\Par_{\beta,h}$.
\end{proof}
% ---------- 04-upper.tex
% ---------------------------------------------------------------------
% Section 4: the upper bound on regular graphs.
% Owner: upper-bound agent. Label prefix: up-.
% ---------------------------------------------------------------------
\providecommand{\upC}{\mathcal C}
\providecommand{\upE}{\widehat{\mathbb E}}
\chapter{The upper bound on regular graphs}\label{sec:up-upper}
The goal of this section is to prove the following proposition, which states
that the pressure of the Edwards--Anderson model on any $D$-regular graph is
asymptotically at most the Parisi pressure.
\begin{proposition}\label{prop:upper}
For every $\beta\ge0$ and $h\in\R$,
\[
\limsup_{D\to\infty}\ \sup_{G\in\cG_D}\ p_G(\beta,h)\le\PSK(\beta,h).
\]
\end{proposition}
The supremum is over all finite simple $D$-regular graphs, with any number of
vertices; no assumption on the spectrum, the girth, or the symmetries of $G$ is
made. We now outline the proof. The strategy is that of Mourrat
\cite{Mourrat2021nonconvex,Mourrat2020vector}, who proved upper bounds for
mean-field models with nonconvex interactions by showing that an enriched free
energy is an approximate viscosity supersolution of a Hamilton--Jacobi equation,
with Ghirlanda--Guerra identities and synchronization enforced at contact points.
Fix a finite cascade as in \cref{sec:up-cascades}.
On the cone of $V$-indexed families of nondecreasing paths we consider an
enriched free energy in which every vertex receives a Gaussian field with the
covariance structure of a Ruelle cascade, and in which each first-level branch
of the cascade carries an independent copy of the couplings. At time $t$ this
free energy is at most $\beta^2t/4-p_G(\beta\sqrt t,h)$ at the origin, and
at time zero it is given by the scalar transform of \cref{sec:up-transform}.
Its time derivative differs from a quadratic Hamiltonian evaluated at its
gradient by $\beta^2(4n)^{-1}\sum_jw_j\tr(SC_j)$, where $C_j$ is a conditional
covariance matrix of the products $\sigma^1_x\sigma^2_x$. Since $S$ has negative
eigenvalues, this term has no sign. For every $\eta>0$, we use
$|\tr(SC_j)|\le\eta\tr C_j+(4\eta)^{-1}\tr(S^2C_j)$, with $\tr C_j\le n$.
We reduce $\tr(S^2C_j)$ to the conditional variances of the overlaps of
$O(n/D)$ rows of $S$ by a sparsification theorem of Cohen, Nelson, and Woodruff, and
control these variances by Mourrat's finitary form of Panchenko's
synchronization theorem, using Ghirlanda--Guerra identities produced by small
Gaussian perturbations. Because
the couplings are independent across first-level branches, the free energy
has fluctuations of order $1/n$ at every fixed parameter, which allows these
identities to be enforced at every point where a test function touches a
deterministic regularization of the free energy. A comparison principle on the
cone then bounds this regularization from
below by an explicit subsolution built from the Hopf--Lax formula of
\cref{sec:up-hopf}, and \cref{lem:up-parisi} compares the value of the
subsolution with the Parisi formula.
\Cref{sec:up-hj} introduces the Hamiltonians on the cone and proves the
comparison principle. \Cref{sec:up-sub} constructs the subsolution.
\Cref{sec:up-rpc} defines the enriched free energy and records its properties.
\Cref{sec:up-sync} states Mourrat's bound on conditional variances, \cref{sec:up-probes} introduces the Gaussian
perturbations, and \cref{sec:up-drift} bounds the drift term. \Cref{sec:up-proof}
completes the proof of \cref{prop:upper}.
Throughout this section we fix $D\ge1$, a graph $G=(V,E)\in\cG_D$ with $n=|V|$
vertices, and the notation of \eqref{eq:pre-model}. In particular $S=A_G/D$ is
symmetric, nonnegative, with zero diagonal, $S\mathbf1=\mathbf1$, $\|S\|\le1$, and
$\tr S^2=n/D$. We also fix a cascade as in \eqref{eq:up-masses} and
\eqref{eq:up-weights}. All constants are independent of $D$, $G$, and $n$ unless
stated otherwise. We write
\[
Z_G(\beta,h)=\sum_{\sigma\in\{-1,1\}^V}\exp\Big(\beta H_G(\sigma)+h\sum_x\sigma_x\Big),
\qquad p_G(\beta,h)=\frac1n\E\log Z_G(\beta,h).
\]
\section{Hamiltonians on the cone and comparison}\label{sec:up-hj}
Let $V_K=V\times\{0,\dots,K\}$. For $p\in\R^{V_K}$ we write $p_x=(p_{xj})_{j\le K}$
and $p_j=(p_{xj})_{x\in V}\in\R^V$. We use the inner product and norms
\[
\langle p,q\rangle_V=\frac1n\sum_{x\in V}\langle p_x,q_x\rangle_w,\qquad
\|p\|_1=\frac1n\sum_{x\in V}\|p_x\|_{1,w},\qquad \|p\|_2=\langle p,p\rangle_V^{1/2},
\]
so that $\|p\|_1\le\|p\|_2$, because the weights $w_j/n$ sum to one. The cone
is $\upC_K^V=\{q\in\R^{V_K}:q_x\in\upC_K\text{ for every }x\}$, and its dual cone for
$\langle\cdot,\cdot\rangle_V$ is $(\upC_K^*)^V$. We write $p\succeq_*p'$ if
$p_x\succeq_*p'_x$ for every $x$. For a function $u$ of $q\in\R^{V_K}$ we write
$D_qu$ for its gradient with respect to $\langle\cdot,\cdot\rangle_V$, that is,
$(D_qu)_{xj}=nw_j^{-1}\partial u/\partial q_{xj}$; for functions on $\upC_K^V$,
differentiability is understood as in \cref{sec:up-notation}.
\subsection*{Hamiltonians}
For $p\in\R^{V_K}$ we define
\begin{equation}\label{eq:up-hams}
\mathsf H_S(p)=\frac{\beta^2}{4n}\sum_{j=0}^Kw_j\,p_j\cdot Sp_j
\end{equation}
and
\begin{equation}\label{eq:up-Hext}
\mathsf H_{\rm ext}(p)=\inf\big\{\mathsf H_S(A):A\in\upC_K^V,\ A\succeq_*p\big\}.
\end{equation}
This is the monotone modification constructed in the proof of
\cite[Lemma~2.4]{ChenXia2022cones}, applied to $\mathsf H_S$. For $z\in\R^{K+1}$ and $0\le\ell\le K+1$ we write
$\tau_\ell(z)=\sum_{j\ge\ell}w_jz_j$, $\bar\tau_\ell(z)=\max_{\ell\le\ell'\le K+1}\tau_{\ell'}(z)$, and $W_\ell=\tau_\ell(\mathbf 1)$. Then $\tau_{K+1}(z)=0$, so that $\bar\tau_\ell(z)\ge0$, and
$z\succeq_*z'$ if and only if $\tau_\ell(z)\ge\tau_\ell(z')$ for every $\ell$, by
\eqref{eq:up-dualcone}.
\begin{lemma}\label{lem:up-ext}
The following hold.
\begin{enumerate}[label=(\alph*)]
\item If $A,B\in\upC_K^V$ and $B\succeq_*A$, then $\mathsf H_S(B)\ge\mathsf H_S(A)$.
\item Let $p\in\R^{V_K}$. For each $x$, let $\mathscr C_x$ be the least
concave majorant on $[0,1]$ of the points $(W_\ell,\bar\tau_\ell(p_x))$,
$0\le\ell\le K+1$, and let $A_{xj}=(\mathscr C_x(W_j)-\mathscr C_x(W_{j+1}))/w_j$.
Then $A\in\upC_K^V$, $A\succeq_*p$, every $B\in\upC_K^V$ with $B\succeq_*p$ satisfies $B\succeq_*A$, and
$\mathsf H_{\rm ext}(p)=\mathsf H_S(A)$. Further, $0\le w_KA_{xK}\le\|p_x\|_{1,w}$ for every $x$. If $p\in(0,\infty)^{V_K}$, then $A\in(0,\infty)^{V_K}$, and every $j_0\le K$ belongs to a
block $\{i,\dots,i'-1\}\ni j_0$ on which $A_x$ is constant, equal to the weighted
mean of $p_x$ over the block, and such that
$\sum_{j=\ell}^{i'-1}w_jp_{xj}\le A_{xj_0}\sum_{j=\ell}^{i'-1}w_j$ for every
$i\le\ell0$ and $p,p'\in\R^{V_K}$ with $\|p\|_2,\|p'\|_2\le R$, we have
$|\mathsf H_{\rm ext}(p)-\mathsf H_{\rm ext}(p')|\le\beta^2Rw_K^{-1}\|p-p'\|_2$.
\end{enumerate}
\end{lemma}
\begin{proof}
(a) For $P\in\upC_K^V$, we have
$(D\mathsf H_S(P))_{xj}=(\beta^2/2)\sum_yS_{xy}P_{yj}$,
which is nonnegative and nondecreasing in $j$, because $S$ has nonnegative entries and $P_{yj}$ is nonnegative and
nondecreasing in $j$. In other words, $D\mathsf H_S(P)\in\upC_K^V$. Since the
segment from $A$ to $B$ lies in $\upC_K^V$ and $B-A\in(\upC_K^*)^V$, we have
$\mathsf H_S(B)-\mathsf H_S(A)=\int_0^1\langle D\mathsf H_S(A+\theta(B-A)),B-A\rangle_V\dd\theta\ge0$.
(b) Fix $x$ and let $a_\ell=\bar\tau_\ell(p_x)$. As $\ell$ increases from $0$ to $K+1$, the abscissae $W_\ell$ decrease strictly from $1$ to $0$, and $a_\ell$ is nonincreasing, with $a_{K+1}=0$. The least
concave majorant $\mathscr C_x$ is the upper boundary of the convex hull of the points. It is
piecewise linear, its vertices are among the points, $\mathscr C_x(0)=a_{K+1}=0$, and $\mathscr C_x(1)=a_0$. Since the maximum of $\mathscr C_x$ is attained at a vertex and $a_0=\max_\ell a_\ell$, it is attained at $1$, and the concave function $\mathscr C_x$ is nondecreasing. Its slopes are then nonnegative and
nonincreasing in $W$, which gives $A_x\in\upC_K$. Since
$\mathscr C_x(0)=0$, we have $\tau_\ell(A_x)=\mathscr C_x(W_\ell)\ge a_\ell\ge\tau_\ell(p_x)$,
that is, $A_x\succeq_*p_x$. Moreover $w_KA_{xK}=\mathscr C_x(W_K)\in[0,a_0]$, and $a_0\le\max_\ell|\tau_\ell(p_x)|\le\|p_x\|_{1,w}$. Let $B_x\in\upC_K$ satisfy $B_x\succeq_*p_x$. The
piecewise linear interpolation $\Phi$ of the points $(W_\ell,\tau_\ell(B_x))$ is concave and nondecreasing,
because its slopes $B_{xj}$ are nonnegative and nondecreasing in $j$. For $\ell\le\ell'$ we have $\Phi(W_\ell)\ge\Phi(W_{\ell'})=\tau_{\ell'}(B_x)\ge\tau_{\ell'}(p_x)$, so that $\Phi(W_\ell)\ge a_\ell$. By the minimality of the least concave majorant, $\Phi\ge\mathscr C_x$, and $B_x\succeq_*A_x$.
By (a), $\mathsf H_S(B)\ge\mathsf H_S(A)$ for every $B$ admissible in \eqref{eq:up-Hext}, and $A$ is admissible, which shows that
$\mathsf H_{\rm ext}(p)=\mathsf H_S(A)$.
Let now $p\in(0,\infty)^{V_K}$. Then $\tau_\ell(p_x)$ is strictly decreasing in $\ell$, and $a_\ell=\tau_\ell(p_x)$. Let $i\le j_00$ for every $j$, so that $b\in(0,\infty)^{V_K}$, and
let $A$ be as in \cref{lem:up-ext}(b) for $p=b$. Fix $x$ and a block $\{i,\dots,i'-1\}$
as there, on which $A_x$ equals a constant $a$. For $i\le\ell0$, the inequality between the arithmetic and geometric means of $a^{-1}$ and $a'^{-1}$ gives $aa'\ge((a^{-1}+a'^{-1})/2)^{-2}$. Using this and Jensen's
inequality for the convex function $a\mapsto a^{-2}$ on $(0,\infty)$, we obtain
\[
\frac1n\sum_{x,y}S_{xy}A_{xj}A_{yj}\ge\frac1n\sum_{x,y}S_{xy}\Big(\frac{A_{xj}^{-1}+A_{yj}^{-1}}2\Big)^{-2}
\ge\mathsf v_j^{-2}.
\]
The tangent line of the convex function $a\mapsto a^{-2}$ at $\mathsf P_j^{-1}$ gives
$\mathsf v_j^{-2}\ge\mathsf P_j^2-2\mathsf P_j^3(\mathsf v_j-\mathsf P_j^{-1})$. Since the vector
$(\mathsf P_j^3)_j$ belongs to $\upC_K$ and $\mathsf P^{-1}-\mathsf v\in\upC_K^*$, we have
\[
\sum_jw_j\mathsf v_j^{-2}\ge\sum_jw_j\mathsf P_j^2+2\big\langle(\mathsf P^3_j)_j,\mathsf P^{-1}-\mathsf v\big\rangle_w
\ge\sum_jw_j\mathsf P_j^2 .
\]
Combining the last two displays with the identity $\mathsf H_{\rm ext}(b)=\mathsf H_S(A)$ from \cref{lem:up-ext}(b), we obtain the claim in
this case. In general, for $\theta>0$ let $\mathsf P^{(\theta)}=\mathsf P+\theta\mathbf 1$,
which belongs to $\upC_K\cap(0,\infty)^{K+1}$, and let $b^{(\theta)}$ be the corresponding
vector. Then $b^{(\theta)}\to b$ as $\theta\to0$, and the continuity of
$\mathsf H_{\rm ext}$ from \cref{lem:up-ext}(c) proves the claim.
\end{proof}
\subsection*{Viscosity solutions and comparison}
Viscosity solutions were introduced by Crandall and Lions
\cite{CrandallLions1983}; see also \cite{CrandallEvansLions1984}. The following
definition, in which test points on the boundary of the cone are allowed, follows
\cite[Definition~1.1]{ChenXia2022cones}.
\begin{definition}\label{def:up-viscosity}
Let $\mathsf H\colon\R^{V_K}\to\R$ be continuous, let $c\in\R$, and let
$u\colon[0,2]\times\upC_K^V\to\R$ be continuous. We say that $u$ is a
\emph{subsolution} of $\partial_tu-\mathsf H(D_qu)\le c$ if, for every
$(t_0,q_0)\in(0,2)\times\upC_K^V$ and every function $\varphi$ that is continuously
differentiable in a neighborhood of $(t_0,q_0)$ in $\R\times\R^{V_K}$ and such that
$u-\varphi$ has a local maximum at $(t_0,q_0)$ relative to $(0,2)\times\upC_K^V$, we have
$\partial_t\varphi(t_0,q_0)-\mathsf H(D_q\varphi(t_0,q_0))\le c$. We say that $u$ is a
\emph{supersolution} of $\partial_tu-\mathsf H(D_qu)\ge c$ if the same holds with
local minima and the reverse inequality.
\end{definition}
The tests are relative to the closed cone: points on its boundary are included,
and no boundary condition is imposed. We use the following elementary fact
repeatedly. Let $u$ be differentiable in $t$ at $t_0\in(0,2)$ and differentiable
on $\upC_K^V$ in $q$, and let $u-\varphi$ have a local minimum at $(t_0,q_0)$
relative to $(0,2)\times\upC_K^V$. Since the point $q_0+\epsilon r$ lies in
$\upC_K^V$ for $r\in\upC_K^V$ and $\epsilon\ge0$, we have
\begin{equation}\label{eq:up-touch}
\partial_tu(t_0,q_0)=\partial_t\varphi(t_0,q_0),\qquad
D_qu(t_0,q_0)-D_q\varphi(t_0,q_0)\in(\upC_K^*)^V .
\end{equation}
\begin{lemma}\label{lem:up-comparison}
Let $\mathsf H\colon\R^{V_K}\to\R$ be such that for every $R>0$ there exists $L_R$ with
$|\mathsf H(p)-\mathsf H(p')|\le L_R\|p-p'\|_2$ whenever $\|p\|_2,\|p'\|_2\le R$. Let $\tau\ge0$, $c_0\in\R$,
and let $\mathsf V,\mathsf E\colon[0,2]\times\upC_K^V\to\R$ be continuous functions
with the following properties, for constants $L_t,L_q$:
\begin{enumerate}[label=(\roman*)]
\item $\mathsf V$ is a subsolution of $\partial_tu-\mathsf H(D_qu)\le0$, and $\mathsf E$ is
a supersolution of $\partial_tu-\mathsf H(D_qu)\ge-\tau$;
\item $|\mathsf E(t,q)-\mathsf E(t',q')|\le L_t|t-t'|+L_q\|q-q'\|_1$ and
$|\mathsf V(t,q)-\mathsf V(t',q)|\le L_t|t-t'|$;
\item $\mathsf V(0,q)\le\mathsf E(0,q)+c_0$ for every $q\in\upC_K^V$.
\end{enumerate}
Then $\mathsf V(t,q)\le\mathsf E(t,q)+c_0+\tau t$ for every $t\in[0,2)$ and $q\in\upC_K^V$.
\end{lemma}
The proof is the doubling-of-variables argument of
\cite{CrandallLions1983,CrandallEvansLions1984}; see
\cite[Theorem~3.5]{DominguezMourrat2024} and
\cite[Proposition~3.1]{ChenXia2022cones} for comparison principles on $\R^d$ and on
cones. We include the proof, because the Hamiltonian is only locally Lipschitz
and $\mathsf V$ is not assumed to be Lipschitz in $q$.
\begin{proof}
Suppose that $\mathsf V(\bar t,\bar q)-\mathsf E(\bar t,\bar q)-c_0-\tau\bar t=4\theta>0$
for some $\bar t<2$; by (iii), $\bar t>0$. Let $L=\max\{1,(L_t^2+L_q^2)^{1/2}\}$. Fix $T'\in(\bar t,2)$, then $\rho>0$ with
$\rho\bar t+\rho/(T'-\bar t)\le\theta$, and then $\gamma\in(0,1]$ with
$\gamma(1+\|\bar q\|_2^2)^{1/2}\le\theta$ and $\gamma L_{2L+1}<\rho$. For $b>0$
and $(t,q,s,r)\in[0,T')\times\upC_K^V\times[0,2]\times\upC_K^V$ let
\[
\Phi_b(t,q,s,r)=\mathsf V(t,q)-\mathsf E(s,r)-c_0-(\tau+\rho)t-\frac{\rho}{T'-t}
-\gamma\langle q\rangle-\frac{|t-s|^2+\|q-r\|_2^2}{2b},
\]
where $\langle q\rangle=(1+\|q\|_2^2)^{1/2}$. By (ii) and (iii), we have
$\mathsf V(t,q)-\mathsf E(s,r)\le c_0+2L_tt+L_t|t-s|+L_q\|q-r\|_2$, and $\Phi_b$ is
continuous and tends to $-\infty$ as $\|q\|_2\to\infty$, as $\|q-r\|_2\to\infty$, or as
$t\to T'$. Since the dimension is finite, $\Phi_b$ attains its maximum at some
point $(t_b,q_b,s_b,r_b)$, and $\Phi_b(t_b,q_b,s_b,r_b)\ge\Phi_b(\bar t,\bar q,\bar t,\bar q)\ge2\theta$.
Comparing with $\Phi_b(t_b,q_b,t_b,q_b)$ and using (ii), we obtain, with
$\mathsf X=(|t_b-s_b|^2+\|q_b-r_b\|_2^2)^{1/2}$,
\[
\frac{\mathsf X^2}{2b}\le\mathsf E(t_b,q_b)-\mathsf E(s_b,r_b)\le L\mathsf X,
\qquad\text{so}\qquad \mathsf X\le2bL .
\]
If $t_b=0$, then by (iii) and (ii),
$\Phi_b\le\mathsf E(0,q_b)-\mathsf E(s_b,r_b)\le L\mathsf X\le2bL^2$. If $s_b=0$, then
$\Phi_b\le\mathsf V(t_b,q_b)-\mathsf V(0,q_b)+\mathsf E(0,q_b)-\mathsf E(0,r_b)\le2L\mathsf X\le4bL^2$.
Both contradict $\Phi_b\ge2\theta$ if $b<\theta/(2L^2)$. Also $s_b\le t_b+2bL<2$ for
small $b$. We conclude that $t_b\in(0,T')$ and $s_b\in(0,2)$ for small $b$.
The function $(t,q)\mapsto\mathsf V(t,q)-\varphi_1(t,q)$, with
\[
\varphi_1(t,q)=(\tau+\rho)t+\frac{\rho}{T'-t}+\gamma\langle q\rangle+\frac{|t-s_b|^2+\|q-r_b\|_2^2}{2b},
\]
has a local maximum at $(t_b,q_b)$ relative to $(0,2)\times\upC_K^V$, and the
function $(s,r)\mapsto\mathsf E(s,r)-\varphi_2(s,r)$, with
$\varphi_2(s,r)=-(|t_b-s|^2+\|q_b-r\|_2^2)/(2b)$, has a local minimum at
$(s_b,r_b)$. Let $\mathsf p=(q_b-r_b)/b$. Since $D_q\langle q\rangle=q/\langle q\rangle$, by (i) we have
\[
\tau+\rho+\frac{\rho}{(T'-t_b)^2}+\frac{t_b-s_b}b-\mathsf H\Big(\mathsf p+\frac{\gamma q_b}{\langle q_b\rangle}\Big)\le0
\le\frac{t_b-s_b}b-\mathsf H(\mathsf p)+\tau .
\]
We have $\|\mathsf p\|_2\le\mathsf X/b\le2L$ and $\|\gamma q_b/\langle q_b\rangle\|_2\le\gamma\le1$. Rearranging, we obtain $\rho\le\mathsf H(\mathsf p+\gamma q_b/\langle q_b\rangle)-\mathsf H(\mathsf p)\le L_{2L+1}\gamma<\rho$,
a contradiction.
\end{proof}
\section{The subsolution}\label{sec:up-sub}
In this subsection $\beta>0$ and $h\in\R$ are fixed, $\psi_h$ is the scalar
transform \eqref{eq:up-Psi}, and $\mathsf U$ is the function \eqref{eq:up-hopfU}.
We evaluate $\mathsf U$ at an average of the paths $q_x$. A similar construction
appears in the Hamilton--Jacobi comparison argument for balanced multispecies
models of Chen, Issa, and Mourrat \cite[Appendix~B]{ChenIssaMourrat2026}, where the
weighted mean of the paths is used and the initial conditions are compared
through the convexity of $\psi_0$. They note that this argument is close in
spirit to the Hamilton--Jacobi comparison for permutation-invariant vector spin
glasses in \cite[Section~7]{Issa2024permutation}. Here we average the square roots of the
paths, which \cref{prop:up-sqrt} allows for every $h$, and the subsolution
property follows from \cref{lem:up-harmonic}.
For $\delta\in(0,1)$ and $q\in\upC_K^V$ we set
\begin{equation}\label{eq:up-gq}
\mathsf r_{xj}(q)=(q_{xj}+\delta)^{1/2},\qquad
M_j(q)=\frac1n\sum_x\mathsf r_{xj}(q),\qquad
\mathsf g_j(q)=M_j(q)^2-\delta .
\end{equation}
Since $\mathsf r_{xj}$ is nondecreasing in $j$ and $M_j\ge\sqrt\delta$, we have
$\mathsf g(q)\in\upC_K$. For $q\in\upC_K^V$ we write $\tilde q$ for the element of $\upC_K^V$ with $\tilde q_{x0}=0$ and $\tilde q_{xj}=q_{xj}$ for $1\le j\le K$. We define
\begin{equation}\label{eq:up-Vdelta}
V_\delta(t,q)=\mathsf U(t,\mathsf g(\tilde q))-\frac{\beta^2\zeta_0t}4 .
\end{equation}
\begin{lemma}\label{lem:up-subsol}
The function $V_\delta$ is continuous on $[0,2]\times\upC_K^V$, and the following
hold.
\begin{enumerate}[label=(\alph*)]
\item $|V_\delta(t,q)-V_\delta(t',q)|\le\beta^2|t-t'|$.
\item $V_\delta$ is a subsolution of $\partial_tu-\mathsf H_{\rm ext}(D_qu)\le0$.
\item $V_\delta(0,q)\le n^{-1}\sum_x\psi_h(\tilde q_x)+\delta$ for every $q\in\upC_K^V$.
\item $V_\delta(1,0)=\mathsf U(1,0)-\beta^2\zeta_0/4$.
\end{enumerate}
\end{lemma}
\begin{proof}
Continuity and (a) follow from \cref{lem:up-hopf}(b),(c), and (d) holds because
$\mathsf g(0)=0$.
(b) Let $\varphi$ be continuously
differentiable near $(t_0,q_0)\in(0,2)\times\upC_K^V$ and let $V_\delta-\varphi$ have a local maximum at $(t_0,q_0)$; we may
assume that its value there is zero. Let $y'$ be a maximizer in
\eqref{eq:up-hopfU} for $\mathsf U(t_0,\mathsf g(\tilde q_0))$, and let
$\mathsf P=2(y'-\mathsf g(\tilde q_0))/(\beta^2t_0)\in\upC_K\cap[0,1]^{K+1}$ be as in
\cref{lem:up-hopf}(a). The function
\[
\mathsf B(t,q)=\psi_h(y')-\frac{\|y'-\mathsf g(\tilde q)\|_{2,w}^2}{\beta^2t}-\frac{\beta^2\zeta_0t}4
\]
is smooth near $(t_0,q_0)$, satisfies $\mathsf B\le V_\delta$ on $(0,2)\times\upC_K^V$
and $\mathsf B(t_0,q_0)=V_\delta(t_0,q_0)$. Then $\mathsf B-\varphi\le V_\delta-\varphi\le0$
near $(t_0,q_0)$, with equality at that point. Applying \eqref{eq:up-touch} to
$\varphi-\mathsf B$, we obtain $\partial_t\varphi=\partial_t\mathsf B$ and
$D_q\varphi\succeq_*D_q\mathsf B$ at $(t_0,q_0)$. Since $\tilde q$ does not depend on the coordinates $q_{x0}$, a direct computation at $(t_0,q_0)$
gives $(D_q\mathsf B)_{x0}=0$ and
\begin{gather*}
\partial_t\mathsf B=\frac{\|y'-\mathsf g(\tilde q_0)\|_{2,w}^2}{\beta^2t_0^2}-\frac{\beta^2\zeta_0}4=\frac{\beta^2}4\big(\|\mathsf P\|_{2,w}^2-\zeta_0\big),\\
(D_q\mathsf B)_{xj}=\frac{\mathsf P_jM_j(\tilde q_0)}{\mathsf r_{xj}(\tilde q_0)}\quad(1\le j\le K).
\end{gather*}
Let $\mathsf P'=(0,\mathsf P_1,\dots,\mathsf P_K)\in\upC_K$. Then $D_q\mathsf B$ is the vector $b$ of \cref{lem:up-harmonic} for $\mathsf P'$ and $\mathsf r=\mathsf r(\tilde q_0)$, and that lemma gives
$\mathsf H_{\rm ext}(D_q\mathsf B)\ge(\beta^2/4)\sum_{j\ge1}w_j\mathsf P_j^2$. Since $w_0=\zeta_0$ and $\mathsf P_0\le1$, we have $\partial_t\mathsf B\le(\beta^2/4)\sum_{j\ge1}w_j\mathsf P_j^2$. By \cref{lem:up-ext}(c), we conclude that
$\partial_t\varphi=\partial_t\mathsf B\le\mathsf H_{\rm ext}(D_q\mathsf B)\le\mathsf H_{\rm ext}(D_q\varphi)$.
(c) We have $V_\delta(0,q)=\psi_h(\mathsf g(\tilde q))$. The vectors $\mathsf g(\tilde q)$ and
$M(\tilde q)^2=(M_j(\tilde q)^2)_j$ belong to $\upC_K$, with $\mathsf g(\tilde q)\le M(\tilde q)^2$
coordinatewise, and \cref{lem:up-psi} implies that $\psi_h(\mathsf g(\tilde q))\le\psi_h(M(\tilde q)^2)$.
Since the vectors $\mathsf r_x(\tilde q)$ belong to the convex set
$\{s:0\le s_0\le\dots\le s_K\}$ and $M(\tilde q)$ is their average, we have
$\psi_h(M(\tilde q)^2)\le n^{-1}\sum_x\psi_h(\tilde q_x+\delta\mathbf 1)$ by
\cref{prop:up-sqrt} and Jensen's inequality. Finally
$\psi_h(\tilde q_x+\delta\mathbf 1)\le\psi_h(\tilde q_x)+\delta$ by \cref{lem:up-psi}, since
$\|\delta\mathbf 1\|_{1,w}=\delta$.
\end{proof}
\section{The enriched free energy}\label{sec:up-rpc}
\subsection*{Ruelle cascades}
Let $\mathbb A=\bigcup_{k=0}^K\N^k$ be the rooted tree of depth $K$ with infinite
degree, with root $\emptyset$ and leaves $\N^K$. For $\alpha\in\N^K$ and $k\le K$ let
$\alpha|_k$ be the ancestor of $\alpha$ at depth $k$, and for leaves
$\alpha,\alpha'$ let $\alpha\wedge\alpha'=\max\{k:\alpha|_k=\alpha'|_k\}$. For each
nonleaf vertex $\gamma\in\N^k$, $k0$, set $X_K=X$ and
$X_{k-1}=\zeta_{k-1}^{-1}\log\E_{\omega_k}\exp(\zeta_{k-1}X_k)$ for $1\le k\le K$, where
$\E_{\omega_k}$ integrates a variable $\omega_k\sim\mathsf P_k$. Then
\[
\E\log\sum_{\alpha\in\N^K}v_\alpha\exp X\big(\omega_{\alpha|_1},\dots,\omega_{\alpha|_K}\big)=X_0 .
\]
Moreover, if $\langle\cdot\rangle$ denotes the random probability measure on $\N^K$
with weights proportional to $v_\alpha\exp X(\omega_{\alpha|_1},\dots,\omega_{\alpha|_K})$,
and $\alpha^1,\alpha^2$ are independent samples from it, then
\begin{equation}\label{eq:up-overlaplaw}
\E\big\langle\mathbf 1\{\alpha^1\wedge\alpha^2=j\}\big\rangle=w_j\qquad(0\le j\le K).
\end{equation}
\end{theorem}
The first statement is \cite[Theorem~5.25]{DominguezMourrat2024}, applied with a
trivial root variable; see also \cite[Theorem~2.9]{PanchenkoBook}. % Theorem 2.9 is also cited for this formula in arXiv:1906.08471 and arXiv:2004.01679.
For $X=0$, \eqref{eq:up-overlaplaw} is \cite[(5.93) in Theorem~5.28]{DominguezMourrat2024}.
For general $X$, it is proved in \cite[Section~6.4, before Lemma~6.2]{DominguezMourrat2024}
for the enriched free energy considered there, by observing that the proof of
\cite[Theorem~5.28]{DominguezMourrat2024}, which rests on the invariance property
\cite[Lemma~5.27]{DominguezMourrat2024} of Bolthausen and Sznitman
\cite{BolthausenSznitman1998}, remains valid when each $w_\alpha$ is multiplied by a
function of variables attached to the vertices of the path of $\alpha$ that are
independent of $(w_\alpha)$ and whose joint law is invariant under bijections of
$\N^K$ preserving $\wedge$. This applies verbatim to the weights
$w_\alpha\exp X(\omega_{\alpha|_1},\dots,\omega_{\alpha|_K})$.
We also use Gaussian integration by parts for Gibbs measures
\cite[Theorem~4.6]{DominguezMourrat2024}. Let $\Sigma$ be a countable set, $\mu$ a
finite measure on $\Sigma$, and $(y(\varsigma))_{\varsigma\in\Sigma}$,
$(\mathsf z(\varsigma))_{\varsigma\in\Sigma}$ centered jointly Gaussian processes
with bounded covariances. Let $\langle\cdot\rangle$ be the Gibbs measure with
density proportional to $e^{y}$ with respect to $\mu$, and
$\mathsf C(\varsigma,\varsigma')=\E\mathsf z(\varsigma)y(\varsigma')$. Then for all
$r\ge1$ and all bounded $f\colon\Sigma^r\to\R$,
\begin{equation}\label{eq:up-gibbsibp}
\E\big\langle f(\varsigma^1,\dots,\varsigma^r)\,\mathsf z(\varsigma^1)\big\rangle
=\E\Big\langle f(\varsigma^1,\dots,\varsigma^r)\Big(\sum_{\ell=1}^r\mathsf C(\varsigma^1,\varsigma^\ell)-r\,\mathsf C(\varsigma^1,\varsigma^{r+1})\Big)\Big\rangle .
\end{equation}
Finally, we use the following invariance of Poisson point processes
\cite[Theorem~5.19 and Proposition~5.18]{DominguezMourrat2024}. Let $(u_i)$ be the
decreasing enumeration of a Poisson point process with intensity
$\zeta x^{-1-\zeta}\dd x$, $\zeta\in(0,1)$. Then $\E(\sum_iu_i)^a<\infty$ for
$00$, differentiable on $\upC_K^V$ in
$q$, and infinitely differentiable in $y$, with continuous derivatives. For every bounded
function $f$ of finitely many replicas, $\E\langle f\rangle$ is a continuous
function of $(t,q,y)$.
\item For $t>0$, $\partial_tF=\beta^2(4n)^{-1}\E\langle u^*\cdot Su^*\rangle$; in
particular $|\partial_tF|\le\beta^2/4$.
\item $D_qF=\mathsf p$, where $\mathsf p_{x0}=0$ and
$\mathsf p_{xj}=\E\langle u^*_x\,|\,\vartheta_{12}=z_j\rangle$ for $1\le j\le K$.
Moreover $\mathsf p\in\upC_K^V\cap[0,1]^{V_K}$, and consequently
$|F(t,q,y)-F(t,q',y)|\le\|q-q'\|_1$.
\item $F$ is concave in $y$, and
$\partial_{y_{x\iota}}F=-n^{-1}\mathsf s\E\langle \mathsf G_{x,\iota}\rangle=-n^{-1}\mathsf s^2y_{x\iota}\big(1-\E\langle\mathsf k_{x\iota,12}\rangle\big)$.
Moreover $|F(t,q,y)-F_0(t,q)|\le r_{\mathsf s}$, where
$r_{\mathsf s}=9|\mathcal S||I|\mathsf s^2/(4n)$.
\item We have
\[
F_0(t,0)=\frac{\beta^2t}4-\frac1{n\zeta_0}\log\E Z_G(\beta\sqrt t,h)^{\zeta_0}\le\frac{\beta^2t}4-p_G(\beta\sqrt t,h).
\]
\item We have $F_0(0,q)=n^{-1}\sum_x\psi_h(\tilde q_x)$, with $\tilde q$ as in \cref{sec:up-sub}.
\end{enumerate}
\end{lemma}
\begin{proof}
(a) We write every Gaussian variable in \eqref{eq:up-enrichedH} as an amplitude
times a fixed standard Gaussian variable; the amplitudes are $\beta\sqrt t$, the
numbers $(2q_{x1})^{1/2}$ and $(2(q_{xk}-q_{x,k-1}))^{1/2}$, and $\mathsf sy_{x\iota}$
times fixed coefficients. By \cref{thm:up-rpc} with
$X=\log\sum_\sigma\exp\mathscr H(\alpha,\sigma)$, we have $F=-X_0/n+\beta^2t/4$,
where $X_0$ is obtained from $X$ by $K$ successive integrations as in
\eqref{eq:up-rec}. By \eqref{eq:up-heat} and differentiation under the integral
sign, $X_0$ is a smooth function of $\beta\sqrt t$, of $y$, and of the variances
$2q_{x1}$, $2(q_{xk}-q_{x,k-1})$ in $[0,\infty)$. This gives the differentiability
of $F$ and the continuity of its derivatives. For the Gibbs averages, fix a
compact set $\mathsf K$ of parameters. The variables
$\bar Z_\alpha=\sup_{\mathsf K}\sum_\sigma\exp\mathscr H(\alpha,\sigma)$ have the same law
for all $\alpha$, have finite mean, and are independent of $(v_\alpha)$, so
$\E\sum_\alpha v_\alpha\bar Z_\alpha=\E\bar Z_\alpha<\infty$. Then, almost surely,
the series defining $\langle\cdot\rangle$ converge uniformly on $\mathsf K$, the
Gibbs weights are continuous on $\mathsf K$, and so is $\langle f\rangle$; by dominated
convergence, $\E\langle f\rangle$ is continuous as well. For a fixed amplitude $a$, multiplying the exponential envelope by
$|\partial_a\mathscr H|$ or $|\partial_a\mathscr H|^2$ preserves integrability,
by Gaussian exponential moments. The same domination therefore gives almost-sure
twice differentiability in $a$. The first derivative of $\log Z$ is
$\langle\partial_a\mathscr H\rangle$. Since
$\mathscr H+\sum_xq_{xK}$ is affine in each amplitude, the second derivative of
$\log Z+\sum_xq_{xK}$ is
$\langle(\partial_a\mathscr H-\langle\partial_a\mathscr H\rangle)^2\rangle$.
The random function $\log Z+\sum_xq_{xK}$ is convex in each amplitude, as a
logarithm of a sum of exponentials of affine functions. Its difference
quotients are monotone, and $\E|\log Z|<\infty$ by \cref{lem:up-fluct} below.
Subtracting the deterministic compensator, we conclude that the derivative
of $F$ in each amplitude is the expectation of the derivative of
$\widetilde F$. We now compute these expectations by \eqref{eq:up-gibbsibp},
applied conditionally on $(v_\alpha)$ with $\Sigma=\N^K\times\{-1,1\}^V$.
(b) The process $H^{(\alpha_1)}(\sigma)$ has covariance
$\mathbf 1\{\alpha_1=\alpha'_1\}(\sigma\odot\sigma')\cdot S(\sigma\odot\sigma')/2$,
which equals $n/2$ on the diagonal. By \eqref{eq:up-gibbsibp} with $r=1$,
\[
\partial_t\E\log Z=\frac\beta{2\sqrt t}\E\big\langle H^{(\alpha_1)}(\sigma)\big\rangle
=\frac{\beta^2}2\Big(\frac n2-\frac12\E\langle u^*\cdot Su^*\rangle\Big),
\]
which gives the formula. Since $\|S\|\le1$ and $|u^*_x|\le1$, we have
$|u^*\cdot Su^*|\le n$.
(c) Suppose first that all the increments $q_{x1}$, $q_{xk}-q_{x,k-1}$ are
positive. For $1\le j\le K$, by \eqref{eq:up-gibbsibp} the amplitude derivatives
give, exactly as in \cite[proof of Lemma~6.2]{DominguezMourrat2024},
\[
\begin{aligned}
\partial_{q_{xj}}\E\log Z&=\E\big\langle1-\mathbf 1\{\alpha^1\wedge\alpha^2\ge j\}u_x\big\rangle\\
&\qquad-\E\big\langle1-\mathbf 1\{\alpha^1\wedge\alpha^2\ge j+1\}u_x\big\rangle\mathbf 1\{j0$, the second moment of $\log\mathsf S$ is bounded by a constant depending only on $\zeta_0$, and so is $\Var(\log Z)$. The bound on $\E|\widetilde F-F|$ follows from the Cauchy--Schwarz inequality, since $\widetilde F-F=-n^{-1}(\log Z-\E\log Z)$.
\end{proof}
\section{Synchronization}\label{sec:up-sync}
We use the following bound of Mourrat on conditional variances, which is
\cite[Proposition~5.5]{Mourrat2021nonconvex}, with the hypothesis of
\cite[Theorem~5.3]{Mourrat2021nonconvex}. It is a finitary form of the
synchronization theorem of Panchenko \cite[Theorem~4]{Panchenko2015multispecies},
which was extended to Potts and vector spins in
\cite{Panchenko2018Potts,Panchenko2018vector}. Its proof uses Panchenko's
ultrametricity theorem \cite{Panchenko2013ultrametricity} for measures that
satisfy the Ghirlanda--Guerra identities \cite{GhirlandaGuerra1998}.
Let $(\mathsf c_i)_{i\ge1}$ be a fixed enumeration of $\mathbb Q\cap[0,1]$.
% Theorem 5.3 and Proposition 5.5 verified in arXiv:2004.01679v8. The Dominguez--Mourrat book
% (arXiv:2311.08976, Section 6.4) cites Proposition 5.5 of the published version in Probab. Math. Phys.
\begin{proposition}\label{prop:up-sync}
For every $\varepsilon>0$ there exists $\delta>0$ such that the following holds.
Let $\mathfrak G$ be a random probability measure supported on the Cartesian
product of the unit balls of two Hilbert spaces. Let $\langle\cdot\rangle$ be the
expectation associated with $\mathfrak G^{\otimes\N}$, with canonical random
variables $(\sigma^\ell_1,\sigma^\ell_2)_{\ell\ge1}$, and let $\E$ be the
expectation with respect to the randomness of $\mathfrak G$. For $a\in\{1,2\}$
and $\ell,\ell',r\ge1$, let $\mathsf O_a^{\ell\ell'}=\sigma_a^\ell\cdot\sigma_a^{\ell'}$ and
$\mathsf O^{\le r}=(\mathsf O_a^{\ell\ell'})_{a\in\{1,2\},\,\ell,\ell'\le r}$. Suppose that
for all $r,i_1,i_2,p\in\{1,\dots,\lfloor\delta^{-1}\rfloor\}$ and every continuous
$f\colon\R^{2\times r\times r}\to\R$ with $|f|\le1$, writing
$\mathsf k_{\ell\ell'}=(\mathsf c_{i_1}\mathsf O_1^{\ell\ell'}+\mathsf c_{i_2}\mathsf O_2^{\ell\ell'})^p$, we have
\begin{equation}\label{eq:up-GGM}
\Big|\E\big\langle f(\mathsf O^{\le r})\mathsf k_{1,r+1}\big\rangle-\frac1r\E\big\langle f(\mathsf O^{\le r})\big\rangle\E\langle\mathsf k_{12}\rangle
-\frac1r\sum_{\ell=2}^r\E\big\langle f(\mathsf O^{\le r})\mathsf k_{1\ell}\big\rangle\Big|\le\delta .
\end{equation}
Suppose further that the law of $\mathsf O_2^{12}$ is $k^{-1}\sum_{\ell=1}^k\delta_{q_\ell}$
for some integer $k\ge1$ and $-1=q_00$ we define the finite set
\begin{equation}\label{eq:up-I}
I_\delta=\Big\{\Big(\frac{\mathsf c_{i_2}}{\mathsf c_{i_1}+\mathsf c_{i_2}},p\Big):
1\le i_1,i_2,p\le\lfloor\delta^{-1}\rfloor,\ \mathsf c_{i_1}+\mathsf c_{i_2}>0\Big\}\subseteq[0,1]\times\N .
\end{equation}
For $x\in V$ we write
\[
\mathcal V_x=\sum_{j=0}^Kw_j\Var\big(\varrho_{x,12}\,\big|\,\vartheta_{12}=z_j\big),
\]
where the variance is computed under $\E\langle\cdot\,|\,\vartheta_{12}=z_j\rangle$.
\section{Gaussian perturbations and the envelope}\label{sec:up-probes}
In this subsection $\mathcal S$, $I$, and $\mathsf s$ are as in the definition of the
model, and $\epsilon\in(0,1/8]$. We define the deterministic envelope
\begin{equation}\label{eq:up-envelope}
F_\star(t,q)=\min_{y\in\mathsf Y}\Big\{F(t,q,y)+\frac{\mathsf s^2}{n\epsilon}\sum_{x\in\mathcal S,\iota\in I}(y_{x\iota}-1)^2\Big\}.
\end{equation}
By \cref{lem:up-free}(a), the minimum is attained. By \cref{lem:up-free}, since
the penalty does not depend on $(t,q)$, vanishes at $y=\mathbf 1$, and is
nonnegative,
\begin{equation}\label{eq:up-envprops}
|F_\star(t,q)-F_\star(t',q')|\le\frac{\beta^2}4|t-t'|+\|q-q'\|_1,\qquad
|F_\star(t,q)-F_0(t,q)|\le r_{\mathsf s}.
\end{equation}
Small Gaussian perturbations that enforce the Ghirlanda--Guerra identities
\cite{GhirlandaGuerra1998} through the concentration of the free energy are a
standard tool; see \cite{Talagrand2003challenge} and
\cite[Section~3]{Panchenko2015multispecies}. The proof of \cref{lem:up-activation}
follows Steps~1--3 of the proof of \cite[Theorem~4.1]{Mourrat2021nonconvex}.
% Theorem 4.1 verified in arXiv:2004.01679v8; arXiv:2010.09114v2 cites it with the same number.
There,
the lower bound on the second derivative of the free energy in the perturbation
parameters comes from a test function that touches the free energy; here it comes
from the minimality of $y_*$ in \eqref{eq:up-envelope}. Because of
\cref{lem:up-fluct}, the strength $\mathsf s$ can be chosen independently of $n$.
\begin{lemma}\label{lem:up-activation}
Suppose that $\mathsf s\ge2$. Let $(t,q)\in(0,2)\times\upC_K^V$ and
let $y_*$ be a minimizer in \eqref{eq:up-envelope}. Then $|y_{*x\iota}-1|\le3\epsilon/2$
for every $(x,\iota)$, and, under the Gibbs measure at $(t,q,y_*)$, for every
$(x,\iota)\in\mathcal S\times I$, every $r\ge1$, and every measurable function $f$ of
$r$ replicas with $|f|\le1$,
\begin{equation}\label{eq:up-GGerror}
\Big|\E\langle f\mathsf k_{x\iota,1,r+1}\rangle-\frac1r\E\langle f\rangle\E\langle\mathsf k_{x\iota,12}\rangle
-\frac1r\sum_{\ell=2}^r\E\langle f\mathsf k_{x\iota,1\ell}\rangle\Big|
\le\frac{C_{\rm GG}}{\mathsf s\sqrt\epsilon},
\end{equation}
where $C_{\rm GG}$ depends only on $\zeta_0$.
\end{lemma}
\begin{proof}
Let $\lambda=\mathsf s^2/(n\epsilon)$. By \cref{lem:up-free}(d) and
\cref{lem:up-kernel}, $|\partial_{y_{x\iota}}F|\le2\mathsf s^2y_{x\iota}/n\le3\mathsf s^2/n$. If
$y_{*x\iota}=3/2$, minimality would require
$\partial_{y_{x\iota}}F+\lambda\le0$, which implies $\lambda\le3\mathsf s^2/n$ and
contradicts $\epsilon\le1/8$; the case $y_{*x\iota}=1/2$ is excluded in the same way.
This shows that $y_*$ is interior, and stationarity gives
$|y_{*x\iota}-1|=|\partial_{y_{x\iota}}F|/(2\lambda)\le3\epsilon/2\le3/16$. In
particular $y_*\pm\mathsf z e_{x\iota}\in\mathsf Y$ for $\mathsf z\in[0,1/4]$.
Fix $(x,\iota)$ and write $\mathsf g(\mathsf y)=F(t,q,y)$ and
$\widetilde{\mathsf g}(\mathsf y)=\widetilde F(t,q,y)$, where $y$ equals $y_*$ except
that $y_{x\iota}=\mathsf y$. By global minimality of $y_*$ and stationarity, we have, for
$|\mathsf z|\le1/4$,
\begin{equation}\label{eq:up-semiconcave}
\mathsf g(\mathsf y_*+\mathsf z)\ge\mathsf g(\mathsf y_*)+\mathsf g'(\mathsf y_*)\mathsf z-\lambda\mathsf z^2 ,
\end{equation}
since the penalty changes by $2\lambda(\mathsf y_*-1)\mathsf z+\lambda\mathsf z^2$ and
$\mathsf g'(\mathsf y_*)=-2\lambda(\mathsf y_*-1)$. Because $\mathsf g$ is smooth, it follows from
\eqref{eq:up-semiconcave} that $\mathsf g''(\mathsf y_*)\ge-2\lambda$. The random function
$\widetilde{\mathsf g}$ is concave and, by the domination argument in the proof of
\cref{lem:up-free}(a), almost surely twice differentiable, with
$\widetilde{\mathsf g}'=-n^{-1}\mathsf s\langle \mathsf G_{x,\iota}\rangle$ and
$\widetilde{\mathsf g}''=-n^{-1}\mathsf s^2\langle(\mathsf G_{x,\iota}-\langle \mathsf G_{x,\iota}\rangle)^2\rangle$,
and $\E\widetilde{\mathsf g}'=\mathsf g'$ by the argument in the proof of
\cref{lem:up-free}(a). By Fatou's lemma applied to the nonnegative quotients
$(\widetilde{\mathsf g}'(\mathsf y_*)-\widetilde{\mathsf g}'(\mathsf y_*+\epsilon'))/\epsilon'$ as
$\epsilon'\downarrow0$,
\begin{equation}\label{eq:up-thermal}
\frac{\mathsf s^2}n\E\big\langle(\mathsf G_{x,\iota}-\langle \mathsf G_{x,\iota}\rangle)^2\big\rangle\le-\mathsf g''(\mathsf y_*)\le2\lambda,
\qquad\text{so}\qquad
\E\big\langle(\mathsf G_{x,\iota}-\langle \mathsf G_{x,\iota}\rangle)^2\big\rangle\le\frac2\epsilon .
\end{equation}
Let $\mathsf z\in(0,1/4]$ and $\Delta=\widetilde{\mathsf g}-\mathsf g$. By concavity of
$\widetilde{\mathsf g}$ and by \eqref{eq:up-semiconcave} at $\pm\mathsf z$,
\[
\widetilde{\mathsf g}'(\mathsf y_*)-\mathsf g'(\mathsf y_*)\le\frac{\Delta(\mathsf y_*)-\Delta(\mathsf y_*-\mathsf z)}{\mathsf z}+\lambda\mathsf z,
\qquad
\widetilde{\mathsf g}'(\mathsf y_*)-\mathsf g'(\mathsf y_*)\ge\frac{\Delta(\mathsf y_*+\mathsf z)-\Delta(\mathsf y_*)}{\mathsf z}-\lambda\mathsf z .
\]
Since the three points $\mathsf y_*$, $\mathsf y_*\pm\mathsf z$ are deterministic,
\cref{lem:up-fluct} applies at each of them, and we obtain
$\E|\widetilde{\mathsf g}'(\mathsf y_*)-\mathsf g'(\mathsf y_*)|\le3C_{\rm fl}/(n\mathsf z)+\lambda\mathsf z$,
that is,
\[
\E\big|\langle \mathsf G_{x,\iota}\rangle-\E\langle \mathsf G_{x,\iota}\rangle\big|
\le\frac{3C_{\rm fl}}{\mathsf s\mathsf z}+\frac{\mathsf s\mathsf z}{\epsilon}
=\frac{3C_{\rm fl}+1}{\sqrt\epsilon}
\]
for the choice $\mathsf z=\mathsf s^{-1}\sqrt\epsilon$, which is at most $1/4$ because $\mathsf s\ge2$ and $\epsilon\le1/8$. Combining this with
\eqref{eq:up-thermal}, we obtain
\begin{equation}\label{eq:up-Gconc}
\E\big\langle|\mathsf G_{x,\iota}-\E\langle \mathsf G_{x,\iota}\rangle|\big\rangle\le\frac{3C_{\rm fl}+1+\sqrt2}{\sqrt\epsilon} .
\end{equation}
Let $f$ be a function of $r$ replicas with $|f|\le1$. By \eqref{eq:up-gibbsibp} and
\cref{lem:up-kernel}, with $\mathsf k_{x\iota,11}=1$,
\[
\E\langle \mathsf G_{x,\iota}(\alpha^1,\sigma^1)f\rangle=\mathsf sy_{*x\iota}\E\Big\langle f\Big(\sum_{\ell=1}^r\mathsf k_{x\iota,1\ell}-r\mathsf k_{x\iota,1,r+1}\Big)\Big\rangle,
\qquad
\E\langle \mathsf G_{x,\iota}\rangle=\mathsf sy_{*x\iota}\big(1-\E\langle\mathsf k_{x\iota,12}\rangle\big).
\]
Subtracting $\E\langle \mathsf G_{x,\iota}\rangle\E\langle f\rangle$ from the first identity and
dividing by $-r\mathsf sy_{*x\iota}$, we see that the left side of \eqref{eq:up-GGerror}
equals
$|\E\langle(\mathsf G_{x,\iota}(\alpha^1,\sigma^1)-\E\langle \mathsf G_{x,\iota}\rangle)f\rangle|/(r\mathsf sy_{*x\iota})$.
Since $y_{*x\iota}\ge13/16$, \eqref{eq:up-GGerror} follows from \eqref{eq:up-Gconc}, with
$C_{\rm GG}=16(3C_{\rm fl}+1+\sqrt2)/13$.
\end{proof}
\section{Sparsification and the drift}\label{sec:up-drift}
We use the following theorem of Cohen, Nelson, and Woodruff on row sparsification
\cite[Theorem~5]{CohenNelsonWoodruff2016}.
\begin{theorem}\label{thm:up-cnw}
There exists an absolute constant $C_0\ge1$ such that the following holds. Let
$\mathsf B$ be a real $n\times d$ matrix with $\|\mathsf B\|\le1$ and
$\|\mathsf B\|_{\rm F}^2\le\mathsf r$, where $\mathsf r>0$, and let $\lambda\in(0,1)$. Then
there exists a diagonal $n\times n$ matrix $\Theta$ with nonnegative entries and at
most $C_0(1+\mathsf r/\lambda^2)$ nonzero entries such that
$\|\mathsf B^{\mathsf T}\Theta^2\mathsf B-\mathsf B^{\mathsf T}\mathsf B\|\le\lambda$.
\end{theorem}
In \cite[Theorem~5]{CohenNelsonWoodruff2016} the diagonal matrix has
$O(\mathsf r/\lambda^2)$ nonzero entries; replacing it by its absolute value does not
change $\mathsf B^{\mathsf T}\Theta^2\mathsf B$. The proof of
\cite[Theorem~5]{CohenNelsonWoodruff2016} adapts the deterministic construction of
Batson, Spielman, and Srivastava \cite{BatsonSpielmanSrivastava2012}, in which the
number of nonzero entries is bounded in terms of the rank of $\mathsf B$. For
$\mathsf B=S$ the rank may be of order $n$, whereas $\|S\|_{\rm F}^2=n/D$. We apply \cref{thm:up-cnw} to
$\mathsf B=S$, with $\mathsf r=\tr S^2=n/D\ge1$, and obtain $\Theta$ with
\begin{equation}\label{eq:up-sparse}
\|S\Theta^2S-S^2\|\le\lambda,\qquad
|\{x:\Theta_{xx}\ne0\}|\le\frac{2C_0n}{D\lambda^2}.
\end{equation}
\begin{lemma}\label{lem:up-drift}
Let $\Theta$ satisfy \eqref{eq:up-sparse} and let $\{x:\Theta_{xx}\ne0\}\subseteq\mathcal S$.
Let $t>0$, $q\in\upC_K^V$, $y\in\mathsf Y$, and $\varepsilon_0>0$, and suppose that
$\mathcal V_x\le\varepsilon_0$ at $(t,q,y)$ for every $x\in\mathcal S$.
Then for every $\eta>0$,
\[
\big|\partial_tF(t,q,y)-\mathsf H_S(D_qF(t,q,y))\big|\le\tau,\qquad
\tau=\frac{\beta^2}4\Big[\eta+\frac{\lambda+(1+\lambda)\varepsilon_0}{4\eta}\Big].
\]
\end{lemma}
\begin{proof}
For $1\le j\le K$ let $\mathsf p_j=\E\langle u^*\,|\,\vartheta_{12}=z_j\rangle\in\R^V$ and let
$C_j$ be the covariance matrix of $u^*$ under $\E\langle\cdot\,|\,\vartheta_{12}=z_j\rangle$,
which is positive semidefinite with $\tr C_j\le n$. Since $u^*=0$ on
$\{\vartheta_{12}=z_0\}$, \cref{lem:up-free}(b),(c) and \eqref{eq:up-overlaplaw} give
\[
\partial_tF=\frac{\beta^2}{4n}\sum_{j=1}^Kw_j\E\langle u^*\cdot Su^*\,|\,\vartheta_{12}=z_j\rangle
=\mathsf H_S(\mathsf p)+\frac{\beta^2}{4n}\sum_{j=1}^Kw_j\tr(SC_j),
\]
with $\mathsf p=D_qF$. For the Sherrington--Kirkpatrick model, the analogous
identity is \cite[Lemma~6.2]{DominguezMourrat2024}; there the error term is the
conditional variance of the overlap given the cascade overlap, which is
nonnegative. For each eigenvalue $\mu$ of $S$ we have
$|\mu|\le\eta+\mu^2/(4\eta)$. Since $C_j$ is positive semidefinite, we deduce that
$|\tr(SC_j)|\le\eta\tr C_j+\tr(S^2C_j)/(4\eta)$. By \eqref{eq:up-sparse},
$\tr(S^2C_j)\le\lambda\tr C_j+\tr(S\Theta^2SC_j)$, and
\[
\tr(S\Theta^2SC_j)=\sum_x\Theta_{xx}^2(SC_jS)_{xx}=\sum_x\Theta_{xx}^2\Var\big(\varrho_{x,12}\,\big|\,\vartheta_{12}=z_j\big).
\]
Let $\omega_x=\Theta_{xx}^2/n$. Testing \eqref{eq:up-sparse} on $\mathbf 1$ and using
$S\mathbf1=\mathbf1$, we find $\sum_x\omega_x\le1+\lambda$. We conclude that
\[
\sum_jw_j\frac{\tr(S^2C_j)}n\le\lambda+\sum_{x\in\mathcal S}\omega_x\mathcal V_x
\le\lambda+(1+\lambda)\varepsilon_0,
\]
and the claim follows.
\end{proof}
\begin{lemma}\label{lem:up-super}
Suppose that $\zeta_j=(j+1)/(K+1)$ for $0\le j\le K$. Let $\delta_{\rm M}$ be given by
\cref{prop:up-sync} for $\varepsilon=K^{-1}(K+1)^{-3}$, and let $I=I_{\delta_{\rm M}}$ be as in \eqref{eq:up-I}. Let
$\Theta$ satisfy \eqref{eq:up-sparse}, let $\mathcal S=\{x:\Theta_{xx}\ne0\}$, and suppose
that
\begin{equation}\label{eq:up-sbig}
\mathsf s\ge2\qquad\text{and}\qquad
\frac{2^{\lfloor\delta_{\rm M}^{-1}\rfloor}C_{\rm GG}}{\mathsf s\sqrt\epsilon}\le\delta_{\rm M} .
\end{equation}
Then $F_\star$ is a supersolution of $\partial_tu-\mathsf H_{\rm ext}(D_qu)\ge-\tau$, with
$\tau$ as in \cref{lem:up-drift} for $\varepsilon_0=13/(K+1)$.
\end{lemma}
\begin{proof}
Let $\varphi$ be continuously differentiable near $(t_0,q_0)\in(0,2)\times\upC_K^V$,
and let $F_\star-\varphi$ have a local minimum at $(t_0,q_0)$. Let $y_*$ be a minimizer in
\eqref{eq:up-envelope} at $(t_0,q_0)$; it is a deterministic point. The function
$(t,q)\mapsto F(t,q,y_*)+\mathsf s^2(n\epsilon)^{-1}\sum_{x,\iota}(y_{*x\iota}-1)^2$ is at least
$F_\star$ and equals it at $(t_0,q_0)$. Then $F(\cdot,\cdot,y_*)-\varphi$ also has a local
minimum at $(t_0,q_0)$. By \cref{lem:up-free}(a) and \eqref{eq:up-touch},
$\partial_t\varphi=\partial_tF$ and $D_qF\succeq_*D_q\varphi$ at $(t_0,q_0,y_*)$.
Fix $x\in\mathcal S$, and let $\mathfrak G$ be the image of the Gibbs measure at $(t_0,q_0,y_*)$ under the map
$(\alpha,\sigma)\mapsto(\mathsf F_x(\alpha,\sigma),\mathsf e(\alpha))$, where
\[
\mathsf F_x(\alpha,\sigma)=\big(S_{xy}^{1/2}\sigma_ye_{\alpha_1}\big)_{y\in V},\qquad
\mathsf e(\alpha)=\sum_{k=1}^K(z_k-z_{k-1})^{1/2}e_{\alpha|_k},
\]
for orthonormal families $(e_a)_{a\in\N}$ and $(e_\gamma)_{\gamma\in\mathbb A\setminus\{\emptyset\}}$. Both components are unit vectors, since $\sum_yS_{xy}=1$ and $z_K-z_0=1$. For the replicas $(\alpha^\ell,\sigma^\ell)$, the overlaps of \cref{prop:up-sync} are $\mathsf O_1^{\ell\ell'}=\varrho_{x,\ell\ell'}$ and $\mathsf O_2^{\ell\ell'}=\vartheta_{\ell\ell'}$. We check the hypotheses of \cref{prop:up-sync} with $\delta=\delta_{\rm M}$. Let $r,i_1,i_2,p\le\lfloor\delta_{\rm M}^{-1}\rfloor$ and let $f$ be as there. If $\mathsf c_{i_1}+\mathsf c_{i_2}=0$, then $\mathsf k_{\ell\ell'}=0$ for all $\ell,\ell'$, and the left side of \eqref{eq:up-GGM} vanishes. Otherwise
\[
\big(\mathsf c_{i_1}\varrho+\mathsf c_{i_2}\vartheta\big)^p=(\mathsf c_{i_1}+\mathsf c_{i_2})^p\,\mathsf k_\iota(\vartheta,\varrho),\qquad
\iota=\Big(\frac{\mathsf c_{i_2}}{\mathsf c_{i_1}+\mathsf c_{i_2}},p\Big)\in I,
\]
and $f(\mathsf O^{\le r})$ is a measurable function of $r$ replicas with $|f(\mathsf O^{\le r})|\le1$. Then the left side of \eqref{eq:up-GGM} is $(\mathsf c_{i_1}+\mathsf c_{i_2})^p\le2^{\lfloor\delta_{\rm M}^{-1}\rfloor}$ times the left side of \eqref{eq:up-GGerror}, and it is at most $\delta_{\rm M}$ by \cref{lem:up-activation} and \eqref{eq:up-sbig}. By \eqref{eq:up-overlaplaw}, the law of $\vartheta_{12}$ is $(K+1)^{-1}\sum_{j=0}^K\delta_{z_j}$. This is the law in \cref{prop:up-sync} with $k=K+1$ and $q_\ell=z_{\ell-1}$ for $1\le\ell\le k$, and the spacings $q_{\ell+1}-q_\ell$ are $1$ for $\ell=0$ and $1/K$ otherwise. Since each event $\{\vartheta_{12}=z_j\}$ has probability $w_j>0$, the left side of the conclusion of \cref{prop:up-sync} equals $\mathcal V_x$, and the proposition implies that
\[
\mathcal V_x\le\frac{12}{K+1}+\varepsilon(K+1)^2K=\frac{13}{K+1}.
\]
This holds for every $x\in\mathcal S$, and \cref{lem:up-drift} implies that
$\partial_tF\ge\mathsf H_S(D_qF)-\tau$ at $(t_0,q_0,y_*)$.
Since $D_qF\in\upC_K^V$ by \cref{lem:up-free}(c), we have
$\mathsf H_S(D_qF)=\mathsf H_{\rm ext}(D_qF)\ge\mathsf H_{\rm ext}(D_q\varphi)$ by
\cref{lem:up-ext}(c). Therefore
$\partial_t\varphi-\mathsf H_{\rm ext}(D_q\varphi)\ge-\tau$ at $(t_0,q_0)$.
\end{proof}
\section{Proof of \texorpdfstring{\cref{prop:upper}}{the upper bound}}\label{sec:up-proof}
If $\beta=0$, then $p_G(0,h)=\log2\cosh h$ by \cref{lem:pre-basic}, and
$\PSK(0,h)=\log2\cosh h$ by the same lemma for the Sherrington--Kirkpatrick model. Let
$\beta>0$, $h\in\R$, and $\upsilon\in(0,1)$.
We first make all choices that do not depend on $D$ or $G$. Let $\nu=\nu_{\beta,h}$ be the Parisi measure, and let $\mathsf Q(\theta)=\min\{s\in[0,1]:\nu([0,s])\ge\theta\}$ for $\theta\in(0,1)$. Extend $\mathsf Q$ to $[0,1]$ by $\mathsf Q(0)=\min\supp\nu$ and $\mathsf Q(1)=\max\supp\nu$. The function $\mathsf Q$ is nondecreasing, and the image of the Lebesgue measure on $(0,1)$ under $\mathsf Q$ is $\nu$. For $K\ge1$, let $r^{(K)}_j=\mathsf Q((j+1/2)/(K+1))$ for $0\le j\le K$, so that $r^{(K)}\in\upC_K\cap[0,1]^{K+1}$, and let $\mu^{(K)}=(K+1)^{-1}\sum_{j=0}^K\delta_{r^{(K)}_j}$. For $f\in C([0,1])$, the integral $\int f\dd\mu^{(K)}$ is a Riemann sum of $f\circ\mathsf Q$ for the partition of $[0,1]$ into $K+1$ intervals of equal length. The function $f\circ\mathsf Q$ is bounded and continuous outside the set of discontinuities of $\mathsf Q$, which is countable. By Lebesgue's criterion it is Riemann integrable, and these Riemann sums converge to $\int_0^1f(\mathsf Q(\theta))\dd\theta=\int f\dd\nu$ as $K\to\infty$. Then $\mu^{(K)}\to\nu$ weakly, and \cref{lem:up-parisi} and \eqref{eq:pre-parisi-formula} imply that $\Par_{\beta,h}(\mu^{(K)})\to\PSK(\beta,h)$.
Let $\eta=2\upsilon/(3\beta^2)$, so that $\beta^2\eta/4=\upsilon/6$. Let
$K\ge1$ be so large that, with $\zeta_j=(j+1)/(K+1)$ (so that
$w_j=\bar w=\zeta_0=1/(K+1)\le1/2$ and $\mu_{r^{(K)}}=\mu^{(K)}$), we have
\[
\Par_{\beta,h}(\mu^{(K)})\le\PSK(\beta,h)+\frac\upsilon6,\qquad\frac{\beta^2\zeta_0}4\le\frac\upsilon6\qquad\text{and}\qquad
\frac{\beta^2}{16\eta}\cdot21\bar w\le\frac\upsilon6 .
\]
Let $\lambda=\bar w$ and $\varepsilon_0=13\bar w$, so that $\lambda+(1+\lambda)\varepsilon_0\le21\bar w$
and the constant $\tau$ of \cref{lem:up-drift} satisfies $\tau\le\upsilon/3$. Let
$\delta=\upsilon/6$ and $\epsilon=1/8$. Let $\delta_{\rm M}$ and $I=I_{\delta_{\rm M}}$ be as in \cref{lem:up-super}, and let $C_{\rm GG}$ be as in \cref{lem:up-activation}. These objects depend only on $\beta$, $h$, and $\upsilon$.
Now let $D\ge1$, let $\mathsf s=D^{1/4}$, and let $D_0$ be such that \eqref{eq:up-sbig}
holds for $D\ge D_0$. Let $D\ge D_0$ and $G\in\cG_D$. Let $\Theta$ be given by
\eqref{eq:up-sparse}, let $\mathcal S=\{x:\Theta_{xx}\ne0\}$, and let the model of
\cref{sec:up-rpc} be constructed with these $\mathcal S$, $I$, and $\mathsf s$. By
\eqref{eq:up-sparse},
\begin{equation}\label{eq:up-rs}
r_{\mathsf s}=\frac94\frac{|\mathcal S||I|\mathsf s^2}{n}\le\frac{9C_0|I|}{2\lambda^2\sqrt D} .
\end{equation}
We apply \cref{lem:up-comparison} with $\mathsf H=\mathsf H_{\rm ext}$, $\mathsf V=V_\delta$, and $\mathsf E=F_\star$, restricted to
$[0,2]\times\upC_K^V$. By \cref{lem:up-ext}(c), $\mathsf H_{\rm ext}$ is Lipschitz on bounded sets. Hypothesis (i) holds by \cref{lem:up-subsol}(b) and
\cref{lem:up-super}; (ii) holds with $L_t=\beta^2$ and $L_q=1$ by
\cref{lem:up-subsol}(a) and \eqref{eq:up-envprops}. For (iii), by
\cref{lem:up-subsol}(c), \cref{lem:up-free}(f), and \eqref{eq:up-envprops},
\[
V_\delta(0,q)\le\frac1n\sum_x\psi_h(\tilde q_x)+\delta=F_0(0,q)+\delta\le F_\star(0,q)+\delta+r_{\mathsf s}.
\]
The lemma then gives $V_\delta(1,0)\le F_\star(1,0)+\delta+r_{\mathsf s}+\tau$. By \cref{lem:up-subsol}(d),
\eqref{eq:up-envprops}, and \cref{lem:up-free}(e),
\[
\mathsf U(1,0)-\frac{\beta^2\zeta_0}4\le F_0(1,0)+\delta+2r_{\mathsf s}+\tau\le\frac{\beta^2}4-p_G(\beta,h)+\delta+2r_{\mathsf s}+\tau .
\]
By \cref{lem:up-parisi} with $r=r^{(K)}$ and the choice of $K$, we have
$\mathsf U(1,0)\ge\beta^2/4-\Par_{\beta,h}(\mu^{(K)})\ge\beta^2/4-\PSK(\beta,h)-\upsilon/6$. Combining these
bounds with the choices above and \eqref{eq:up-rs}, we obtain
\[
p_G(\beta,h)\le\PSK(\beta,h)+\frac\upsilon6+\frac{\beta^2\zeta_0}4+\delta+\tau+2r_{\mathsf s}
\le\PSK(\beta,h)+\frac{5\upsilon}6+\frac{9C_0|I|}{\lambda^2\sqrt D}.
\]
The right side does not depend on $G\in\cG_D$ or on $n$. Taking the supremum over
$G\in\cG_D$ and letting $D\to\infty$, we obtain
$\limsup_{D\to\infty}\sup_{G\in\cG_D}p_G(\beta,h)\le\PSK(\beta,h)+5\upsilon/6$. Since
$\upsilon$ is arbitrary, this completes the proof of \cref{prop:upper}.
% ---------- 05-centres.tex
% 05-centres.tex: message-passing centres on tori and hypercubes.
% Label prefix: amp-. Owner: AMP agent.
\chapter{Message-passing centers on tori and hypercubes}\label{sec:amp}
The goal of this section is to prove \cref{prop:amp-centre}, which constructs
the centers used in the lower bound on tori and hypercubes. A center is a
vector $m\in(-1,1)^V$ computed from finitely many products of the coupling
matrix $W$ with vectors, each of which is a function of the earlier products.
The following definition records this structure, which is the form required
by \cref{thm:band}.
\begin{definition}\label{def:legal}
Let $G\in\cG_D$ and let $T\ge1$ be an integer. A \emph{query history of depth
$T$} on $G$ consists of a random variable $\mathsf U$ independent of
$(g_e)_{e\in E}$ and random vectors $x^1,\dots,x^T\in\R^V$ such that, for each
$t\le T$, the vector $x^t$ is measurable with respect to the $\sigma$-algebra
generated by $\mathsf U$ and $Wx^1,\dots,Wx^{t-1}$. We write $\mathcal H$ for
the $\sigma$-algebra generated by $\mathsf U$ and $Wx^1,\dots,Wx^T$. A
\emph{center} of the history is an $\mathcal H$-measurable vector
$m\in(-1,1)^V$ such that $m=x^t$ for some $t\le T$.
\end{definition}
Throughout this section, $(G_d)_{d\ge1}$ denotes one of two sequences of
graphs: either the tori $G_d=\T^d_L$ for a fixed $L\ge3$, with $D=2d$ and
$n=L^d$, or the hypercubes $G_d=Q_d$, with $D=d$ and $n=2^d$. Both are
vertex-transitive, since they are Cayley graphs of $(\Z/L\Z)^d$ and
$(\Z/2\Z)^d$. We use the notation of \eqref{eq:pre-model} on $G_d$, and we
write $\|w\|_n=(n^{-1}\sum_xw_x^2)^{1/2}$ for the normalized Euclidean norm
of $w\in\R^V$. The Parisi measure $\nu=\nu_{\beta,h}$, the endpoints
$q_-\le q_+$ of its support, the functions $a$ and $v$, the Parisi diffusion
$X_t$, the martingale $M_t=\partial_x\Phi(t,X_t)$ and the law
$\mu_+=\operatorname{Law}(M_{q_+})$ are those introduced with
\eqref{eq:pre-parisi-pde} and \eqref{eq:pre-parisi-formula}.
\begin{proposition}\label{prop:amp-centre}
Let $\beta>0$ and $h>0$. Let $(G_d)_{d\ge1}$ be either the tori $\T^d_L$
for a fixed $L\ge3$, with $D=2d$, or the hypercubes $Q_d$, with $D=d$. For
every $\epsilon>0$ there exist an integer $T\ge1$, a probability measure
$\mu_\epsilon$ on $[-1,1]$ and, for every $d$, a query history of depth $T$ on
$G_d$ with center $m=m^{(d)}$, such that the following three statements hold.
\begin{enumerate}[label=(\alph*)]
\item With $\kappa_x=1-m_x^2$, we have
\[
\liminf_{d\to\infty}\frac1n\E\Big[\beta H_{G_d}(m)+h\sum_{x}m_x
+\sum_{x}\ent(m_x)+\frac{\beta^2}4\langle\kappa,S\kappa\rangle\Big]
\ge\PSK(\beta,h)-\epsilon .
\]
\item We have $\mathrm W_1(\mu_\epsilon,\mu_+)\le\epsilon$, where $\mathrm W_1$
is the Wasserstein distance of order one on $[-1,1]$.
\item Let $\Upsilon\colon[-1,1]\times[0,\infty)\to\R$ be continuous, and let
$L_\Upsilon\ge0$ be such that
$|\Upsilon(m',r)-\Upsilon(m',r')|\le L_\Upsilon|r-r'|$ for all
$m'\in[-1,1]$ and $r,r'\ge0$. Let $a,c,B\ge0$. Then
\[
\lim_{d\to\infty}\E\sup_{r\in a\mathbf 1+cS[0,B]^V}\Big|\frac1n\sum_{x}
\Upsilon(m_x,r_x)-\frac1n\sum_{x}\int\Upsilon(m',r_x)\,\mu_\epsilon(\dd m')\Big|
=0 .
\]
\end{enumerate}
\end{proposition}
\begin{remark}\label{rem:amp-centre}
We record three comments on the statement.
\begin{itemize}
\item The integer $T$ and the measure $\mu_\epsilon$ depend on $\beta$, $h$,
$\epsilon$, and, for tori, on $L$. The history is given by the same formulas
for every $d$: polynomials, real coefficients, and a continuous output
function, none of which depend on $d$.
\item All terms in (a) are integrable, since $|H_{G_d}(m)|\le\sum_{e}|J_e|$.
\item The supremum in (c) is a random variable: since $a\mathbf 1+cS[0,B]^V$
is compact and the expression inside the absolute value is continuous in
$r$, the supremum may be taken over a countable dense subset.
\end{itemize}
\end{remark}
The centers are produced by approximate message passing with full memory and
deterministic Onsager coefficients. The Gaussian state evolution of message
passing was first proved by Bolthausen~\cite{Bolthausen2014} and by Bayati and
Montanari~\cite{BayatiMontanari2011}. \Cref{sec:amp-recursion} introduces this
recursion and states its state evolution, \cref{prop:amp-SE}. We deduce
\cref{prop:amp-SE} from a theorem of Hachem~\cite{Hachem2024} on a recursion
whose Onsager coefficients are computed from the disorder.
\Cref{sec:amp-imported} states this theorem and a bound on $\|W\|$,
\cref{sec:amp-tame} proves moment bounds and a coefficient-replacement
estimate, and \cref{sec:amp-SE-proof} proves \cref{prop:amp-SE}.
\Cref{sec:amp-energy} computes the energy of the output.
\Cref{sec:amp-controls,sec:amp-parisi} adapt to this recursion the two-phase
algorithm of Sellke~\cite{Sellke2021field}. Its first phase is an iteration of
the type used by Bolthausen~\cite{Bolthausen2014}, and its second phase is the
incremental message passing of Montanari~\cite{Montanari2021} and El Alaoui,
Montanari, and Sellke~\cite{ElAlaouiMontanariSellke2021}, whose value is
described by the stochastic control problem
of~\cite[Section~4]{ElAlaouiMontanariSellke2021}. For the SK model without
field, Montanari showed that the incremental algorithm constructs approximate
solutions of the TAP equations for all large $\beta$, under the assumption that
the support of the Parisi measure is an interval $[0,q_*]$ for all large
$\beta$~\cite[Theorem~5]{Montanari2021}.
\Cref{sec:amp-homog} proves the homogenization estimate, and
\cref{sec:amp-proof} completes the proof of \cref{prop:amp-centre}.
% CHECK: Montanari2021 Theorem 5 and ElAlaouiMontanariSellke2021 Section 4 were
% verified in arXiv:1812.10897v2 and arXiv:2001.00904v1 only.
% CHECK: Montanari2021 (2.1), (2.37), Lemma 2.3 and Appendix A (arXiv:1812.10897v2),
% JavanmardMontanari2013 Theorem 1 (arXiv:1211.5164v2), Hachem2024 Lemma 21 and
% Theorems 2, 4 (arXiv:2302.09847v3 and v4, same numbering), Hachem2026sparse
% Theorem 5 (arXiv:2604.25535v1); BayatiLelargeMontanari2015 Appendix A.2 checked in
% arXiv:1207.7321v2, which is the published version.
The Onsager coefficients of our recursion are real numbers computed from the
state evolution. Using Hachem's coefficients, which explicitly involve squared
individual couplings, would require information beyond the earlier products
and would not in general satisfy \cref{def:legal}. Deterministic
Onsager coefficients for sparse variance profiles also appear in Theorems~2
and~4 of~\cite{Hachem2024}, which treat exact and approximate iterates,
respectively, under a nondegeneracy assumption. At high temperature,
\cite[Theorem~5]{Hachem2026sparse} treats an exact recursion with the additional
sparsity condition $K_n\ge\log n$. These recursions have nonlinearities that
act only on the last iterate; the incremental phase below needs the full
history. We instead import the sampled-coefficient polynomial
result~\cite[Proposition~8]{Hachem2024} and prove the replacement by
deterministic coefficients in \cref{sec:amp-SE-proof}.
Throughout this section, $Z$ denotes a standard Gaussian variable,
$\|\xi\|_p=(\E|\xi|^p)^{1/p}$ for a random variable $\xi$, and a function on
$\R^k$ has \emph{polynomial growth} if it is bounded by $C(1+|z|^p)$ for some
$C,p\ge0$. Functions act on vectors coordinatewise: for $f\colon\R^t\to\R$ and
$u^1,\dots,u^t\in\R^V$, the vector $f(u^1,\dots,u^t)\in\R^V$ has coordinates
$f(u^1_x,\dots,u^t_x)$. We write $w\odot w'$ for the coordinatewise product.
\section{Message passing with deterministic Onsager coefficients}\label{sec:amp-recursion}
Fix an integer $T\ge1$, a constant $f_0\in\R$ and polynomials
$f_t\colon\R^t\to\R$ for $1\le t\le T-1$. For a sequence $u=(u^1,u^2,\dots)$ we
write $\bar u^t=(u^1,\dots,u^t)$, with $\bar u^0$ empty and
$f_0(\bar u^0)=f_0$. The \emph{state evolution} is the centered Gaussian vector
$U=(U_1,\dots,U_T)$ whose covariance is given recursively by
\begin{equation}\label{eq:amp-SE}
\E\,U_{r+1}U_{s+1}=\E\,f_r(\bar U^r)f_s(\bar U^s)\qquad(0\le r,s\le T-1).
\end{equation}
The recursion \eqref{eq:amp-SE} defines the law of $U$, since the covariance
of $\bar U^{t+1}$ is determined by the law of $\bar U^t$; degenerate covariances
are allowed.
The \emph{Onsager coefficients} are the real numbers
\begin{equation}\label{eq:amp-b}
b_{t,s}=\E\,\partial_sf_t(\bar U^t)\qquad(1\le s\le t\le T-1).
\end{equation}
On a graph $G\in\cG_D$ we define $u^1,\dots,u^T\in\R^V$ by $u^1=f_0W\mathbf 1$
and
\begin{equation}\label{eq:amp-recursion}
u^{t+1}=Wf_t(\bar u^t)-\sum_{s=1}^tb_{t,s}f_{s-1}(\bar u^{s-1})\qquad
(1\le t\le T-1),
\end{equation}
where $f_0(\bar u^0)=f_0\mathbf 1$. The coefficients $b_{t,s}$ depend only on
$f_0,\dots,f_{T-1}$; they do not depend on $G$ or on the couplings. If
$b_{t,s}$ is replaced by the empirical average
$n^{-1}\sum_x\partial_sf_t(\bar u^t_x)$, then \eqref{eq:amp-recursion} is the
message-passing iteration of~\cite[(2.1)]{Montanari2021}, whose state
evolution is \eqref{eq:amp-SE}.
\begin{lemma}\label{lem:amp-history}
Let $G\in\cG_D$, let $1\le K\le T$, and let $\theta\colon\R^K\to(-1,1)$ be
measurable. Set $x^t=f_{t-1}(\bar u^{t-1})$ for $1\le t\le K$, and
$x^{K+1}=\theta(\bar u^K)$. Then $x^1,\dots,x^{K+1}$, with $\mathsf U$
constant, form a query history of depth $K+1$ on $G$, and
$m=\theta(\bar u^K)$ is a center of it.
\end{lemma}
\begin{proof}
We show by induction on $t\le K$ that $\bar u^t$ is measurable with respect
to $\sigma(Wx^1,\dots,Wx^t)$. Since $x^1=f_0\mathbf 1$, we have $u^1=Wx^1$.
If the claim holds for some $t0$ and every sequence of nonempty sets
$I_k\subset[n_k]$ with $|I_k|\le CD_k$, as $k\to\infty$,
\[
\frac1{D_k}\sum_{x\in I_k}\big(\psi(\mathbf z^t_x)
-\E\psi(\boldsymbol\zeta^t_x)\big)\to0
\quad\text{and}\quad
\frac1{n_k}\sum_{x\in[n_k]}\big(\psi(\mathbf z^t_x)
-\E\psi(\boldsymbol\zeta^t_x)\big)\to0
\]
in probability.
\end{enumerate}
\end{theorem}
In the notation of~\cite{Hachem2024}, \cref{thm:amp-hachem} is Proposition~8,
equations (19)--(21), with sparsity parameter $K_n=D_k$, constants
$C_{\mathrm{card}}=C_S=1$, standard Gaussian entries (which satisfy
Assumption~1 there), polynomial maps and test polynomials that do not depend
on the vertex or on $n$, and initial vector zero. Assumption~2
of~\cite{Hachem2024} requires only $K_n\to\infty$, $K_n\le n$, the bound on
the number of nonzero entries in each row, and the bound on the entries; in
particular, no relation between $K_n$ and $\log n$ is needed. Equation (20)
there bounds the moments of monomials, which implies the moment bound in (i)
for polynomials. The normalization in (ii) is $D_k^{-1}$ for every admissible
set, as in equation (21a) there; we use this local conclusion only for neighborhoods, which
have exactly $D_k$ elements. The theorem requires neither a nondegeneracy
assumption on the covariances nor a bound on $\|W\|$; these enter only
Hachem's results for nonpolynomial activation functions.
The second imported result is a bound of Bandeira and van
Handel~\cite[Theorem~1.1]{BandeiraVanHandel2016} on the norm of a Gaussian
matrix with a variance profile.
\begin{theorem}\label{thm:amp-bvh}
Let $X$ be the $n\times n$ symmetric matrix with $X_{xy}=b_{xy}g_{xy}$, where
$\{g_{xy}:x\ge y\}$ are independent standard Gaussian variables and
$\{b_{xy}:x\ge y\}$ are given real numbers, with $b_{xy}=b_{yx}$. Set
$\sigma=\max_x(\sum_yb_{xy}^2)^{1/2}$ and $\sigma_*=\max_{x,y}|b_{xy}|$. Then
for every $0<\epsilon'\le1/2$,
\[
\E\|X\|\le(1+\epsilon')\Big(2\sigma
+\frac{6}{\sqrt{\log(1+\epsilon')}}\,\sigma_*\sqrt{\log n}\Big).
\]
\end{theorem}
\begin{lemma}\label{lem:amp-norm}
Let $G\in\cG_D$ have $n\ge2$ vertices. Then
$\E\|W\|\le3+15(D^{-1}\log n)^{1/2}$. For the tori $\T^d_L$ with a fixed
$L\ge3$ and for the hypercubes $Q_d$, we have $\sup_d\E\|W\|^p<\infty$ for
every $p\ge1$.
\end{lemma}
\begin{proof}
We apply \cref{thm:amp-bvh} with $b_{xy}=S_{xy}^{1/2}$ and $b_{xx}=0$, so
that $\sigma=1$ and $\sigma_*=D^{-1/2}$. With $\epsilon'=1/2$ we have
$9(\log(3/2))^{-1/2}\le15$, which gives the first bound. Next, we view $\|W\|$
as a function of $g=(g_e)_{e\in E}$. For two values $g,g'$,
\[
\big|\|W(g)\|-\|W(g')\|\big|\le\|W(g)-W(g')\|_{\mathrm{HS}}
=\Big(\frac2D\sum_{e\in E}(g_e-g'_e)^2\Big)^{1/2},
\]
that is, $\|W\|$ is $(2/D)^{1/2}$-Lipschitz. By Gaussian
concentration~\ref{F:concentration}, we have
$\|\|W\|-\E\|W\|\|_p\le C_p(2/D)^{1/2}$, where $C_p$ depends only on $p$. For
tori, $D^{-1}\log n=(\log L)/2$, and for hypercubes, $D^{-1}\log n=\log2$.
By the first bound, we have $\sup_d\E\|W\|<\infty$, and the moment bound
follows from Minkowski's inequality.
\end{proof}
\section{Tame polynomials and coefficient replacement}\label{sec:amp-tame}
In this subsection $G\in\cG_D$ is arbitrary, and the entries $W_e=J_e$,
$e\in E$, are independent with law $N(0,D^{-1})$. Moment bounds that do not
depend on $D$ come from a structural property: every monomial of an iterate
at $x$ is supported on a connected set of edges attached to $x$. There are at
most $CD^{s}$ such sets with $s$ edges, and each edge carrying a nonzero even
power contributes a factor at most $C/D$. This bookkeeping follows the moment
method of~\cite{BayatiLelargeMontanari2015}. The letter $C$ denotes constants
that depend only on the indicated quantities, never on $D$, on $G$, or on the
chosen vertices; its value may change from line to line.
A \emph{monomial} is a product $M=\prod_{e\in E}W_e^{k_e}$ with integers
$k_e\ge0$, finitely many of them nonzero. Its degree is
$\deg M=\sum_ek_e$, and its support is $\supp M=\{e:k_e\ge1\}$. A set of
edges is \emph{connected} if the graph formed by these edges and their
endpoints is connected.
\begin{definition}\label{def:amp-tame}
Let $x\in V$, let $k\ge0$ be an integer, and let $C_0\ge0$. A finite sum
$Y=\sum_Mc_MM$ of monomials is \emph{tame rooted at $x$, of type $(k,C_0)$},
if every monomial $M$ with $c_M\neq0$ satisfies $\deg M\le k$ and
$|c_M|\le C_0$, and $\supp M$ is either empty or a connected set of edges at
least one of which has $x$ as an endpoint.
\end{definition}
\begin{lemma}\label{lem:amp-tame-closure}
Let $x\in V$.
\begin{enumerate}[label=(\alph*)]
\item If $Y$ and $Y'$ are tame rooted at $x$, of types $(k,C_0)$ and
$(k',C_0')$, then $Y+Y'$ is tame rooted at $x$, of type
$(\max(k,k'),C_0+C_0')$, and $YY'$ is tame rooted at $x$, of type
$(k+k',2^{k+k'}C_0C_0')$. A polynomial of degree at most $m$,
with coefficients bounded by $C_1$, in $j$ tame polynomials rooted at $x$ of
type $(k,C_0)$ is tame rooted at $x$, of a type depending only on
$k,C_0,m,C_1,j$.
\item If $Y_y$ is tame rooted at $y$, of type $(k,C_0)$, for every neighbor
$y$ of $x$, then $\sum_{y\sim x}W_{xy}Y_y$ and $\sum_{y\sim x}W_{xy}^2Y_y$ are
tame rooted at $x$, of types $(k+1,(k+1)C_0)$ and $(k+2,(k+2)C_0)$.
\end{enumerate}
\end{lemma}
\begin{proof}
(a) The statement on sums is immediate. The coefficient of $M$ in $YY'$ is
$\sum_{M_1M_2=M}c_{M_1}c'_{M_2}$, and $M$ has at most
$\prod_e(k_e+1)\le2^{\deg M}$ factorizations. Further,
$\supp(M_1M_2)=\supp M_1\cup\supp M_2$ is empty or connected with an edge at
$x$, because both sets are, and each of them that is nonempty contains the
vertex $x$. Constants are tame rooted at $x$, of type $(0,|c|)$, which gives
the statement on polynomials.
(b) Let $M_y$ be a monomial of $Y_y$. The support of $W_{xy}M_y$ is
$\supp M_y\cup\{xy\}$. It is connected and contains the edge $xy$ at $x$:
either $\supp M_y$ is empty, or it is connected and contains an edge at $y$,
which shares the vertex $y$ with $xy$. A given monomial $M$ arises as
$W_{xy}M_y$ only for neighbors $y$ with $xy\in\supp M$, which are at most
$\deg M\le k+1$ in number, and for each of them $M_y=M/W_{xy}$ is determined.
The coefficient of $M$ is therefore at most $(k+1)C_0$. The same argument applies
to $W_{xy}^2M_y$.
\end{proof}
\begin{lemma}\label{lem:amp-counting}
Let $x\in V$ and let $v\ge2$. The number of connected sets of edges that
contain an edge at $x$ and span exactly $v$ vertices is at most
$4^v2^{v^2}D^{v-1}$.
\end{lemma}
\begin{proof}
Such a set $\mathcal S$ has a spanning tree. We root it at $x$ and list its
vertices in breadth-first order. There are at most $4^v$ rooted plane trees
with $v$ vertices. Given the shape, there are at most $D^{v-1}$ labelings, since the root is
$x$ and each other vertex is a neighbor of its parent, which is listed before
it. Finally, $\mathcal S$ is a set of pairs of these $v$
vertices, which leaves at most $2^{v^2}$ choices.
\end{proof}
\begin{lemma}\label{lem:amp-tame-moments}
If $Y_1,\dots,Y_j$ are tame rooted at the same vertex $x$, of type $(k,C_0)$,
then $|\E[Y_1\cdots Y_j]|\le C$, where $C$ depends only on $j$, $k$, and
$C_0$. In particular, if $Y$ is tame rooted at $x$, of type $(k,C_0)$, then
$\E|Y|^p\le C$ for every $p\ge1$, with $C$ depending only on $p$, $k$, and
$C_0$.
\end{lemma}
\begin{proof}
We expand $\E\prod_iY_i=\sum\prod_ic_{M_i}\E\prod_iM_i$ over $j$-tuples of
monomials. For a tuple, let $\mathcal S=\bigcup_i\supp M_i$ and
$\mu_e=\sum_ik_e(M_i)$ for $e\in\mathcal S$. By independence,
$\E\prod_iM_i=\prod_{e\in\mathcal S}\E W_e^{\mu_e}$. This vanishes unless
every $\mu_e$ is even, and then
\[
\Big|\E\prod_iM_i\Big|=\prod_{e\in\mathcal S}(\mu_e-1)!!\,D^{-\mu_e/2}
\le(jk)!\,D^{-|\mathcal S|},
\]
since $\mu_e\ge2$ for $e\in\mathcal S$. If $\mathcal S$ is empty, all $M_i$
are constant, and the tuple contributes at most $C_0^j$. Otherwise
$\mathcal S$ is connected and contains an edge at $x$, because each nonempty
$\supp M_i$ does and contains $x$; it spans some number
$2\le v\le jk+1$ of vertices, and $|\mathcal S|\ge v-1$. Given $\mathcal S$,
there are at most $(k+1)^{j|\mathcal S|}\le(k+1)^{j^2k}$ tuples of monomials
of degree at most $k$ supported in $\mathcal S$. By \cref{lem:amp-counting},
\[
|\E[Y_1\cdots Y_j]|\le C_0^j\Big(1+\sum_{v=2}^{jk+1}4^v2^{v^2}D^{v-1}
(k+1)^{j^2k}(jk)!\,D^{-(v-1)}\Big),
\]
which depends only on $j$, $k$, and $C_0$. The second statement follows by
taking $Y_1=\dots=Y_{2p'}=Y$ for an integer $p'\ge p/2$ and using
H\"older's inequality.
\end{proof}
The next lemma replaces squared couplings by their variances in an Onsager
coefficient. For dense matrices, the corresponding fourth-moment bound is
proved in~\cite[Appendix~A.2]{BayatiLelargeMontanari2015}. Hachem stated a
version for sparse variance profiles in~\cite[Lemma~21]{Hachem2024} and
omitted its proof. We give a complete proof for tame polynomials, along the lines
of~\cite{BayatiLelargeMontanari2015}.
\begin{lemma}\label{lem:amp-coefficient}
Let $x\in V$, and for every neighbor $y$ of $x$ let $Y_y$ be tame rooted at
$y$, of type $(k,C_0)$. Then
\[
\E\Big(\sum_{y\sim x}\big(W_{xy}^2-D^{-1}\big)Y_y\Big)^4\le\frac{C}{D^2},
\]
where $C$ depends only on $k$ and $C_0$.
\end{lemma}
\begin{proof}
Expanding the fourth power and each $Y_{y}$, we write the left side as a sum
over $4$-tuples $(y_a,M_a)_{a\le4}$, where $y_a\sim x$ and $M_a$ is a monomial
of $Y_{y_a}$, of
\[
\prod_{a=1}^4c^{(y_a)}_{M_a}\cdot
\E\prod_{a=1}^4\big(W_{xy_a}^2-D^{-1}\big)M_a .
\]
Fix a tuple. Let $e_a=xy_a$, let
$\mathcal S=\bigcup_{a}(\supp M_a\cup\{e_a\})$, and for $e\in\mathcal S$ let
$\nu_e=|\{a:e_a=e\}|$ and $\mu_e=\sum_ak_e(M_a)$. By independence,
\[
\E\prod_{a=1}^4\big(W_{xy_a}^2-D^{-1}\big)M_a
=\prod_{e\in\mathcal S}w_e,\qquad
w_e=\E\big[W_e^{\mu_e}\big(W_e^2-D^{-1}\big)^{\nu_e}\big].
\]
Writing $W_e=D^{-1/2}g_e$, we have
$w_e=D^{-(\mu_e+2\nu_e)/2}\E[g_e^{\mu_e}(g_e^2-1)^{\nu_e}]$. We claim that
either some $w_e$ vanishes, or
\begin{equation}\label{eq:amp-weight}
\Big|\prod_{e\in\mathcal S}w_e\Big|\le C\,D^{-|\mathcal S|-2}.
\end{equation}
Suppose that no $w_e$ vanishes. Then every $\mu_e$ is even. If $\nu_e=0$,
then $\mu_e\ge2$, since $e\in\mathcal S$, and $|w_e|\le CD^{-1}$. If $\nu_e=1$, then $\mu_e\ge2$, since $\E[g_e^2-1]=0$, and $|w_e|\le CD^{-2}$
because $\mu_e+2\nu_e\ge4$. If $\nu_e\ge2$, then
$|w_e|\le CD^{-\nu_e}$. In all cases with $\nu_e\ge1$, we obtain
$|w_e|\le CD^{-1-\theta_e}$ with $\theta_e=\max(1,\nu_e-1)$. Since
$\sum_e\nu_e=4$, either a single edge has $\nu_e=4$ and $\theta_e=3$, or at
least two edges have $\nu_e\ge1$; in both cases
$\sum_{e:\nu_e\ge1}\theta_e\ge2$. This proves \eqref{eq:amp-weight}.
Each set $\supp M_a\cup\{e_a\}$ is connected and contains the edge $e_a$ at
$x$, because $M_a$ is tame rooted at $y_a$ and $e_a=xy_a$. It follows that
$\mathcal S$ is connected, contains an edge at $x$, has at most $4(k+1)$
edges, and spans some number $2\le v\le4k+5$ of vertices, with
$|\mathcal S|\ge v-1$. Given $\mathcal S$, the number of tuples is at most
$|\mathcal S|^4(k+1)^{4|\mathcal S|}\le C$. By \cref{lem:amp-counting} and
\eqref{eq:amp-weight}, the left side of the lemma is at most
\[
C_0^4\sum_{v=2}^{4k+5}4^v2^{v^2}D^{v-1}\cdot C\cdot C\,D^{-(v-1)-2}
\le\frac{C}{D^2}. \qedhere
\]
\end{proof}
\section{Proof of the state evolution}\label{sec:amp-SE-proof}
We compare \eqref{eq:amp-recursion} with the recursion in which the Onsager
coefficients are computed from the disorder: $\tilde u^1=f_0W\mathbf 1$ and,
for $1\le t\le T-1$,
\begin{equation}\label{eq:amp-sampled}
\tilde u^{t+1}=Wf_t(\bar{\tilde u}^t)
-\sum_{s=1}^t\tilde b_{t,s}\odot f_{s-1}(\bar{\tilde u}^{s-1}),\qquad
(\tilde b_{t,s})_x=\sum_{y}W_{xy}^2\,\partial_sf_t(\bar{\tilde u}^t_y).
\end{equation}
This recursion is used only in the proof; it is not a query history.
\begin{lemma}\label{lem:amp-iterates-tame}
Let $G\in\cG_D$, $x\in V$, and $1\le t\le T$. Every coordinate of $\bar u^t_x$
and of $\bar{\tilde u}^t_x$ is tame rooted at $x$, of a type depending only on
$t$ and $f_0,\dots,f_{T-1}$. For every polynomial
$\psi\colon\R^{2t}\to\R$ and every $p\ge1$, we have
\[
\sup_{D\ge1}\sup_{G\in\cG_D}\max_{x\in V}
\E|\psi(\bar u^t_x,\bar{\tilde u}^t_x)|^p<\infty .
\]
In particular, \cref{prop:amp-SE}(a) holds.
\end{lemma}
\begin{proof}
We induct on $t$. The vectors $u^1_x=\tilde u^1_x=f_0\sum_{y\sim x}W_{xy}$
are tame rooted at $x$, by \cref{lem:amp-tame-closure}(b) with $Y_y=f_0$.
Suppose that the claim holds up to time $t$. By
\cref{lem:amp-tame-closure}(a), $f_t(\bar u^t_y)$, $f_t(\bar{\tilde u}^t_y)$,
and $\partial_sf_t(\bar{\tilde u}^t_y)$ are tame rooted at $y$, and
$f_{s-1}(\bar u^{s-1}_x)$ and $f_{s-1}(\bar{\tilde u}^{s-1}_x)$ are tame
rooted at $x$. By \cref{lem:amp-tame-closure}(b),
$\sum_{y\sim x}W_{xy}f_t(\bar u^t_y)$,
$\sum_{y\sim x}W_{xy}f_t(\bar{\tilde u}^t_y)$, and
$(\tilde b_{t,s})_x=\sum_{y\sim x}W_{xy}^2\partial_sf_t(\bar{\tilde u}^t_y)$
are tame rooted at $x$. Then \cref{lem:amp-tame-closure}(a), applied to
\eqref{eq:amp-recursion} and \eqref{eq:amp-sampled}, implies that $u^{t+1}_x$
and $\tilde u^{t+1}_x$ are tame rooted at $x$. The types depend only on $t$,
on the polynomials, and on the numbers $b_{t,s}$, which are determined by the
polynomials. The moment bound follows from \cref{lem:amp-tame-closure}(a) and
\cref{lem:amp-tame-moments}.
\end{proof}
The next lemma identifies \eqref{eq:amp-sampled} as an instance of
\eqref{eq:amp-hachem}. The vector iterate stores the history of the scalar
iterates; once a coordinate has appeared, the recursion recomputes the same
value at every later step. The same reduction of a recursion with memory to a
vector recursion appears in~\cite[Appendix~A]{Montanari2021}, where it is
combined with~\cite[Theorem~1]{JavanmardMontanari2013}.
\begin{lemma}\label{lem:amp-embed}
Let $(G_d)$ be the tori $\T^d_L$ for a fixed $L\ge3$ or the hypercubes $Q_d$.
Let $1\le t\le T$, let $\psi\colon\R^t\to\R$ be a polynomial, and let
$x=x_d\in V$. Then, as $d\to\infty$,
\[
\frac1D\sum_{y\sim x}\psi(\bar{\tilde u}^t_y)\to\E\psi(\bar U^t)
\quad\text{and}\quad
\frac1n\sum_{y\in V}\psi(\bar{\tilde u}^t_y)\to\E\psi(\bar U^t)
\]
in probability.
\end{lemma}
\begin{proof}
The graphs $G_d$ satisfy the hypotheses of \cref{thm:amp-hachem}, with $k=d$,
$n_k=n$, $D_k=D$, and $S^{(k)}=S$: each row of $S$ has exactly $D$ nonzero
entries, all equal to $D^{-1}$, and $D\le n$. Further, $W^{(k)}$ has the law
of $W$, if we let $g_{xy}$ for nonadjacent $x,y$ be independent dummy
variables. Let $q=T$, and for $0\le t'\le T-1$ define
$\mathbf f_{t'}\colon\R^T\to\R^T$ by
\[
\mathbf f_{t',r}(z)=f_{r-1}(z_1,\dots,z_{r-1})\quad(r\le t'+1),\qquad
\mathbf f_{t',r}(z)=0\quad(r>t'+1).
\]
We claim that $\mathbf z^{t'}_x=(\tilde u^1_x,\dots,\tilde u^{t'}_x,0,\dots,0)$
for $0\le t'\le T$ and $x\in V$. This holds for $t'=0$, and for $t'=1$ since
$\mathbf f_0(0)=(f_0,0,\dots,0)$. Suppose it holds at times $t'-1$ and $t'$,
for some $1\le t'\le T-1$. If $r>t'+1$, then $\mathbf f_{t',r}=0$, so
$\mathbf z^{t'+1}_x(r)=0$. Let $r\le t'+1$. Then
$\mathbf f_{t',r}(\mathbf z^{t'}_y)=f_{r-1}(\bar{\tilde u}^{r-1}_y)$, because
$r-1\le t'$. The derivative $\partial_s\mathbf f_{t',r}$ equals
$\partial_sf_{r-1}$ for $s\le r-1$ and vanishes for $s\ge r$, and for
$s\le r-1\le t'$ we have
$\mathbf f_{t'-1,s}(\mathbf z^{t'-1}_x)=f_{s-1}(\bar{\tilde u}^{s-1}_x)$.
Inserting these identities into \eqref{eq:amp-hachem}, we find
\[
\mathbf z^{t'+1}_x(r)=\sum_yW_{xy}f_{r-1}(\bar{\tilde u}^{r-1}_y)
-\sum_{s=1}^{r-1}f_{s-1}(\bar{\tilde u}^{s-1}_x)\sum_yW_{xy}^2\,
\partial_sf_{r-1}(\bar{\tilde u}^{r-1}_y)=\tilde u^r_x,
\]
by \eqref{eq:amp-sampled} at time $r-1$ (for $r=1$, by $\tilde u^1=f_0W\mathbf 1$).
This proves the claim.
Next, we claim that $\Xi^{t'}_x$ is the covariance matrix of
$(U_1,\dots,U_{t'},0,\dots,0)$ for all $x$. For $t'=1$ this holds because
$\sum_yS_{xy}=1$ and $\E U_1^2=f_0^2$. If it holds at time $t'$, then
$\boldsymbol\zeta^{t'}_y$ has the law of $(\bar U^{t'},0,\dots,0)$ for every
$y$, and, using $\sum_yS_{xy}=1$ again, $\Xi^{t'+1}_x$ has entries
$\E f_{r-1}(\bar U^{r-1})f_{r'-1}(\bar U^{r'-1})=\E U_rU_{r'}$ for
$r,r'\le t'+1$, by \eqref{eq:amp-SE}, and zero entries otherwise.
Finally, we apply the local conclusion of \cref{thm:amp-hachem}(ii) at time
$t$ to the polynomial $z\mapsto\psi(z_1,\dots,z_t)$, with $I_k$ the set of
neighbors of $x_d$, which has $D$ elements. For the vertex average, we use the
separate global conclusion of that theorem. By the two claims,
$\psi(\mathbf z^t_y)=\psi(\bar{\tilde u}^t_y)$ and
$\E\psi(\boldsymbol\zeta^t_y)=\E\psi(\bar U^t)$ for all $y$.
\end{proof}
We next use the symmetry of tori and hypercubes.
\begin{lemma}\label{lem:amp-symmetry}
Let $G=G_d$, let $1\le t\le T$, let $\psi\colon\R^{2t}\to\R$ be measurable,
and set $\xi_y=\psi(\bar u^t_y,\bar{\tilde u}^t_y)$ for $y\in V$. Then the
laws of $\xi_x$, of $D^{-1}\sum_{y\sim x}\xi_y$, and of $(W\xi)_x$ do not
depend on $x\in V$.
\end{lemma}
\begin{proof}
Let $\varphi$ be an automorphism of $G$, and let
$W^\varphi_{xy}=W_{\varphi(x)\varphi(y)}$. Since $\varphi$ permutes the edges
and the couplings are independent and identically distributed, $W^\varphi$
has the law of $W$. The recursions \eqref{eq:amp-recursion} and
\eqref{eq:amp-sampled} use the same polynomials and the same real
coefficients at every vertex. By induction on $t$, computing them from
$W^\varphi$ instead of $W$ produces the vectors $(u^t_{\varphi(x)})_x$ and
$(\tilde u^t_{\varphi(x)})_x$; for instance,
$\sum_yW^\varphi_{xy}f_t(\bar u^t_{\varphi(y)})
=\sum_{y'}W_{\varphi(x)y'}f_t(\bar u^t_{y'})$. Writing $\xi(W)$ for the
vector $\xi$ computed from $W$, we obtain $\xi(W^\varphi)_x=\xi(W)_{\varphi(x)}$,
\[
\frac1D\sum_{y\sim x}\xi(W^\varphi)_y=\frac1D\sum_{y'\sim\varphi(x)}\xi(W)_{y'},
\qquad
\big(W^\varphi\xi(W^\varphi)\big)_x=\big(W\xi(W)\big)_{\varphi(x)}.
\]
Since $G$ is vertex-transitive, for all $x,x'$ there exists an automorphism
$\varphi$ with $\varphi(x)=x'$, and the lemma follows.
\end{proof}
\begin{lemma}\label{lem:amp-neighbour}
Let $(G_d)$ be the tori $\T^d_L$ for a fixed $L\ge3$ or the hypercubes $Q_d$.
For $1\le s\le t\le T-1$ and $x\in V$,
$\lim_{d\to\infty}\E|(\tilde b_{t,s})_x-b_{t,s}|^4=0$.
\end{lemma}
\begin{proof}
Since $S_{xy}=D^{-1}$ for $y\sim x$ and $W_{xy}=0$ otherwise,
\[
(\tilde b_{t,s})_x-b_{t,s}
=\sum_{y\sim x}\big(W_{xy}^2-D^{-1}\big)\partial_sf_t(\bar{\tilde u}^t_y)
+\Big(\frac1D\sum_{y\sim x}\partial_sf_t(\bar{\tilde u}^t_y)-b_{t,s}\Big).
\]
By \cref{lem:amp-iterates-tame,lem:amp-tame-closure}, the polynomial
$\partial_sf_t(\bar{\tilde u}^t_y)$ is tame rooted at $y$, of a type that
does not depend on $d$ or $y$. Then \cref{lem:amp-coefficient} implies that
the fourth moment of the first term is at most $CD^{-2}$. The second term tends to zero
in probability by \cref{lem:amp-embed} with $\psi=\partial_sf_t$, since
$b_{t,s}=\E\partial_sf_t(\bar U^t)$. Its eighth moment is bounded uniformly in
$d$, by Jensen's inequality and \cref{lem:amp-iterates-tame}, so its fourth
power is uniformly integrable, and its fourth moment tends to zero. The
lemma follows from Minkowski's inequality.
\end{proof}
\begin{proof}[Proof of \cref{prop:amp-SE}]
Part (a) is contained in \cref{lem:amp-iterates-tame}. For the proof of (b),
we use two elementary facts. First, if random variables $\xi_d$ satisfy
$\E\xi_d^2\to0$ and $\sup_d\E|\xi_d|^{2p}<\infty$ for every $p\ge1$, then
$\E|\xi_d|^p\le(\E\xi_d^2)^{1/2}(\E|\xi_d|^{2p-2})^{1/2}\to0$ for every
$p\ge1$. Second, for every polynomial $P\colon\R^t\to\R$ there exist $C,k$
with $|P(z)-P(z')|\le C|z-z'|(1+|z|^k+|z'|^k)$.
\emph{Step 1: comparison.} Let $w^t=u^t-\tilde u^t$. We show by induction on
$t\le T$ that $\E(w^t_x)^2\to0$ for all $x$. By \cref{lem:amp-symmetry},
the law of $w^t_x$ does not depend on $x$. We have $w^1=0$. Suppose that
$\E(w^s_x)^2\to0$ for all $s\le t$, where $t\le T-1$. By
\cref{lem:amp-iterates-tame} and the two facts above, for every polynomial
$P\colon\R^t\to\R$ and every $p\ge1$,
\begin{equation}\label{eq:amp-compare}
\E|P(\bar u^t_x)-P(\bar{\tilde u}^t_x)|^p
\le C\big(\E|\bar u^t_x-\bar{\tilde u}^t_x|^{2p}\big)^{1/2}
\big(\E(1+|\bar u^t_x|^k+|\bar{\tilde u}^t_x|^k)^{2p}\big)^{1/2}\to0 .
\end{equation}
By \eqref{eq:amp-recursion} and \eqref{eq:amp-sampled},
$w^{t+1}=W\xi-\sum_{s=1}^t\rho^s$, where $\xi=f_t(\bar u^t)-f_t(\bar{\tilde u}^t)$
and
\[
\rho^s=b_{t,s}\big(f_{s-1}(\bar u^{s-1})-f_{s-1}(\bar{\tilde u}^{s-1})\big)
+\big(b_{t,s}\mathbf 1-\tilde b_{t,s}\big)\odot f_{s-1}(\bar{\tilde u}^{s-1}).
\]
By \cref{lem:amp-symmetry} and the Cauchy--Schwarz inequality,
\[
\E(W\xi)_x^2=\frac1n\E\|W\xi\|^2
\le\big(\E\|W\|^4\big)^{1/2}\Big(\E\Big(\frac1n\sum_y\xi_y^2\Big)^2\Big)^{1/2}
\le\big(\E\|W\|^4\big)^{1/2}\big(\E\xi_x^4\big)^{1/2},
\]
where in the last inequality we used Jensen's inequality and that the law of
$\xi_y$ does not depend on $y$. The right side tends to zero, by
\cref{lem:amp-norm} and \eqref{eq:amp-compare}. The first part of $\rho^s_x$
tends to zero in $L^2$ by \eqref{eq:amp-compare}. For the second part, we
have
\[
\E\big[(b_{t,s}-(\tilde b_{t,s})_x)^2f_{s-1}(\bar{\tilde u}^{s-1}_x)^2\big]
\le\big(\E|b_{t,s}-(\tilde b_{t,s})_x|^4\big)^{1/2}
\big(\E f_{s-1}(\bar{\tilde u}^{s-1}_x)^4\big)^{1/2}\to0,
\]
by \cref{lem:amp-neighbour,lem:amp-iterates-tame}. We conclude that
$\E(w^{t+1}_x)^2\to0$, which completes the induction. In particular,
\eqref{eq:amp-compare} holds for all $t\le T$.
\emph{Step 2: polynomial test functions.} Let $\psi\colon\R^t\to\R$ be a
polynomial. By \cref{lem:amp-symmetry},
\[
\E\Big|\frac1n\sum_y\psi(\bar u^t_y)-\E\psi(\bar U^t)\Big|
\le\E|\psi(\bar u^t_x)-\psi(\bar{\tilde u}^t_x)|
+\E\Big|\frac1n\sum_y\psi(\bar{\tilde u}^t_y)-\E\psi(\bar U^t)\Big|.
\]
The first term tends to zero by \eqref{eq:amp-compare}. The random variable
in the second term tends to zero in probability by \cref{lem:amp-embed}, and
it tends to zero in $L^1$ because its second moment is bounded uniformly in
$d$ by Jensen's inequality and \cref{lem:amp-iterates-tame}. The same
argument applies to $D^{-1}\sum_{y\sim x}\psi(\bar u^t_y)$. Finally, by
\cref{lem:amp-symmetry},
$\E\psi(\bar u^t_x)=\E[n^{-1}\sum_y\psi(\bar u^t_y)]\to\E\psi(\bar U^t)$.
This proves (b) for polynomials.
\emph{Step 3: continuous test functions.} By Step 2, all moments of
$\bar u^t_x$ converge to those of the Gaussian vector $\bar U^t$,
and~\ref{F:moments} implies that $\bar u^t_x\to\bar U^t$ in law. Let $\varphi$ be continuous
with polynomial growth. By (a), we have $\sup_d\E\varphi(\bar u^t_x)^2<\infty$. Then the
variables $\varphi(\bar u^t_x)$ are uniformly integrable, and
$\E\varphi(\bar u^t_x)\to\E\varphi(\bar U^t)$. This is the first assertion
of (b). Now let $\psi$ be continuous with polynomial growth, and let
$\eta>0$. By~\ref{F:density}, there exists a polynomial $P$ with
$\E|\psi(\bar U^t)-P(\bar U^t)|\le\eta$. By \cref{lem:amp-symmetry},
\[
\E\Big|\frac1n\sum_y\psi(\bar u^t_y)-\E\psi(\bar U^t)\Big|
\le\E|(\psi-P)(\bar u^t_x)|+\E\Big|\frac1n\sum_yP(\bar u^t_y)-\E P(\bar U^t)\Big|
+\E|(P-\psi)(\bar U^t)|.
\]
As $d\to\infty$, the first term tends to $\E|(\psi-P)(\bar U^t)|\le\eta$, by
the first assertion of (b) applied to $\varphi=|\psi-P|$, and the second
term tends to zero by Step 2. The left side therefore has $\limsup$ at most
$2\eta$, and $\eta$ was arbitrary. The neighbor average is treated in the
same way, using
$\E|D^{-1}\sum_{y\sim x}(\psi-P)(\bar u^t_y)|\le\E|(\psi-P)(\bar u^t_x)|$.
\end{proof}
\section{Energy of the final message}\label{sec:amp-energy}
The energy of the final polynomial message is determined by the next
iterate and the Onsager term. A similar computation appears in the proof
of~\cite[Lemma~2.3]{Montanari2021}.
\begin{lemma}\label{lem:amp-energy}
In the setting of \cref{prop:amp-SE}, let $1\le K\le T-1$, and let
$F=f_K(\bar u^K)\in\R^V$ and $\mathbf F=f_K(\bar U^K)$. Then
\[
\lim_{d\to\infty}\frac1n\E H_{G_d}(F)
=\lim_{d\to\infty}\frac1{2n}\E\langle F,WF\rangle=\E[\mathbf FU_{K+1}] .
\]
\end{lemma}
\begin{proof}
Let $O=\sum_{s=1}^Kb_{K,s}f_{s-1}(\bar u^{s-1})$. By \eqref{eq:amp-recursion},
we have $WF=u^{K+1}+O$ and
\[
\frac1n\langle F,WF\rangle=\frac1n\sum_xf_K(\bar u^K_x)
\Big(u^{K+1}_x+\sum_{s=1}^Kb_{K,s}f_{s-1}(\bar u^{s-1}_x)\Big),
\]
which is the empirical average of a polynomial in $\bar u^{K+1}_x$. By
\cref{prop:amp-SE}(b), its expectation converges to
\[
\E[\mathbf FU_{K+1}]+\sum_{s=1}^Kb_{K,s}\E[\mathbf Ff_{s-1}(\bar U^{s-1})].
\]
By Gaussian integration by parts~\ref{F:ibp}, \eqref{eq:amp-SE}, and
\eqref{eq:amp-b}, we have
\[
\E[U_{K+1}f_K(\bar U^K)]=\sum_{s=1}^K\Cov(U_{K+1},U_s)\,\E\partial_sf_K(\bar U^K)
=\sum_{s=1}^K\E[f_K(\bar U^K)f_{s-1}(\bar U^{s-1})]\,b_{K,s}.
\]
We conclude that the limit equals $2\E[\mathbf FU_{K+1}]$. Since
$2H_{G_d}(F)=\langle F,WF\rangle$, this proves the lemma.
\end{proof}
The Onsager term contributes exactly half of the energy. The center is a
bounded function of the history close to $F$, and its energy is controlled
by the norm of $W$.
\begin{lemma}\label{lem:amp-wrapper}
In the setting of \cref{lem:amp-energy}, let $\theta\colon\R^K\to[-1,1]$ be
continuous, and set $m=\theta(\bar u^K)$ and $\Theta=\theta(\bar U^K)$. Let
$M_W=\sup_d(\E\|W\|^2)^{1/2}$, which is finite by \cref{lem:amp-norm}. Then
\[
\limsup_{d\to\infty}\Big|\frac1n\E H_{G_d}(m)-\E[\mathbf FU_{K+1}]\Big|
\le\frac{M_W}2\,\|\Theta-\mathbf F\|_2\big(\|\Theta\|_2+\|\mathbf F\|_2\big).
\]
\end{lemma}
\begin{proof}
Since $2(H_{G_d}(m)-H_{G_d}(F))=\langle m-F,W(m+F)\rangle$, we have
$2n^{-1}|H_{G_d}(m)-H_{G_d}(F)|\le\|W\|\,\Xi$, where
$\Xi=\|m-F\|_n(\|m\|_n+\|F\|_n)$. By the Cauchy--Schwarz inequality,
$\E[\|W\|\,\Xi]\le M_W(\E\Xi^2)^{1/2}$. By \cref{lem:amp-energy}, it
suffices to show that
$\E\Xi^2\to\|\Theta-\mathbf F\|_2^2(\|\Theta\|_2+\|\mathbf F\|_2)^2$.
Let $A_1=\|m-F\|_n^2$, $A_2=\|m\|_n^2$, and $A_3=\|F\|_n^2$. Each $A_i$ is an
empirical average $n^{-1}\sum_y\psi_i(\bar u^K_y)$, where $\psi_1=(\theta-f_K)^2$,
$\psi_2=\theta^2$, and $\psi_3=f_K^2$ are continuous and have polynomial
growth. By \cref{prop:amp-SE}(b), $A_1$, $A_2$, and $A_3$ converge in $L^1$
and in probability to $\|\Theta-\mathbf F\|_2^2$, $\|\Theta\|_2^2$,
and $\|\mathbf F\|_2^2$. Since $\Xi^2=A_1(A_2^{1/2}+A_3^{1/2})^2$, it
converges in probability to
$\|\Theta-\mathbf F\|_2^2(\|\Theta\|_2+\|\mathbf F\|_2)^2$. Further, for every
$p\ge1$, we have $A_i^p\le n^{-1}\sum_y\psi_i(\bar u^K_y)^p$ by Jensen's
inequality, and $\psi_i^p$ is bounded by the polynomial $2^p(1+f_K^2)^{\lceil p\rceil}$, because
$|\theta|\le1$. Then \cref{prop:amp-SE}(a) implies that
$\sup_d\E A_i^p<\infty$, and by H\"older's inequality,
$\sup_d\E\Xi^4<\infty$. Then $\Xi^2$ is uniformly integrable, and
$\E\Xi^2$ converges to the same limit.
\end{proof}
\section{The two-phase scheme and controls}\label{sec:amp-controls}
We now choose the polynomials $f_t$ so that the state evolution follows a
stochastic control. We use the root-then-incremental architecture of
Sellke~\cite{Sellke2021field}, with the incremental nonlinearities of
Montanari~\cite{Montanari2021}: a fixed-point iteration first produces a
field of variance $q_0>0$ and a prescribed function of it. Since
\eqref{eq:amp-SE} involves only the polynomials $f_t$, the computations of
this subsection concern Gaussian vectors alone.
For $q_0\in(0,1)$ and $\tau\in(0,1]$, let $\mathsf N=\mathsf N_{q_0,\tau}$ be
the law of $(Y,\Delta_1,\Delta_2,\dots)$, where $Y\sim N(0,q_0)$ and the
$\Delta_j$ are independent $N(0,\tau)$ variables, independent of $Y$. For a
function $p$ with $\E p(\sqrt{q_0}Z)^2<\infty$ and $c\in[0,q_0]$, let $\phi_p(c)=\E[p(A)p(A')]$, where $(A,A')$ is a centered
Gaussian pair with $\Var A=\Var A'=q_0$ and $\Cov(A,A')=c$. Writing
$A=\sqrt c\,Z_0+\sqrt{q_0-c}\,Z_1$ and $A'=\sqrt c\,Z_0+\sqrt{q_0-c}\,Z_2$
with independent standard Gaussian $Z_0,Z_1,Z_2$, we have
$\phi_p(c)=\E[(\E[p(A)\mid Z_0])^2]\ge0$.
\begin{definition}\label{def:amp-twophase}
Let $q_0\in(0,1)$, $\tau\in(0,1]$, and let $K_2\ge0$ be an integer with
$q_0+K_2\tau\le1$. Let $K_1\ge1$ be an integer and $p$ a nonconstant
polynomial with $\E p(\sqrt{q_0}Z)^2=q_0$. Let $P_j(y,z_1,\dots,z_j)$,
$0\le j0$ by the next lemma. Each $\Delta^i$ is a linear combination of
$u^1,\dots,u^{K_1+i}$ with real coefficients, so each $f_t$ is a polynomial
in $\bar u^t$, and \cref{prop:amp-SE,lem:amp-history} apply. By part (b) of
the next lemma, the notation $(Y,\Delta_1,\Delta_2,\dots)$ is consistent with
the law $\mathsf N$.
\begin{lemma}\label{lem:amp-twophase}
In the setting of \cref{def:amp-twophase}, the following hold.
\begin{enumerate}[label=(\alph*)]
\item The variables $U_1,\dots,U_{K_1+1}$ are $N(0,q_0)$ variables with a
nonsingular covariance matrix; in particular $\tau_1>0$. Further,
$\varrho_k=\E U_kU_{k+1}$ satisfies
$\varrho_1=\sqrt{q_0}\,|\E p(\sqrt{q_0}Z)|0$, and
$\xi\sim N(0,1)$ is independent of $U_1,\dots,U_{k-1}$. Conditionally on
$U_1,\dots,U_{k-1}$, the message $p(U_k)$ is a nonconstant polynomial of a
nondegenerate Gaussian variable, and it is therefore not almost surely equal
to a function of $U_1,\dots,U_{k-1}$. Since every element of $\mathfrak F_k$ is
such a function, we have $p(U_k)\notin\mathfrak F_k$. By \cref{lem:amp-isometry},
the variable $U_{k+1}$ has a nonzero component orthogonal to $\mathfrak U_k$,
which implies that $(U_1,\dots,U_{k+1})$ is nondegenerate. At $k=K_1$, the squared norm of this
component is $\tau_1$.
(b) Let $\mathcal G$ be the $\sigma$-algebra generated by
$U_1,\dots,U_{K_1}$. Then $\Delta_1\sim N(0,\tau)$ is independent of
$\mathcal G$, since the residual $U_{K_1+1}-\sum_j\gamma_jU_j$ is Gaussian,
uncorrelated with $U_1,\dots,U_{K_1}$, and of variance $\tau_1$. Now let
$1\le r\le K_2$, and assume that $\Delta_1,\dots,\Delta_r$ are independent $N(0,\tau)$ variables, independent of $\mathcal G$. Let
$\mathcal G_r$ be the $\sigma$-algebra generated by $\mathcal G$ and
$\Delta_1,\dots,\Delta_{r-1}$. By \eqref{eq:amp-incremental},
$f_{K_1+r}-f_{K_1+r-1}=P_{r-1}\Delta_r$, where $f_{K_1}=\hat F_0$. Every
message $f_j$ with $ji$. Further, $\E[P_{i-1}\Delta_i^2]=\tau\E P_{i-1}$. This gives
$\E\mathbf F^2=\E\hat F_0^2+K_2\tau\le q_0+(1-q_0)=1$. For the correlation,
we write $U_{K+1}=U_{K_1+1}+\sum_{r=2}^{K_2+1}\Delta_r$ and
$U_{K_1+1}=\sum_j\gamma_jU_j+(\tau_1/\tau)^{1/2}\Delta_1$. The term
$\hat F_0$ correlates only with $U_{K_1+1}$; the term $P_0\Delta_1$ correlates only with
$(\tau_1/\tau)^{1/2}\Delta_1$, giving $(\tau\tau_1)^{1/2}\E P_0$; and for
$i\ge2$ the term $P_{i-1}\Delta_i$ correlates only with $\Delta_i$, giving
$\tau\E P_{i-1}$. Finally, \ref{F:ibp} applied to the pair
$(U_{K_1},U_{K_1+1})$ gives
$\E[\hat F_0U_{K_1+1}]=\E[p(U_{K_1})U_{K_1+1}]=\varrho_{K_1}\E p'(U_{K_1})$.
\end{proof}
A control specifies the ideal root message, the ideal coefficients, of which
the $P_j$ are polynomial approximations, and a bounded output.
\begin{definition}\label{def:amp-control}
Let $q_0\in(0,1)$, $\tau\in(0,1]$, and $K_2\ge0$ be as in
\cref{def:amp-twophase}. A \emph{control} of step $\tau$ and length $K_2$ with
root variance $q_0$ consists of the following data.
\begin{enumerate}[label=(\alph*)]
\item A \emph{root map}: a nonconstant $C^1$ function
$f\colon\R\to[-1,1]$ with bounded derivative, such that
$\E f(\sqrt{q_0}Z)^2=q_0$ and $\phi_f(c)>c$ for every $c\in[0,q_0)$.
\item Random variables $G_j=g_j(Y,\Delta_1,\dots,\Delta_j)$, $0\le j0$. Since $f$ is continuous and
nonconstant and $N(0,q_0)$ has full support, $\|f-\E f\|_2>0$.
By~\ref{F:density}, there
exists a polynomial $Q_\star$ with
\[
\|Q_\star-f\|_2\le\min\Big\{\frac1\ell,\frac{c_\eta}{8\sqrt{q_0}},
\frac{\|f-\E f\|_2}{3}\Big\}
\]
in $L^2(N(0,q_0))$. This gives
$\|Q_\star\|_2\ge\|f\|_2-\|Q_\star-f\|_2\ge2\sqrt{q_0}/3>0$,
so we set $p=\sqrt{q_0}\,Q_\star/\|Q_\star\|_2$. Let $K_1$ be
the least $k\ge1$ with $\varrho_k>q_0-\eta$, where $\varrho_k$ is given by
\cref{lem:amp-twophase}(a).
\item \emph{Coefficient polynomials.} By~\ref{F:density}, there exist
polynomials $Q_j$ in $(y,z_1,\dots,z_j)$ with $\|Q_j-G_j\|_2\le1/\ell$ under
$\mathsf N$. Since $\|Q_j\|_2\ge1-1/\ell\ge1/2$, we set $P_j=Q_j/\|Q_j\|_2$.
\item \emph{Center.} We run the two-phase scheme of \cref{def:amp-twophase}
with $p$, $K_1$, and $(P_j)$ on $G_d$, and we set
$m=\Theta(y,\Delta^1,\dots,\Delta^{K_2})\in(-1,1)^V$. By
\cref{lem:amp-history}, applied with $\theta$ the composition of $\Theta$ with
the linear map $\bar u^K\mapsto(y,\Delta^1,\dots,\Delta^{K_2})$, the vectors
$x^t=f_{t-1}(\bar u^{t-1})$, $1\le t\le K$, and $x^{K+1}=m$ form a query
history of depth $K+1$ with center $m$.
\end{enumerate}
The update rule depends only on the control and the accuracy parameters, not
on $d$.
\begin{lemma}\label{lem:amp-root}
In the algorithm, $p$ is a nonconstant polynomial with
$\E p(\sqrt{q_0}Z)^2=q_0$ and $\|p-f\|_2\le2/\ell$, the index $K_1$ exists,
$K_1\le1+2q_0/c_\eta$, and $q_0-\eta<\varrho_{K_1}\le q_0$. For every $jq_0-\eta$ is at most
$1+2q_0/c_\eta$, and $\varrho_{K_1}\le q_0$ by \cref{lem:amp-twophase}(a).
For the coefficients, $\|Q_j\|_2\ge1-1/\ell>0$, and
$\|P_j-G_j\|_2\le|1-\|Q_j\|_2|+\|Q_j-G_j\|_2\le2/\ell$.
\end{proof}
\begin{proposition}\label{prop:amp-transfer}
Let $(G_d)$ be the tori $\T^d_L$ for a fixed $L\ge3$ or the hypercubes $Q_d$,
and let $M_W$ be as in \cref{lem:amp-wrapper}. Fix a control with root
variance $q_0$, step $\tau$, length $K_2$, value $V$, and defect $\vartheta$,
and fix accuracy parameters $\ell$ and $\eta$. Let $m$ be the center
produced by the algorithm on $G_d$, and set $\kappa_x=1-m_x^2$. Then the
following hold.
\begin{enumerate}[label=(\alph*)]
\item As $d\to\infty$, we have $n^{-1}\E\sum_xm_x\to\E\Theta$,
$n^{-1}\E\sum_x\ent(m_x)\to\E\ent(\Theta)$, and
$n^{-1}\E\langle\kappa,S\kappa\rangle\to(1-\E\Theta^2)^2$, where $\Theta$ has
law $\mu_\Theta$.
\item We have
\[
\limsup_{d\to\infty}\Big|\frac1n\E H_{G_d}(m)-V\Big|
\le2\sqrt\tau\,\mathbf 1_{\{K_2\ge1\}}+\eta+\frac4\ell
+M_W\Big(\vartheta+\frac4\ell\Big).
\]
\item For every $x\in V$, the law of $m_x$ does not depend on $x$, and it
converges weakly to $\mu_\Theta$ as $d\to\infty$.
\end{enumerate}
\end{proposition}
\begin{proof}
Let $\theta$ be as in Step (iii) of the algorithm, so that $m=\theta(\bar u^K)$.
By \cref{lem:amp-twophase}(b), $\theta(\bar U^K)=\Theta(Y,\Delta_1,\dots,\Delta_{K_2})$
has law $\mu_\Theta$, and we denote it by $\Theta$.
(a) and (c). Since the functions $\theta$ and $\ent\circ\theta$ are bounded
and continuous, the first two limits follow from \cref{prop:amp-SE}(b). Part
(c) follows in the same way, because $\varphi\circ\theta$ is bounded and
continuous for every bounded continuous $\varphi$; the law of $m_x$ does not depend on $x$ by
\cref{lem:amp-symmetry}. Let $\bar\kappa=\E(1-\Theta^2)$. We have
$(S\kappa)_x=D^{-1}\sum_{y\sim x}(1-\theta(\bar u^K_y)^2)$, and
\cref{prop:amp-SE}(b) gives $\E|(S\kappa)_x-\bar\kappa|\to0$. Since
$0\le\kappa_x\le1$, we have
\[
\Big|\frac1n\E\langle\kappa,S\kappa\rangle-\bar\kappa\,\frac1n\E\sum_x\kappa_x\Big|
\le\frac1n\sum_x\E|(S\kappa)_x-\bar\kappa|\to0,
\]
and $n^{-1}\E\sum_x\kappa_x\to\bar\kappa$. This proves (a).
(b) Let $\mathbf F=f_K(\bar U^K)=\hat F_0+\sum_{j0$ and $h>0$. By \cref{thm:pre-supporth}, we have
$\supp\nu=[q_-,q_+]$ with $00$ and $h>0$. There exists $C<\infty$,
depending only on $\beta$ and $h$, such that the following hold.
\begin{enumerate}[label=(\alph*)]
\item If $q_-=q_+$, the data above form a control with root variance $q_-$
and length $0$, and $V=\mathcal E(\beta,h)$, $\vartheta=0$, and
$\mu_\Theta=\mu_+$.
\item If $q_-0$,
we have $\lambda_j>0$. Then $G_j$ is well defined and $\E G_j^2=1$.
\emph{Admissibility.} By the last of the listed facts, $f$ is a root map.
Each $G_j$ is a continuous function of $(Y,\Delta_1,\dots,\Delta_j)$,
since $x\mapsto x+v(t_j,x)\tau$ is Lipschitz. The output $\Theta$ is a
continuous function of $(Y,\Delta_1,\dots,\Delta_{K_2})$ with values in
$(-1,1)$, and $q_-+K_2\tau=q_+\le1$. This shows that \eqref{eq:amp-euler} is a control.
\emph{Coefficient error.} For $r\in[t_j,t_{j+1}]$, splitting as before and
using $\|G_j-a(t_j,\hat X_j)\|_2=|1-\lambda_j|$, we have
\[
\|G_j-a(r,X_r)\|_2\le|1-\lambda_j|+4\beta\|\hat e_j\|_2
+4\beta\|X_{t_j}-X_r\|_2+4\sqrt3\,\beta^2\sqrt\tau\le C\sqrt\tau .
\]
\emph{Defect, output, and value.} By the martingale identity,
$M_{q_+}=M_{q_-}+\sum_{j0$, and let $(\xi_x)_{x\in V}$ be random variables with values in
$[-\Lambda,\Lambda]$. Suppose that $V$ is partitioned into $\chi$ classes such
that, within each class, the variables $\xi_x$ are independent. Then for
every $b>0$,
\[
\P\Big(\Big|\sum_{x\in V}(\xi_x-\E\xi_x)\Big|\ge nb\Big)
\le2\exp\Big(-\frac{nb^2}{2\chi\Lambda^2}\Big).
\]
\end{lemma}
\begin{proof}
Let $V_1,\dots,V_\chi$ be the classes and $\lambda>0$. By Hoeffding's
lemma~\cite{Hoeffding1963}, $\E e^{s(\xi_x-\E\xi_x)}\le e^{s^2\Lambda^2/2}$ for
$s\in\R$. By H\"older's inequality with $\chi$ factors and independence
within classes,
\[
\E\exp\Big(\lambda\sum_x(\xi_x-\E\xi_x)\Big)
\le\prod_{i=1}^\chi\Big(\prod_{x\in V_i}\E e^{\chi\lambda(\xi_x-\E\xi_x)}\Big)^{1/\chi}
\le\exp\Big(\frac{\chi\lambda^2\Lambda^2n}{2}\Big).
\]
Using Markov's inequality with $\lambda=b/(\chi\Lambda^2)$, we bound the
probability of the upper deviation by $e^{-nb^2/(2\chi\Lambda^2)}$. The lower deviation is
treated in the same way.
\end{proof}
\begin{proposition}\label{prop:amp-homog}
Let $(G_d)$ be the tori $\T^d_L$ for a fixed $L\ge3$ or the hypercubes $Q_d$.
Let $\mu$ be a probability measure on $[-1,1]$, let $k\ge1$ be an integer,
and for each $d$ let $m=m^{(d)}$ be a random vector in $[-1,1]^V$ with the
following two properties.
\begin{enumerate}[label=(\roman*)]
\item The set $V$ can be partitioned into at most $(D+1)^k$ classes such
that, within each class, the variables $m_x$ are independent.
\item The law of $m_x$ does not depend on $x\in V$, and it converges weakly
to $\mu$ as $d\to\infty$.
\end{enumerate}
Then the conclusion of \cref{prop:amp-centre}(c) holds with $\mu$ in place of
$\mu_\epsilon$.
\end{proposition}
\begin{proof}
Fix $\Upsilon$, $L_\Upsilon$, $a$, $c$, and $B$ as in
\cref{prop:amp-centre}(c), and let $\mathcal Z=a\mathbf 1+cS[0,B]^V$. Let
$\bar\Upsilon(s)=\int\Upsilon(m',s)\,\mu(\dd m')$ for $s\ge0$, and for
$r\in[0,\infty)^V$ let
\[
\mathcal A(r)=\frac1n\sum_x\big(\Upsilon(m_x,r_x)-\bar\Upsilon(r_x)\big).
\]
Both $\Upsilon(m',\cdot)$ and $\bar\Upsilon$ are $L_\Upsilon$-Lipschitz.
By the Cauchy--Schwarz inequality, we have, for all $r,r'\in[0,\infty)^V$,
\begin{equation}\label{eq:amp-A-lip}
|\mathcal A(r)-\mathcal A(r')|\le\frac{2L_\Upsilon}n\sum_x|r_x-r'_x|
\le2L_\Upsilon\|r-r'\|_n .
\end{equation}
Let $M=a+cB$ and
$\Lambda=1+\max_{[-1,1]\times[0,M]}|\Upsilon|$, so that $|\mathcal A(r)|\le2\Lambda$
for $r\in[0,M]^V$. Fix auxiliary parameters $\alpha\in(0,1]$, $\zeta>0$, and
$\xi>0$, and let $\operatorname{clip}\colon\R^V\to[0,M]^V$ be the
coordinatewise projection onto $[0,M]$.
\emph{Step 1: reduction to a ball in the range of $\Pi_\alpha$.} Let
$r\in\mathcal Z$, and write $r=a\mathbf 1+cSu$ with $u\in[0,B]^V$. Since $S$
is nonnegative with unit row sums, $r\in[0,M]^V$. Let $\Pi_\alpha$ and
$k_\alpha$ be as in \cref{lem:amp-rank}. Since $S\mathbf 1=\mathbf 1$, the vector $\mathbf 1$
lies in the range of $\Pi_\alpha$. Let $r_H=a\mathbf 1+c\,\Pi_\alpha Su$.
Then $r_H$ belongs to the set
\[
\mathcal K_\alpha=\{w\in\operatorname{range}\Pi_\alpha:\|w\|_n\le M\},
\]
because $\|\Pi_\alpha Su\|_n\le\|u\|_n\le B$. Further,
$\|r-r_H\|_n=c\|(I-\Pi_\alpha)Su\|_n\le c\alpha B$, since every eigenvalue
of $S$ on the range of $I-\Pi_\alpha$ has absolute value less than $\alpha$.
Since $\operatorname{clip}$ is $1$-Lipschitz in each coordinate and fixes
$r$, we also have $\|r-\operatorname{clip}(r_H)\|_n\le c\alpha B$. By
\eqref{eq:amp-A-lip},
\begin{equation}\label{eq:amp-homog-step1}
\sup_{r\in\mathcal Z}|\mathcal A(r)|
\le2L_\Upsilon c\alpha B
+\sup_{w\in\mathcal K_\alpha}|\mathcal A(\operatorname{clip}(w))| .
\end{equation}
\emph{Step 2: a net.} The set $\mathcal K_\alpha$ is a Euclidean ball of
radius $M$ in the $k_\alpha$-dimensional space $\operatorname{range}\Pi_\alpha$,
with the norm $\|\cdot\|_n$. A maximal $\zeta$-separated subset
$\mathcal N\subset\mathcal K_\alpha$ is a $\zeta$-net, and the balls of
radius $\zeta/2$ around its points are disjoint and contained in the ball of
radius $M+\zeta/2$; comparing volumes gives
$|\mathcal N|\le(1+2M/\zeta)^{k_\alpha}$. For $w\in\mathcal K_\alpha$ and
$w'\in\mathcal N$ with $\|w-w'\|_n\le\zeta$, we have
$\|\operatorname{clip}(w)-\operatorname{clip}(w')\|_n\le\zeta$. Then
\eqref{eq:amp-A-lip} implies that
\begin{equation}\label{eq:amp-homog-step2}
\sup_{w\in\mathcal K_\alpha}|\mathcal A(\operatorname{clip}(w))|
\le2L_\Upsilon\zeta+\max_{w\in\mathcal N}|\mathcal A(\operatorname{clip}(w))| .
\end{equation}
\emph{Step 3: a fixed field.} Let $r\in[0,M]^V$ be deterministic. By (ii), the law $\lambda_d$ of $m_x$
does not depend on $x$. Then
\[
|\E\mathcal A(r)|\le\varepsilon_d,\qquad
\varepsilon_d=\sup_{s\in[0,M]}\Big|\int\Upsilon(m',s)\,\lambda_d(\dd m')
-\int\Upsilon(m',s)\,\mu(\dd m')\Big| .
\]
For each $s$, the difference inside the supremum tends to zero by (ii),
since the function $\Upsilon(\cdot,s)$ is bounded and continuous on
$[-1,1]$. Both integrals are $L_\Upsilon$-Lipschitz in $s$, and a finite net
of $[0,M]$ shows that $\varepsilon_d\to0$. Next, by (i), the variables
$\Upsilon(m_x,r_x)\in[-\Lambda,\Lambda]$ satisfy the hypothesis of
\cref{lem:amp-hoeffding} with $\chi\le(D+1)^k$ classes, and we obtain
\[
\P\big(|\mathcal A(r)-\E\mathcal A(r)|\ge\xi\big)
\le2\exp\Big(-\frac{n\xi^2}{2(D+1)^k\Lambda^2}\Big).
\]
\emph{Step 4: union bound.} By Step 3, the bound on $|\mathcal N|$, and
\cref{lem:amp-rank},
\[
\P\Big(\max_{w\in\mathcal N}|\mathcal A(\operatorname{clip}(w))|\ge\varepsilon_d+\xi\Big)
\le2\exp\Big(n\Big[2e^{-d\alpha^2/2}\log\Big(1+\frac{2M}\zeta\Big)
-\frac{\xi^2}{2(D+1)^k\Lambda^2}\Big]\Big).
\]
Since $D\le2d$, we have $e^{-d\alpha^2/2}(D+1)^k\to0$, and the bracket is at
most $-\xi^2/(4(D+1)^k\Lambda^2)$ for large $d$. Since $n\ge2^d$, we have
$n/(D+1)^k\to\infty$, and the probability tends to zero. Since
$|\mathcal A(\operatorname{clip}(w))|\le2\Lambda$, we obtain
\[
\limsup_{d\to\infty}\E\max_{w\in\mathcal N}|\mathcal A(\operatorname{clip}(w))|
\le\limsup_{d\to\infty}\Big(\varepsilon_d+\xi
+2\Lambda\,\P\Big(\max_{w\in\mathcal N}|\mathcal A(\operatorname{clip}(w))|\ge\varepsilon_d+\xi\Big)\Big)=\xi .
\]
Combining this with \eqref{eq:amp-homog-step1} and
\eqref{eq:amp-homog-step2}, we find
\[
\limsup_{d\to\infty}\E\sup_{r\in\mathcal Z}|\mathcal A(r)|
\le2L_\Upsilon c\alpha B+2L_\Upsilon\zeta+\xi .
\]
The left side does not depend on $\alpha$, $\zeta$, and $\xi$, which may be
taken arbitrarily small. This completes the proof.
\end{proof}
\section{Proof of \texorpdfstring{\cref{prop:amp-centre}}{the main proposition of this section}}\label{sec:amp-proof}
\begin{proof}[Proof of \cref{prop:amp-centre}]
Fix $\beta>0$, $h>0$, and $\epsilon>0$, and let $M_W$ be as in
\cref{lem:amp-wrapper}. By \cref{lem:pre-parisiTAP}(a), (b), and (c), we
have $\E M_{q_+}^2=q_+$, $M_{q_+}=\tanh X_{q_+}$, and
\begin{equation}\label{eq:amp-TAP}
\PSK(\beta,h)=\beta\mathcal E(\beta,h)+h\E M_{q_+}+\E\ent(M_{q_+})
+\frac{\beta^2}4(1-q_+)^2 .
\end{equation}
We choose the control of \cref{prop:amp-parisi-control}. If $q_-=q_+$, it is
the control of part (a), and we set $\tau=1$. If $q_-0$ and $\beta^2\E\operatorname{sech}^2(\beta\sqrt qZ+h)\le1$, where $q=\E\tanh^2(\beta\sqrt qZ+h)$ and $Z$ is a standard Gaussian variable. Fan and Wu adapted Bolthausen's argument to couplings with an orthogonally invariant law~\cite{FanWu2021RS}. These works use a conditional second-moment method. Here the gap between the conditional quenched and annealed free energies is the relative entropy above, and we bound it through the planted model. For the SK model in zero field, the normalized partition function is likewise the likelihood ratio of a planted model with respect to the null model, and El Alaoui, Montanari, and Sellke used this together with contiguity at high temperature~\cite[Section~4.1]{ElAlaouiMontanariSellke2022sampling}. Planting that preserves the typical properties of the null ensemble is called quiet~\cite{KrzakalaZdeborova2009quiet}.
Restricting the free energy to a thin band around the output of an algorithm was used by Fan, Li, and Sen \cite{FanLiSen2022} for spin glasses with orthogonally invariant couplings. Bands around a magnetization also appear in Subag's TAP representation of the free energy of spherical models~\cite{Subag2018FreeEnergyLandscapes}, and they define the generalized TAP free energy of Chen, Panchenko, and Subag~\cite{CPSGeneralizedTAP}. Huang and Sellke~\cite{HuangSellke2023constructive} proved the lower bound in the Parisi formula for spherical models by constructing an ultrametric tree of states, partly with an optimization algorithm, and bounding the free energy of bands around its leaves. For mixed $p$-spin models with Ising spins, in any field, Chen and Panchenko proved a TAP representation of the free energy~\cite[Theorem~1]{ChenPanchenko2018TAP}. For the SK model, let $q_+$ be the largest point of the support of the Parisi measure, as in \cref{sec:pre-SK}. Then their theorem states that $\PSK(\beta,h)$ is the limit, as $N\to\infty$ and then $\varepsilon\downarrow0$, of the expected maximum of the TAP free energy over magnetizations with self-overlap at least $q_+-\varepsilon$. In particular, the TAP free energy at a magnetization with self-overlap at least $q_+$ is asymptotically at most $\PSK(\beta,h)$, and \cref{thm:band} is an inequality of the same type on $G$, under different conditions on $m$.
% CHECK: ChenPanchenko2018TAP Theorem 1 verified in arXiv:1709.03468v2; ElAlaouiMontanariSellke2022sampling
% Section 4.1 (Proposition 4.2) in arXiv:2203.05093v2; BrenneckeYau2022 condition (1.5) in arXiv:2109.07354v2.
\Cref{sec:bd-statement} states \cref{def:planted} and \cref{thm:band}. \Cref{sec:bd-conditioning} describes the law of the disorder given a query history. \Cref{sec:bd-expansion} expands the partition function around the center, and \Cref{sec:bd-band} restricts the expansion to a set of configurations of large product probability on which the error terms are small. \Cref{sec:bd-entropy} bounds the remaining Gaussian likelihood in terms of $\mathcal D(m)$, and \Cref{sec:bd-proof} proves \cref{thm:band}.
\section{The planted model and the main result}\label{sec:bd-statement}
We use the query histories of \cref{def:legal}. We recall that a query history of depth $T$ on $G$ consists of a random variable $\mathsf U$ independent of $(g_e)_{e\in E}$ and random vectors $x^1,\dots,x^T\in\R^V$ such that $x^t$ is measurable with respect to the $\sigma$-algebra generated by $\mathsf U$ and $Wx^1,\dots,Wx^{t-1}$. The $\sigma$-algebra $\mathcal H$ is generated by $\mathsf U$ and $Wx^1,\dots,Wx^T$, and a center is an $\mathcal H$-measurable vector $m\in(-1,1)^V$ such that $m=x^t$ for some $t\le T$. In particular, the response $Wm$ of the disorder to the center is observed; see \cref{lem:bd-conditioning}(a).
The following definition is also used in the proof of \cref{thm:nd}.
\begin{definition}\label{def:planted}
Let $S$ be a symmetric $n\times n$ matrix with nonnegative entries, zero diagonal, and unit row sums, and let $\mathsf E=\{\{x,y\}:S_{xy}>0\}$. Fix $\beta\ge0$ and $m\in[-1,1]^n$. Let $\sigma^*$ have law $\pi_m$, set $v^*=\sigma^*-m$, and let $Z$ be a standard Gaussian vector in $\R^{\mathsf E}$ independent of $\sigma^*$. The \emph{planted product model} is the law $\mathsf Q_m$ of $Y=\beta a(v^*)+Z$, where $a(v)_{xy}=\sqrt{S_{xy}}\,v_xv_y$. We write $\mathsf P_0$ for the standard Gaussian law on $\R^{\mathsf E}$ and $\mathcal D(m)=\KL{\mathsf Q_m}{\mathsf P_0}$.
\end{definition}
For $G\in\cG_D$ we always apply \cref{def:planted} with $S=A_G/D$, and we identify $\{1,\dots,n\}$ with $V$. Then $\mathsf E=E$, the planted model lives on the edge space $\R^E$ of the disorder, and $a(v)_{xy}=D^{-1/2}v_xv_y$ for $xy\in E$. We write $\E_{\pi_m}$ for the expectation over $\sigma$ with law $\pi_m$, and in such expectations we always set $v=\sigma-m$. Since the Gaussian law on $\R^E$ with mean $b$ and identity covariance has density $\exp(\langle y,b\rangle-\|b\|^2/2)$ with respect to $\mathsf P_0$, the law $\mathsf Q_m$ has density
\begin{equation}\label{eq:bd-Lm}
L_m(y)=\E_{\pi_m}\exp\Big(\beta\langle y,a(v)\rangle-\frac{\beta^2}2\|a(v)\|^2\Big)\qquad(y\in\R^E)
\end{equation}
with respect to $\mathsf P_0$. By \cref{lem:bd-restrict} below, the function $m\mapsto\mathcal D(m)$ is continuous on $[-1,1]^V$. In particular, $\mathcal D(m)$ is a bounded random variable whenever $m$ is a random vector in $[-1,1]^V$. If $m$ is a random center, then $\mathcal D(m)$ denotes the value of this deterministic function at $m$. The planted configuration and the noise of \cref{def:planted} are integrated out in the definition of $\mathcal D$, and they have no relation to the disorder $g$.
\begin{theorem}\label{thm:band}
Let $\beta\ge0$, $h\in\R$, and $\delta\ge0$, and let $T\ge1$ be an integer. For each $k\ge1$, let $D_k\ge1$, let $G_k\in\cG_{D_k}$ have $n_k$ vertices, let $S_k=A_{G_k}/D_k$, and let a query history of depth $T$ on $G_k$ with center $m=m^{(k)}$ be given. Let $\mathcal D$ be as in \cref{def:planted} with $S=S_k$. Suppose that $D_k\to\infty$ and
\begin{equation}\label{eq:bd-ND}
\limsup_{k\to\infty}\frac1{n_k}\E\,\mathcal D\big(m^{(k)}\big)\le\delta.
\end{equation}
Then
\begin{equation}\label{eq:bd-main}
\liminf_{k\to\infty}p_{G_k}(\beta,h)\ge\liminf_{k\to\infty}\frac1{n_k}\E\Big[\beta H_{G_k}(m)+h\sum_{x}m_x+\sum_{x}\ent(m_x)+\frac{\beta^2}4\langle\kappa,S_k\kappa\rangle\Big]-4\beta\sqrt\delta,
\end{equation}
where $m=m^{(k)}$ and $\kappa_x=1-m_x^2$.
\end{theorem}
\begin{remark}\label{rem:bd-scope}
\Cref{thm:band} follows from the nonasymptotic bound of \cref{prop:bd-quant}, which holds for every $G\in\cG_D$ and every query history of depth $T$ on $G$. The only property of the sequence $(G_k)$ used in the proof is $D_k\to\infty$. No transitivity, no bound on the operator norm of $W$, and no relation between $n_k$ and $D_k$ is needed.
\end{remark}
\section{Gaussian conditioning}\label{sec:bd-conditioning}
In this subsection, we fix $G\in\cG_D$ and a query history of depth $T$ on $G$, defined on a probability space $(\Omega,\mathcal A,\P)$. We write $g=(g_e)_{e\in E}$ and $\mathscr B(\R^E)$ for the Borel $\sigma$-algebra of $\R^E$. For $0\le t\le T$, let $\mathcal F_t$ be the $\sigma$-algebra generated by $\mathsf U$ and $Wx^1,\dots,Wx^t$, so that $\mathcal F_T=\mathcal H$.
For $b\in\R^V$, let $\Lambda_b\colon\R^E\to\R^V$ be the linear map defined by
\[
(\Lambda_bw)_x=D^{-1/2}\sum_{y\colon xy\in E}w_{xy}b_y\qquad(x\in V).
\]
Then $Wb=\Lambda_bg$; each query reveals $n$ linear functionals of $g$. For $0\le t\le T$, let $\Lambda^{(t)}\colon\R^E\to(\R^V)^t$ be the map $w\mapsto(\Lambda_{x^1}w,\dots,\Lambda_{x^t}w)$, and let $\Pi_t$ be the orthogonal projection of $\R^E$ onto $\ker\Lambda^{(t)}$; in particular, $\Pi_0=I$. We set
\[
\Pi=\Pi_T,\qquad \Pi^\perp=I-\Pi,\qquad \bar g=\Pi^\perp g,
\]
and we let $\bar W$ be the symmetric $V\times V$ matrix with $\bar W_{xy}=D^{-1/2}\bar g_{xy}$ for $xy\in E$ and $\bar W_{xy}=0$ otherwise. For a symmetric matrix $K$, we write $|K|=(K^2)^{1/2}$ and $\|K\|_{\mathrm{tr}}=\tr|K|$ for its trace norm.
The next lemma is a version of the conditioning lemmas of Bolthausen~\cite{Bolthausen2014} and Bayati and Montanari~\cite[Lemmas~11 and~12]{BayatiMontanari2011}, for linear functionals of the edge disorder that are chosen adaptively and may depend on the auxiliary randomness $\mathsf U$.
% CHECK: Lemmas 11 and 12 of BayatiMontanari2011 verified in arXiv:1001.3448v4 only.
\begin{lemma}\label{lem:bd-conditioning}
Let $G\in\cG_D$, and let a query history of depth $T$ on $G$ be given. Then the following statements hold.
\begin{enumerate}[label=(\alph*)]
\item The projection $\Pi$ and the vector $\bar g$ are $\mathcal H$-measurable, and $\operatorname{rank}\Pi^\perp\le nT$. If $m$ is a center of the history, then $Wm$ is $\mathcal H$-measurable.
\item For every $\mathcal H\otimes\mathscr B(\R^E)$-measurable function $F\colon\Omega\times\R^E\to[0,\infty]$, we have
\begin{equation}\label{eq:bd-conditional-law}
\E\big[F(\cdot,g)\,\big|\,\mathcal H\big]=\int F(\cdot,\bar g+\Pi w)\,\mathsf P_0(\mathrm dw)\qquad\text{almost surely.}
\end{equation}
\item We have $\E\|\bar g\|^2=\E\operatorname{rank}\Pi^\perp\le nT$ and $\E\|\bar W\|_{\mathrm{tr}}\le n(2T/D)^{1/2}$.
\end{enumerate}
\end{lemma}
Part (b) states that, conditional on $\mathcal H$, the disorder has the law of $\bar g+\Pi z$, where $z$ is a standard Gaussian vector on $\R^E$ independent of $\mathcal H$. The function $F$ may depend on $\mathcal H$-measurable quantities, such as the center and the projection $\Pi$; this is how the lemma is used below. In particular, $\E[g\,|\,\mathcal H]=\bar g$ and $\bar W=\E[W\,|\,\mathcal H]$ is the conditional mean of the coupling matrix.
\begin{proof}
For a linear map $M$ between Euclidean spaces, let $M^+$ denote its Moore--Penrose pseudoinverse. Then $M^+M$ is the orthogonal projection onto $(\ker M)^\perp$, and $M\mapsto M^+$ is Borel measurable, being the pointwise limit of $(M^{\mathsf T}M+\epsilon I)^{-1}M^{\mathsf T}$ as $\epsilon\downarrow0$. In particular, $\Pi_t^\perp=I-\Pi_t=(\Lambda^{(t)})^+\Lambda^{(t)}$.
(a) Fix $1\le t\le T$. Since $x^s$ is $\mathcal F_{s-1}$-measurable and $b\mapsto\Lambda_b$ is linear, the map $\Lambda^{(t)}$ is $\mathcal F_{t-1}$-measurable, and so is $\Pi_t$. Further, $\Lambda^{(t)}g=(Wx^1,\dots,Wx^t)$ is $\mathcal F_t$-measurable, and so is $\Pi_t^\perp g=(\Lambda^{(t)})^+\Lambda^{(t)}g$. For $t=T$ this shows that $\Pi$ and $\bar g$ are $\mathcal H$-measurable. The rank of $\Pi^\perp$ equals that of $\Lambda^{(T)}$, which is at most $nT$, since $\Lambda^{(T)}$ takes values in a space of dimension $nT$. If $m$ is a center, let $\tau$ be the least $t$ with $m=x^t$. The events $\{\tau=t\}$ belong to $\mathcal H$, because $m$ and $x^1,\dots,x^T$ are $\mathcal H$-measurable, and $Wm=\sum_{t\le T}\mathbf 1_{\{\tau=t\}}Wx^t$ is $\mathcal H$-measurable.
(b) For $0\le t\le T$, let $(\mathrm b_t)$ denote the statement (b) with $\mathcal F_t$ and $\Pi_t$ in place of $\mathcal H$ and $\Pi$. We prove $(\mathrm b_t)$ by induction on $t$. In $(\mathrm b_t)$, the right side of \eqref{eq:bd-conditional-law} is $\mathcal F_t$-measurable by Tonelli's theorem, since $\Pi_t^\perp g$ and $\Pi_t$ are $\mathcal F_t$-measurable by the proof of (a). The bounded $\mathcal F_t\otimes\mathscr B(\R^E)$-measurable functions $F$ that satisfy \eqref{eq:bd-conditional-law} form a vector space that contains the constants and is closed under bounded monotone limits, and nonnegative $F$ are monotone limits of bounded ones. By the monotone-class theorem, it therefore suffices to prove $(\mathrm b_t)$ for $F(\omega,w)=\mathbf 1_A(\omega)f(w)$ with $A\in\mathcal F_t$ and $f\colon\R^E\to[0,\infty)$ bounded and Borel. Since $\mathbf 1_A$ is $\mathcal F_t$-measurable, this amounts to
\begin{equation}\label{eq:bd-cond-f}
\E\big[f(g)\,\big|\,\mathcal F_t\big]=\int f(\Pi_t^\perp g+\Pi_tw)\,\mathsf P_0(\mathrm dw).
\end{equation}
For $t=0$, \eqref{eq:bd-cond-f} holds because $\Pi_0=I$ and $g$ is independent of $\mathcal F_0$, which is generated by $\mathsf U$.
Let $1\le t\le T$, and assume $(\mathrm b_{t-1})$. We abbreviate $\Pi'=\Pi_{t-1}$ and $\Lambda=\Lambda_{x^t}$. Since $\ker\Lambda^{(t)}\subset\ker\Lambda^{(t-1)}$, the operator $\Xi=\Pi'-\Pi_t$ is the orthogonal projection onto the orthogonal complement of the range of $\Pi_t$ in the range of $\Pi'$. The operators $\Pi'$, $\Pi_t$, $\Xi$, and $\Lambda$ are $\mathcal F_{t-1}$-measurable, and
\begin{equation}\label{eq:bd-cond-algebra}
\Lambda\Pi_t=0,\qquad \Lambda\Pi'=\Lambda\Xi,\qquad \Pi_t^\perp\Pi'^\perp=\Pi'^\perp,\qquad \Pi_t^\perp\Pi'=\Xi.
\end{equation}
The events $A'\cap\{Wx^t\in C\}$, with $A'\in\mathcal F_{t-1}$ and $C\subset\R^V$ Borel, form a $\pi$-system that generates $\mathcal F_t$. By Dynkin's $\pi$--$\lambda$ theorem and the measurability of the right side of \eqref{eq:bd-cond-f}, it suffices to show that
\begin{equation}\label{eq:bd-cond-goal}
\E\big[\mathbf 1_{A'}\mathbf 1_C(\Lambda g)f(g)\big]=\E\big[\Psi(\cdot,g)\big],
\end{equation}
where
\[
\Psi(\omega,w)=\mathbf 1_{A'}(\omega)\mathbf 1_C(\Lambda w)\int f(\Pi_t^\perp w+\Pi_tw')\,\mathsf P_0(\mathrm dw'),
\]
and
where we used $Wx^t=\Lambda g$. The function $\Psi$ is $\mathcal F_{t-1}\otimes\mathscr B(\R^E)$-measurable. By $(\mathrm b_{t-1})$ and \eqref{eq:bd-cond-algebra}, we have
\begin{align*}
\E\big[\Psi(\cdot,g)\big]&=\E\int\Psi(\cdot,\Pi'^\perp g+\Pi'w)\,\mathsf P_0(\mathrm dw)\\
&=\E\bigg[\mathbf 1_{A'}\iint\mathbf 1_C\big(\Lambda\Pi'^\perp g+\Lambda\Xi w\big)f\big(\Pi'^\perp g+\Xi w+\Pi_tw'\big)\,\mathsf P_0(\mathrm dw)\,\mathsf P_0(\mathrm dw')\bigg].
\end{align*}
For fixed $\omega$, the Gaussian vectors $\Xi w$ and $\Pi_tw$ are independent under $\mathsf P_0$, because $\Xi\Pi_t=0$. Then the pair $(\Xi w,\Xi w+\Pi_tw')$ under $\mathsf P_0\otimes\mathsf P_0$ has the same law as $(\Xi w,\Pi'w)$ under $\mathsf P_0$. Using also $\Lambda\Xi w=\Lambda\Pi'w$, we find that the double integral equals
\[
\int\mathbf 1_C\big(\Lambda(\Pi'^\perp g+\Pi'w)\big)f\big(\Pi'^\perp g+\Pi'w\big)\,\mathsf P_0(\mathrm dw).
\]
By $(\mathrm b_{t-1})$, applied to $(\omega,w)\mapsto\mathbf 1_{A'}(\omega)\mathbf 1_C(\Lambda(\omega)w)f(w)$, the expectation of $\mathbf 1_{A'}$ times this integral is the left side of \eqref{eq:bd-cond-goal}. This proves $(\mathrm b_t)$, and (b) is the case $t=T$.
(c) By (b) with $F(\omega,w)=\|w\|^2$, and since $\bar g$ is orthogonal to the range of $\Pi$, we have
\[
\E\big[\|g\|^2\,\big|\,\mathcal H\big]=\int\|\bar g+\Pi w\|^2\,\mathsf P_0(\mathrm dw)=\|\bar g\|^2+\tr\Pi=\|\bar g\|^2+|E|-\operatorname{rank}\Pi^\perp.
\]
Taking expectations and using $\E\|g\|^2=|E|$, we find $\E\|\bar g\|^2=\E\operatorname{rank}\Pi^\perp\le nT$. Next, the squared Hilbert--Schmidt norm of $\bar W$ is $\sum_{x,y}\bar W_{xy}^2=2D^{-1}\|\bar g\|^2$. Since $\bar W$ has $n$ eigenvalues, the Cauchy--Schwarz inequality implies that $\|\bar W\|_{\mathrm{tr}}\le(2n/D)^{1/2}\|\bar g\|$. Then
\[
\E\|\bar W\|_{\mathrm{tr}}\le\Big(\frac{2n}D\Big)^{1/2}\big(\E\|\bar g\|^2\big)^{1/2}\le n\Big(\frac{2T}D\Big)^{1/2}.
\]
This completes the proof.
\end{proof}
\section{Expansion around the center}\label{sec:bd-expansion}
The following identity is deterministic. It holds for every realization of the disorder and every orthogonal projection $\Pi$ of $\R^E$, with $\bar g=\Pi^\perp g$ and $\bar W$ defined from $\bar g$ as in \cref{sec:bd-conditioning}. For $m\in(-1,1)^V$, we define $r\in\R^V$ by
\begin{equation}\label{eq:bd-r}
r_x=\beta(Wm)_x+h-\operatorname{atanh}(m_x)-\beta^2m_x(S\kappa)_x\qquad(x\in V).
\end{equation}
\begin{lemma}\label{lem:bd-expansion}
Let $G\in\cG_D$, $\beta\ge0$, $h\in\R$, and $m\in(-1,1)^V$, and let $\Pi$ be an orthogonal projection of $\R^E$. Then
\begin{equation}\label{eq:bd-expansion}
\begin{aligned}
Z_G=e^{\Theta_G(m)}\,\E_{\pi_m}\exp\Big(&\langle r,v\rangle+\frac\beta2\langle v,\bar Wv\rangle+\beta^2\langle m\odot v,S(m\odot v)\rangle-\frac{\beta^2}2\|\Pi^\perp a(v)\|^2\\
&+\beta\langle g,\Pi a(v)\rangle-\frac{\beta^2}2\|\Pi a(v)\|^2\Big).
\end{aligned}
\end{equation}
\end{lemma}
\begin{proof}
Let $y_x=\operatorname{atanh}(m_x)$. Then $\pi_m(\sigma)=\prod_x e^{y_x\sigma_x}/(2\cosh y_x)$ is positive for all $\sigma$, and for every function $f$ on $\{-1,1\}^V$ we have
\[
\sum_\sigma f(\sigma)=\E_{\pi_m}\Big[f(\sigma)\prod_{x\in V}2\cosh(y_x)\,e^{-y_x\sigma_x}\Big].
\]
By \eqref{eq:pre-ent-tanh}, we have $\log2\cosh y_x-y_xm_x=\ent(m_x)$. With $\sigma=m+v$, it follows that the product inside the expectation equals $\exp(\sum_x\ent(m_x)-\langle y,v\rangle)$. Since $W$ is symmetric, $H_G(m+v)=H_G(m)+\langle Wm,v\rangle+H_G(v)$. Applying the previous display to $f(\sigma)=\exp(\beta H_G(\sigma)+h\sum_x\sigma_x)$, we obtain
\begin{equation}\label{eq:bd-expansion-first}
Z_G=\exp\Big(\beta H_G(m)+h\sum_xm_x+\sum_x\ent(m_x)\Big)\,\E_{\pi_m}\exp\big(\langle\beta Wm+h\mathbf 1-y,v\rangle+\beta H_G(v)\big).
\end{equation}
Next, $H_G(v)=\sum_{xy\in E}D^{-1/2}g_{xy}v_xv_y=\langle g,a(v)\rangle$. We split $g=\bar g+\Pi g$. Since $\langle\bar g,a(v)\rangle=\sum_{xy\in E}D^{-1/2}\bar g_{xy}v_xv_y=\langle v,\bar Wv\rangle/2$ and $\langle\Pi g,a(v)\rangle=\langle g,\Pi a(v)\rangle$, we have
\begin{equation}\label{eq:bd-H-split}
\beta H_G(v)=\frac\beta2\langle v,\bar Wv\rangle+\beta\langle g,\Pi a(v)\rangle.
\end{equation}
Finally, we use the identity $v_x^2=\kappa_x-2m_xv_x$, which holds because $\sigma_x^2=1$. Since $S$ is symmetric with zero diagonal, a sum over edges of a symmetric expression is half the corresponding sum over ordered pairs, and we find
\begin{align*}
\|a(v)\|^2&=\sum_{xy\in E}S_{xy}(\kappa_x-2m_xv_x)(\kappa_y-2m_yv_y)\\
&=\frac12\langle\kappa,S\kappa\rangle-2\langle m\odot S\kappa,v\rangle+2\langle m\odot v,S(m\odot v)\rangle.
\end{align*}
Since $\|\Pi a(v)\|^2=\|a(v)\|^2-\|\Pi^\perp a(v)\|^2$, this gives
\begin{equation}\label{eq:bd-binary}
\frac{\beta^2}2\|\Pi a(v)\|^2=\frac{\beta^2}4\langle\kappa,S\kappa\rangle-\beta^2\langle m\odot S\kappa,v\rangle+\beta^2\langle m\odot v,S(m\odot v)\rangle-\frac{\beta^2}2\|\Pi^\perp a(v)\|^2.
\end{equation}
We insert \eqref{eq:bd-H-split} into \eqref{eq:bd-expansion-first}, and we add and subtract $(\beta^2/2)\|\Pi a(v)\|^2$ in the exponent. Using \eqref{eq:bd-binary} for the added term, moving the constant $(\beta^2/4)\langle\kappa,S\kappa\rangle$ out of the expectation, and recalling \eqref{eq:bd-r}, we obtain \eqref{eq:bd-expansion}.
\end{proof}
\section{A band of typical configurations}\label{sec:bd-band}
We begin with a moment bound for the disorder. It holds on every $G\in\cG_D$ and uses no bound on the operator norm of $W$. The corresponding bound for the Hamiltonian is \ref{F:gaussian-max}.
\begin{lemma}\label{lem:bd-max}
Let $G\in\cG_D$. Then
\[
\E\max_\sigma\|W\sigma\|^2\le8n\Big(\log2+\frac14\Big)\le8n.
\]
Further, for every realization of the disorder and every $w\in[-1,1]^V$, we have $\|Ww\|\le\max_\sigma\|W\sigma\|$.
\end{lemma}
\begin{proof}
Fix $\sigma\in\{-1,1\}^V$. Then $W\sigma$ is a centered Gaussian vector. Its coordinate $(W\sigma)_x=D^{-1/2}\sum_{y\colon xy\in E}g_{xy}\sigma_y$ has variance $1$, and for $x\neq y$ the coordinates $(W\sigma)_x$ and $(W\sigma)_y$ share only the variable $g_{xy}$, which is present if $xy\in E$. The covariance matrix of $W\sigma$ is $C_\sigma=I+\Delta_\sigma S\Delta_\sigma$, where $\Delta_\sigma$ is the diagonal matrix with entries $\sigma_x$. By \eqref{eq:pre-S}, the eigenvalues $c_1,\dots,c_n$ of $C_\sigma$ lie in $[0,2]$, and $\sum_jc_j=\tr C_\sigma=n$. Diagonalizing $C_\sigma$, we obtain
\[
\E\exp\Big(\frac18\|W\sigma\|^2\Big)=\prod_{j=1}^n\Big(1-\frac{c_j}4\Big)^{-1/2}\le\exp\Big(\sum_{j=1}^n\frac{c_j}4\Big)=e^{n/4},
\]
where we used that $-\log(1-s)\le2s$ for $s\in[0,1/2]$. By Jensen's inequality,
\[
\E\max_\sigma\|W\sigma\|^2\le8\log\E\max_\sigma e^{\|W\sigma\|^2/8}\le8\log\sum_\sigma\E e^{\|W\sigma\|^2/8}\le8\Big(n\log2+\frac n4\Big).
\]
For the final claim, let $w\in[-1,1]^V$. Since $W$ is linear, we have $Ww=\E_{\pi_w}W\sigma$, and by Jensen's inequality, $\|Ww\|\le\E_{\pi_w}\|W\sigma\|\le\max_\sigma\|W\sigma\|$.
\end{proof}
We now define the band. For the rest of this section, we write
\begin{equation}\label{eq:bd-constants}
C_{\beta,h}=4\big(1+h^2+8\beta^2+\beta^4\big),\qquad
\varepsilon_D(\eta)=\frac{C_{\beta,h}}{\eta^2D}+\frac{1+(2T)^{1/2}}{\eta D^{1/2}}+\frac T{\eta D}\qquad(\eta>0),
\end{equation}
and we note that $\varepsilon_D(\eta)$ depends only on $\beta$, $h$, $T$, $D$, and $\eta$.
\begin{lemma}\label{lem:bd-band}
Let $G\in\cG_D$, $\beta\ge0$, and $h\in\R$, and let a query history of depth $T$ on $G$ with center $m$ be given. Let $\Pi$ and $\bar W$ be as in \cref{sec:bd-conditioning}, and let $r$ be as in \eqref{eq:bd-r}. For $\eta>0$, let $B_\eta$ be the set of $\sigma\in\{-1,1\}^V$ such that $v=\sigma-m$ satisfies
\begin{equation}\label{eq:bd-band}
|\langle r,v\rangle|\le\eta n,\qquad\langle v,|\bar W|v\rangle\le\eta n,\qquad\langle m\odot v,|S|(m\odot v)\rangle\le\eta n,\qquad\|\Pi^\perp a(v)\|^2\le\eta n.
\end{equation}
Then $\{\sigma\in B_\eta\}\in\mathcal H$ for every $\sigma$, and $p_\eta=\pi_m(B_\eta)$ satisfies $\E(1-p_\eta)\le\varepsilon_D(\eta)$.
\end{lemma}
\begin{proof}
By \cref{lem:bd-conditioning}(a), the vectors $m$, $Wm$, and $r$; the projection $\Pi$; and the matrix $\bar W$ are $\mathcal H$-measurable, and $|\bar W|$ is a continuous function of $\bar W$. This proves the measurability claim, and it shows that $p_\eta$ is $\mathcal H$-measurable.
Fix a realization. Under $\pi_m$, the coordinates of $v$ are independent and centered, and $v_x$ has variance $\kappa_x$. We bound the probability that each condition in \eqref{eq:bd-band} fails. First, $\E_{\pi_m}\langle r,v\rangle^2=\sum_x\kappa_xr_x^2$, and Chebyshev's inequality bounds the probability of the first failure by $\sum_x\kappa_xr_x^2/(\eta n)^2$. Second, since $|\bar W|$ is positive semidefinite,
\[
\E_{\pi_m}\langle v,|\bar W|v\rangle=\sum_x|\bar W|_{xx}\kappa_x\le\tr|\bar W|=\|\bar W\|_{\mathrm{tr}},
\]
and Markov's inequality bounds the probability of the second failure by $\|\bar W\|_{\mathrm{tr}}/(\eta n)$. Third, in the same way,
\[
\E_{\pi_m}\langle m\odot v,|S|(m\odot v)\rangle=\sum_x|S|_{xx}m_x^2\kappa_x\le\tr|S|\le(n\tr S^2)^{1/2}=\frac n{D^{1/2}},
\]
where we used the Cauchy--Schwarz inequality for the eigenvalues of $S$ and \eqref{eq:pre-S}. Fourth, the entries of $a(v)$ are centered and uncorrelated under $\pi_m$. Indeed, for distinct edges $e,e'$, some vertex $x$ belongs to exactly one of them, and the factor $v_x$ in $a(v)_ea(v)_{e'}$ is centered and independent of the other factors. Further, $\E_{\pi_m}a(v)_{xy}^2=S_{xy}\kappa_x\kappa_y\le D^{-1}$. In other words, the covariance matrix of $a(v)$ is diagonal with entries at most $D^{-1}$, and
\[
\E_{\pi_m}\|\Pi^\perp a(v)\|^2=\sum_{e\in E}(\Pi^\perp)_{ee}\,\E_{\pi_m}a(v)_e^2\le\frac{\tr\Pi^\perp}D=\frac{\operatorname{rank}\Pi^\perp}D.
\]
By the union bound, we conclude that
\begin{equation}\label{eq:bd-band-pathwise}
1-p_\eta\le\frac{\sum_x\kappa_xr_x^2}{\eta^2n^2}+\frac{\|\bar W\|_{\mathrm{tr}}}{\eta n}+\frac1{\eta D^{1/2}}+\frac{\operatorname{rank}\Pi^\perp}{\eta nD}.
\end{equation}
It remains to bound the expectation of the first term. Let $s_x=\operatorname{atanh}(m_x)$. Then $\kappa_x=\operatorname{sech}^2s_x\le4e^{-2|s_x|}$ and $\kappa_xs_x^2\le4\sup_{s\ge0}s^2e^{-2s}=4e^{-2}\le1$. This bound is uniform in $m_x\in(-1,1)$. Using $(c_1+c_2+c_3+c_4)^2\le4(c_1^2+c_2^2+c_3^2+c_4^2)$, $\kappa_x\le1$, $|m_x|\le1$, and $0\le(S\kappa)_x\le1$, we obtain from \eqref{eq:bd-r} that
\[
\kappa_xr_x^2\le4\big(\beta^2(Wm)_x^2+h^2+\kappa_xs_x^2+\beta^4\big)\le4\big(\beta^2(Wm)_x^2+h^2+1+\beta^4\big).
\]
Summing over $x$ and using \cref{lem:bd-max}, which gives $\E\|Wm\|^2\le\E\max_\sigma\|W\sigma\|^2\le8n$, we find
\[
\E\sum_x\kappa_xr_x^2\le32\beta^2n+4n\big(h^2+1+\beta^4\big)=C_{\beta,h}n.
\]
We take expectations in \eqref{eq:bd-band-pathwise}, and we use this bound together with \cref{lem:bd-conditioning}(c). Since every $D$-regular simple graph has $n\ge D+1$ vertices, we obtain $\E(1-p_\eta)\le\varepsilon_D(\eta)$.
\end{proof}
On the band, the error terms in \cref{lem:bd-expansion} cost at most a multiple of $\eta n$. To state this, we set
\begin{equation}\label{eq:bd-c-beta}
c_\beta=1+\frac\beta2+\frac{3\beta^2}2,
\end{equation}
and on the event $\{p_\eta>0\}$ we define the function
\begin{equation}\label{eq:bd-L-eta}
L_\eta(y)=\E_{\pi_m}\Big[\exp\Big(\beta\langle y,\Pi a(v)\rangle-\frac{\beta^2}2\|\Pi a(v)\|^2\Big)\,\Big|\,B_\eta\Big]\qquad(y\in\R^E).
\end{equation}
On $\{p_\eta=0\}$, we set $L_\eta(y)=1$ for all $y$.
It depends on $y$ only through $\Pi y$, and $(\omega,y)\mapsto L_\eta(y)$ is $\mathcal H\otimes\mathscr B(\R^E)$-measurable on $\{p_\eta>0\}\times\R^E$, because it is a finite sum of continuous functions of $(m,\Pi,y)$ multiplied by $\mathcal H$-measurable indicators.
\begin{lemma}\label{lem:bd-pathwise}
Under the assumptions of \cref{lem:bd-band}, for every $\eta>0$ we have, on the event $\{p_\eta>0\}$,
\[
\log Z_G\ge\Theta_G(m)-c_\beta\eta n+\log p_\eta+\log L_\eta(g).
\]
\end{lemma}
\begin{proof}
We apply \cref{lem:bd-expansion} with the projection $\Pi$ of \cref{sec:bd-conditioning}, and we restrict the expectation in \eqref{eq:bd-expansion} to $\sigma\in B_\eta$; this decreases it, since the integrand is positive. Since a symmetric matrix $K$ satisfies $-|K|\le K\le|K|$ in the order of quadratic forms, we have, for $\sigma\in B_\eta$,
\begin{gather*}
\langle r,v\rangle\ge-\eta n,\qquad\frac\beta2\langle v,\bar Wv\rangle\ge-\frac\beta2\eta n,\\
\beta^2\langle m\odot v,S(m\odot v)\rangle\ge-\beta^2\eta n,\qquad-\frac{\beta^2}2\|\Pi^\perp a(v)\|^2\ge-\frac{\beta^2}2\eta n.
\end{gather*}
The sum of the right sides is $-c_\beta\eta n$. Therefore
\[
Z_G\ge e^{\Theta_G(m)-c_\beta\eta n}\,\E_{\pi_m}\Big[\mathbf 1_{B_\eta}(\sigma)\exp\Big(\beta\langle g,\Pi a(v)\rangle-\frac{\beta^2}2\|\Pi a(v)\|^2\Big)\Big]=e^{\Theta_G(m)-c_\beta\eta n}\,p_\eta L_\eta(g).
\]
Taking logarithms completes the proof.
\end{proof}
\section{Relative entropy of the restricted and projected model}\label{sec:bd-entropy}
In this subsection, $S$, $\mathsf E$, $\beta$, and $n$ are as in \cref{def:planted}. Since $S$ has zero diagonal and unit row sums, $\sum_{\{x,y\}\in\mathsf E}S_{xy}=n/2$. Since $|v_x|\le1+|m_x|\le2$ for $v=\sigma-m$, we have
\begin{equation}\label{eq:bd-a-bound}
\|a(v)\|^2=\sum_{\{x,y\}\in\mathsf E}S_{xy}v_x^2v_y^2\le16\sum_{\{x,y\}\in\mathsf E}S_{xy}=8n
\end{equation}
for every $m\in[-1,1]^n$ and $\sigma\in\{-1,1\}^n$.
\begin{lemma}\label{lem:bd-restrict}
The function $m\mapsto\mathcal D(m)$ is continuous on $[-1,1]^n$. Further, let $m\in[-1,1]^n$, let $\Pi$ be an orthogonal projection of $\R^{\mathsf E}$, and let $B\subset\{-1,1\}^n$ satisfy $p=\pi_m(B)>0$. Define
\[
L_{B,\Pi}(y)=\E_{\pi_m}\Big[\exp\Big(\beta\langle y,\Pi a(v)\rangle-\frac{\beta^2}2\|\Pi a(v)\|^2\Big)\,\Big|\,B\Big]\qquad(y\in\R^{\mathsf E}).
\]
Then $\mathsf Q_{B,\Pi}=L_{B,\Pi}\mathsf P_0$ is the law of $\beta\Pi a(v^*)+Z$, where $\sigma^*$ has law $\pi_m(\cdot\mid B)$, $v^*=\sigma^*-m$, and $Z$ is a standard Gaussian vector in $\R^{\mathsf E}$ independent of $\sigma^*$. Further,
\begin{equation}\label{eq:bd-restrict}
p\,\KL{\mathsf Q_{B,\Pi}}{\mathsf P_0}\le\mathcal D(m)-p\log p-(1-p)\log(1-p)\le\mathcal D(m)+\log2,
\end{equation}
with the convention $0\log0=0$.
\end{lemma}
\begin{proof}
The identification of $\mathsf Q_{B,\Pi}$ follows from the formula for the density of a Gaussian law with identity covariance, as in \eqref{eq:bd-Lm}. By \eqref{eq:bd-a-bound}, we have
\begin{equation}\label{eq:bd-L-bounds}
\exp\big(-\beta(8n)^{1/2}\|y\|-4\beta^2n\big)\le L_{B,\Pi}(y)\le\exp\big(\beta(8n)^{1/2}\|y\|\big),
\end{equation}
and in particular all relative entropies in this proof are finite. For the continuity claim, the density $L_m$ of $\mathsf Q_m$, given by \eqref{eq:bd-Lm}, is continuous in $(m,y)$, and by \eqref{eq:bd-L-bounds} with $B=\{-1,1\}^n$ and $\Pi=I$,
\[
|L_m(y)\log L_m(y)|\le\big(\beta(8n)^{1/2}\|y\|+4\beta^2n\big)e^{\beta(8n)^{1/2}\|y\|}.
\]
The right side does not depend on $m$ and is integrable under $\mathsf P_0$. By dominated convergence, $\mathcal D(m)=\int L_m\log L_m\dd\mathsf P_0$ is continuous in $m$.
We now prove \eqref{eq:bd-restrict}, first for $\Pi=I$. We write $L_B=L_{B,I}$ and $\mathsf Q_B=\mathsf Q_{B,I}$. If $p=1$, then $\mathsf Q_B=\mathsf Q_m$ and there is nothing to prove. If $p<1$, then $L_m=pL_B+(1-p)L_{B^c}$, where $L_{B^c}$ is defined in the same way. Since $L_m\ge pL_B$, we have $\log L_B\le\log L_m-\log p$ and
\[
\KL{\mathsf Q_B}{\mathsf P_0}=\int L_B\log L_B\dd\mathsf P_0\le\int L_B\log L_m\dd\mathsf P_0-\log p.
\]
Since $L_m\ge(1-p)L_{B^c}$, we also have
\[
\int L_{B^c}\log L_m\dd\mathsf P_0\ge\log(1-p)+\int L_{B^c}\log L_{B^c}\dd\mathsf P_0\ge\log(1-p),
\]
because the last integral is a relative entropy and is nonnegative. Then
\begin{align*}
\mathcal D(m)=\int L_m\log L_m\dd\mathsf P_0&=p\int L_B\log L_m\dd\mathsf P_0+(1-p)\int L_{B^c}\log L_m\dd\mathsf P_0\\
&\ge p\int L_B\log L_m\dd\mathsf P_0+(1-p)\log(1-p).
\end{align*}
Combining the last three displays, we obtain \eqref{eq:bd-restrict} for $\Pi=I$. The second inequality in \eqref{eq:bd-restrict} holds because the binary entropy is at most $\log2$.
For a general $\Pi$, it suffices to show $\KL{\mathsf Q_{B,\Pi}}{\mathsf P_0}\le\KL{\mathsf Q_B}{\mathsf P_0}$. This is the data-processing inequality for the Markov kernel $y\mapsto\Pi y+\Pi^\perp z'$, with $z'$ an independent standard Gaussian vector, which maps $\mathsf P_0$ to itself and $\mathsf Q_B$ to $\mathsf Q_{B,\Pi}$; we give the short direct argument. Under $\mathsf P_0$, the vectors $\Pi y$ and $\Pi^\perp y$ are independent, and $\langle\Pi^\perp y,a\rangle$ is a centered Gaussian variable with variance $\|\Pi^\perp a\|^2$ for each $a\in\R^{\mathsf E}$. Since $\langle y,a\rangle=\langle\Pi y,\Pi a\rangle+\langle\Pi^\perp y,\Pi^\perp a\rangle$ and $\|a\|^2=\|\Pi a\|^2+\|\Pi^\perp a\|^2$, integrating out $\Pi^\perp y$ in each term of the finite average defining $L_B$ gives
\[
\E_{\mathsf P_0}\big[L_B(y)\,\big|\,\Pi y\big]=L_{B,\Pi}(y).
\]
By the convexity of $s\mapsto s\log s$ on $[0,\infty)$ and the conditional Jensen inequality, we have
\[
\KL{\mathsf Q_{B,\Pi}}{\mathsf P_0}=\E_{\mathsf P_0}\big[L_{B,\Pi}\log L_{B,\Pi}\big]\le\E_{\mathsf P_0}\big[L_B\log L_B\big]=\KL{\mathsf Q_B}{\mathsf P_0}.
\]
This completes the proof.
\end{proof}
The next lemma converts the relative entropy of a Gaussian mixture with respect to the Gaussian law into the relative entropy in the reverse direction. We recall \ref{F:transport}: for every $k\ge1$ and every probability measure $\rho$ on $\R^k$, we have $W_2(\rho,\gamma_k)^2\le2\KL{\rho}{\gamma_k}$, where $\gamma_k$ is the standard Gaussian measure on $\R^k$ and $W_2$ is the quadratic Wasserstein distance.
\begin{lemma}\label{lem:bd-transport}
Let $k\ge1$ and $\ell\ge0$, and let $\nu$ be a probability measure on $\{b\in\R^k:\|b\|\le\ell\}$. Define
\[
L(y)=\int\exp\Big(\langle y,b\rangle-\frac12\|b\|^2\Big)\,\nu(\mathrm db)\qquad(y\in\R^k),
\]
and let $\mathsf Q=L\gamma_k$. Then
\[
0\le-\int\log L\dd\gamma_k=\KL{\gamma_k}{\mathsf Q}\le\ell\big(2\KL{\mathsf Q}{\gamma_k}\big)^{1/2}.
\]
\end{lemma}
\begin{proof}
The function $L$ is positive and smooth, and $\nabla\log L(y)=\int b\,\nu_y(\mathrm db)$, where $\nu_y$ is the probability measure proportional to $\exp(\langle y,b\rangle-\|b\|^2/2)\nu(\mathrm db)$. Since $\nu_y$ is supported in the ball of radius $\ell$, we have $\|\nabla\log L\|\le\ell$, and $\log L$ is $\ell$-Lipschitz. The measure $\mathsf Q$ is the law of $b+Z$ with $b\sim\nu$ and $Z\sim\gamma_k$ independent, so $\mathsf Q$ and $\gamma_k$ have finite second moments, and $\log L$ is integrable under both. By definition, $\KL{\mathsf Q}{\gamma_k}=\int\log L\dd\mathsf Q$ and $\KL{\gamma_k}{\mathsf Q}=-\int\log L\dd\gamma_k\ge0$. Let $(Y,Y')$ be a coupling of $\mathsf Q$ and $\gamma_k$. Then
\[
\KL{\mathsf Q}{\gamma_k}+\KL{\gamma_k}{\mathsf Q}=\E\big[\log L(Y)-\log L(Y')\big]\le\ell\,\E\|Y-Y'\|\le\ell\big(\E\|Y-Y'\|^2\big)^{1/2}.
\]
Taking the infimum over couplings and using \ref{F:transport}, we find that the left side is at most $\ell W_2(\mathsf Q,\gamma_k)\le\ell(2\KL{\mathsf Q}{\gamma_k})^{1/2}$. Since $\KL{\mathsf Q}{\gamma_k}\ge0$, the claim follows.
\end{proof}
\begin{corollary}\label{cor:bd-reverse}
Under the assumptions of \cref{lem:bd-restrict}, we have
\[
\int\log L_{B,\Pi}\dd\mathsf P_0\ge-4\beta\big(n\KL{\mathsf Q_{B,\Pi}}{\mathsf P_0}\big)^{1/2}\ge-4\beta\Big(\frac{n(\mathcal D(m)+\log2)}p\Big)^{1/2}.
\]
\end{corollary}
\begin{proof}
By \cref{lem:bd-restrict}, $L_{B,\Pi}$ has the form of \cref{lem:bd-transport} with $k=|\mathsf E|$ and $\nu$ the law of $\beta\Pi a(v^*)$. By \eqref{eq:bd-a-bound} and $\|\Pi a\|\le\|a\|$, we may take $\ell=\beta(8n)^{1/2}$. Then \cref{lem:bd-transport} implies the first inequality, since $\beta(8n)^{1/2}\cdot2^{1/2}=4\beta n^{1/2}$. The second inequality is \eqref{eq:bd-restrict}.
\end{proof}
\section{Proof of the main result}\label{sec:bd-proof}
We first prove a nonasymptotic version of \cref{thm:band}. Recall the constants $\varepsilon_D(\eta)$ and $c_\beta$ from \eqref{eq:bd-constants} and \eqref{eq:bd-c-beta}.
\begin{proposition}\label{prop:bd-quant}
Let $\beta\ge0$, $h\in\R$, $D\ge1$, and $T\ge1$. Let $G\in\cG_D$ have $n$ vertices, let $\mathcal D$ be as in \cref{def:planted} with $S=A_G/D$, and let a query history of depth $T$ on $G$ with center $m$ be given. Then for every $\eta>0$ and $\varepsilon\in(0,1)$,
\begin{equation}\label{eq:bd-quant}
\begin{split}
p_G(\beta,h)\ge\frac1n\E\,\Theta_G(m)&-c_\beta\eta-C'_{\beta,h}\Big(\frac{\varepsilon_D(\eta)}\varepsilon\Big)^{1/2}\\
&-4\beta\Big(\frac{n^{-1}\E\,\mathcal D(m)+n^{-1}\log2}{1-\varepsilon}\Big)^{1/2}+\frac{\log(1-\varepsilon)}n,
\end{split}
\end{equation}
where $C'_{\beta,h}=\beta C_2^{1/2}+|h|+\log2+\beta^2/4$ and $C_2$ is the constant of \ref{F:gaussian-max} with $p=2$.
\end{proposition}
\begin{proof}
We have $|\log Z_G|\le n\log2+\beta\max_\sigma|H_G(\sigma)|+|h|n$, which is integrable by \ref{F:gaussian-max}. Further, $|\ent(m_x)|\le\log2$ and $0\le\langle\kappa,S\kappa\rangle\le\sum_{x,y}S_{xy}=n$, since $\kappa\in[0,1]^V$. Together with \ref{F:gaussian-max}, this gives the pathwise bound
\[
|\Theta_G(m)|\le\beta\max_\sigma|H_G(\sigma)|+n\Big(|h|+\log2+\frac{\beta^2}4\Big),
\]
and, by the Minkowski inequality,
\begin{equation}\label{eq:bd-UI}
\big(\E\,\Theta_G(m)^2\big)^{1/2}\le\beta\big(\E\max_\sigma H_G(\sigma)^2\big)^{1/2}+n\Big(|h|+\log2+\frac{\beta^2}4\Big)\le C'_{\beta,h}n.
\end{equation}
Let $\mathcal G=\{p_\eta\ge1-\varepsilon\}$, where $p_\eta=\pi_m(B_\eta)$ is as in \cref{lem:bd-band}. Then $\mathcal G\in\mathcal H$, and by Markov's inequality and \cref{lem:bd-band}, we have
\begin{equation}\label{eq:bd-bad-event}
\P(\mathcal G^c)=\P(1-p_\eta>\varepsilon)\le\frac{\varepsilon_D(\eta)}\varepsilon.
\end{equation}
On $\mathcal G$, \cref{lem:bd-pathwise} gives
\begin{equation}\label{eq:bd-on-good}
\mathbf 1_{\mathcal G}\log Z_G\ge\mathbf 1_{\mathcal G}\big(\Theta_G(m)-c_\beta\eta n+\log(1-\varepsilon)\big)+\mathbf 1_{\mathcal G}\log L_\eta(g).
\end{equation}
We compute the expectation of the last term by conditioning on $\mathcal H$. Let $F(\omega,y)=\mathbf 1_{\mathcal G}(\omega)\log L_\eta(y)$. By the discussion after \eqref{eq:bd-L-eta}, $F$ is $\mathcal H\otimes\mathscr B(\R^E)$-measurable, and by \eqref{eq:bd-L-bounds} we have $|F(\omega,y)|\le\beta(8n)^{1/2}\|y\|+4\beta^2n$. In particular, $F(\cdot,g)$ is integrable. Since $\Pi\bar g=0$ and $L_\eta(y)$ depends on $y$ only through $\Pi y$, we have $L_\eta(\bar g+\Pi w)=L_\eta(w)$. Applying \cref{lem:bd-conditioning}(b) to the positive and negative parts of $F$, we obtain
\[
\E\big[\mathbf 1_{\mathcal G}\log L_\eta(g)\big]=\E\Big[\mathbf 1_{\mathcal G}\int\log L_\eta\dd\mathsf P_0\Big].
\]
For each fixed $\omega\in\mathcal G$, the function $L_\eta$ is the function $L_{B,\Pi}$ of \cref{lem:bd-restrict}, with the deterministic data $m(\omega)$, $\Pi(\omega)$, and $B=B_\eta(\omega)$, and with $p=p_\eta(\omega)\ge1-\varepsilon$. Then \cref{cor:bd-reverse}, applied with $\mathcal D(m)$ the value of the deterministic function $\mathcal D$ at $m(\omega)$, implies that
\[
\mathbf 1_{\mathcal G}\int\log L_\eta\dd\mathsf P_0\ge-4\beta\Big(\frac{n(\mathcal D(m)+\log2)}{1-\varepsilon}\Big)^{1/2}.
\]
Taking expectations and using Jensen's inequality for the square root, we find
\begin{equation}\label{eq:bd-likelihood}
\E\big[\mathbf 1_{\mathcal G}\log L_\eta(g)\big]\ge-4\beta\Big(\frac{n(\E\,\mathcal D(m)+\log2)}{1-\varepsilon}\Big)^{1/2}.
\end{equation}
On $\mathcal G^c$ we use a cruder bound. Since $W$ has zero diagonal, the average of $\beta H_G(\sigma)+h\sum_x\sigma_x$ under the uniform law on $\{-1,1\}^V$ is zero. By Jensen's inequality, we have $\log Z_G\ge n\log2\ge0$. Combining this with \eqref{eq:bd-on-good} and \eqref{eq:bd-likelihood}, and using $\P(\mathcal G)\le1$ and $\log(1-\varepsilon)<0$, we obtain
\[
\E\log Z_G\ge\E\big[\mathbf 1_{\mathcal G}\Theta_G(m)\big]-c_\beta\eta n+\log(1-\varepsilon)-4\beta\Big(\frac{n(\E\,\mathcal D(m)+\log2)}{1-\varepsilon}\Big)^{1/2}.
\]
Finally, by the Cauchy--Schwarz inequality, \eqref{eq:bd-UI}, and \eqref{eq:bd-bad-event},
\[
\E\big[\mathbf 1_{\mathcal G}\Theta_G(m)\big]\ge\E\,\Theta_G(m)-\P(\mathcal G^c)^{1/2}\big(\E\,\Theta_G(m)^2\big)^{1/2}\ge\E\,\Theta_G(m)-C'_{\beta,h}n\Big(\frac{\varepsilon_D(\eta)}\varepsilon\Big)^{1/2}.
\]
Dividing by $n$ gives \eqref{eq:bd-quant}.
\end{proof}
\begin{proof}[Proof of \cref{thm:band}]
Fix $\eta>0$ and $\varepsilon\in(0,1)$, and let $b_k$ denote the sum of the four error terms in \eqref{eq:bd-quant} for $G=G_k$, so that $p_{G_k}(\beta,h)\ge n_k^{-1}\E\,\Theta_{G_k}(m^{(k)})-b_k$ by \cref{prop:bd-quant}. As $k\to\infty$, we have $\varepsilon_{D_k}(\eta)\to0$, because $D_k\to\infty$ and $\beta$, $h$, $T$, and $\eta$ are fixed. Further, the terms involving $n_k^{-1}\log2$ and $n_k^{-1}\log(1-\varepsilon)$ tend to zero, since $n_k\ge D_k+1\to\infty$. Then \eqref{eq:bd-ND} and the continuity of the square root imply that $\limsup_kb_k\le c_\beta\eta+4\beta(\delta/(1-\varepsilon))^{1/2}$. By \eqref{eq:bd-UI}, the sequence $n_k^{-1}\E\,\Theta_{G_k}(m^{(k)})$ is bounded. Since $\liminf_k(a_k-b_k)\ge\liminf_ka_k-\limsup_kb_k$ whenever the right side is defined, we obtain
\[
\liminf_{k\to\infty}p_{G_k}(\beta,h)\ge\liminf_{k\to\infty}\frac1{n_k}\E\,\Theta_{G_k}\big(m^{(k)}\big)-c_\beta\eta-4\beta\Big(\frac\delta{1-\varepsilon}\Big)^{1/2}.
\]
Letting $\eta\downarrow0$ and then $\varepsilon\downarrow0$ gives \eqref{eq:bd-main}.
\end{proof}
\begin{remark}\label{rem:bd-query}
The requirement that the center is itself a query is used through \cref{lem:bd-conditioning}(a): the vector $Wm$ appears in the linear coefficient $r$, which must be $\mathcal H$-measurable for the band $B_\eta$ and the function $L_\eta$ to be $\mathcal H$-measurable. If $Wm$ were not observed, then $\langle\beta Wm,v\rangle$ would contain a part that is linear in the unobserved Gaussian vector, and \cref{lem:bd-conditioning}(b) could not be applied with $\mathcal H$-measurable $L_\eta$.
\end{remark}
% ---------- 07-nondetection.tex
% Section 7: Relative entropy of the planted product model.
% Label prefix: nd-. Bibliography: bib/nondetection.bib. Audit: audit/nondetection.md.
\chapter{Relative entropy of the planted product model}\label{sec:nondetection}
The goal of this section is to prove \cref{thm:nd}, which states that the relative entropy $\mathcal D(m)$ of the planted product model of \cref{def:planted} is small compared to $n$, provided that two conditions hold. First, the scalar transforms $\psi_{m_x}$ of the coordinates of the center must be close to the transform $\psi_\mu$ of a single law $\mu$, on average over the vertices and uniformly over a compact set of site profiles. Second, $\mu$ must satisfy $\Gamma_\mu\le0$ on $[0,1]$. The bound holds for every deterministic center, and since its error terms depend on the center only through the discrepancy between the two transforms, it can be averaged over random centers. We also prove \cref{prop:nd-scalar}, which states that the law $\mu_+$ of the Parisi center satisfies $\Gamma_{\mu_+}(s)\le0$ for all $s\ge0$. Its proof uses only the optimality of the Parisi measure. \Cref{cor:nd} combines the two results.
The planted product model is a spiked Wigner model with a sparse variance profile and a prior that depends on the vertex. For the spiked Wigner model with independent and identically distributed noise, the limit of the relative entropy, or equivalently of the mutual information between the spike and the observation, was computed in~\cite{LelargeMiolane2019}, where earlier work is discussed, and for bounded centered unit-variance priors, El Alaoui, Krzakala, and Jordan determined the region in which the spiked and the unspiked models are contiguous~\cite{ElAlaouiKrzakalaJordan2020}. Guionnet, Ko, Krzakala, and Zdeborov\'a computed this limit for inhomogeneous noise whose inverse variances form a positive-semidefinite profile with a block structure or a regular limit~\cite{GuionnetKoKrzakalaZdeborova2025}. Behne and Reeves computed it when the signal-to-noise ratio is constant on each of finitely many blocks, for every nonnegative matrix of block ratios~\cite{BehneReeves2022groupwise}. On the Nishimori line, multispecies spin glasses are spiked models with binary signals and a block variance profile. Alberici, Camilli, Contucci, and Mingione computed their free energy when the interaction is elliptic~\cite{AlbericiCamilliContucciMingione2021multi}, and for the deep Boltzmann machine, whose interaction matrix is not positive semidefinite~\cite{AlbericiCamilliContucciMingione2021deep}. For general variance profiles and certain priors with independent and identically distributed symmetric coordinates of variance at most one, De and Kunisky proved that strong detection is impossible below a spectral threshold; their assumptions include a 1-moment-subgaussian condition on the product of two independent samples of the prior and, in our setting, require $D$ to be of order $n$~\cite[Theorem~1.11]{DeKunisky2025inhomogeneous}.
Interpolations of Guerra type~\cite{Guerra2003} between a matrix channel and scalar channels were used by Korada and Macris~\cite{KoradaMacris2009gauge} for $p$-spin models on the Nishimori line and by Krzakala, Xu, and Zdeborov\'a~\cite{KrzakalaXuZdeborova2016mutual} for spiked models. Barbier and Macris made the interpolation path adaptive~\cite{BarbierMacris2019adaptive}. Our proof of \cref{thm:nd} follows the Hamilton--Jacobi approach of Mourrat~\cite{Mourrat2021HJ,Mourrat2020matrix}, in which the relative entropy of a family of interpolating channels is compared with the solution of a Hamilton--Jacobi equation given by a Hopf--Lax formula. For inference of matrix tensor products, Chen and Xia showed that the solution of the corresponding Hamilton--Jacobi equation is an upper bound on the limit of the free energy for every interaction matrix, also when the nonlinearity is not convex~\cite{ChenXia2022tensor}. We need only an upper bound, the entries of $S$ are of order at most $1/D$, and $S$ need not be positive semidefinite. For the SK model, the function $\Gamma_\mu$ coincides with a function introduced by Chen, Panchenko, and Subag in their study of the TAP free energy~\cite{CPSGeneralizedTAP}; see the discussion after \eqref{eq:nd-Gamma}.
\Cref{sec:nd-channels} collects identities for Gaussian channels. \Cref{sec:nd-scalar} defines the scalar transform and proves \cref{lem:nd-psi}. \Cref{sec:nd-parisi} proves \cref{prop:nd-scalar}, and \cref{sec:nd-statement} states \cref{thm:nd,cor:nd}. The remaining subsections prove \cref{thm:nd}. In \cref{sec:nd-interpolation}, we interpolate between independent scalar channels at the vertices and the planted edge channel. We compare the normalized relative entropy $F(t,q)$ of the interpolating channel with a scalar Hopf--Lax formula, which is treated in \cref{sec:nd-hopf}. The comparison uses only the behavior of $F$ at points where it is touched from above by a function with bounded Hessian. At such points, the time derivative of $F$ is close to a quadratic form in the gradient of $F$ (\cref{prop:nd-contact}). The proof of this estimate rests on an exact covariance identity for the derivative of the quenched relative entropy (\cref{sec:nd-score}) and on concentration over a family of perturbations of small metric entropy (\cref{sec:nd-concentration}). \Cref{sec:nd-proof} completes the proofs.
Throughout this section, $S$ is a symmetric $n\times n$ matrix with nonnegative entries, zero diagonal, and unit row sums, as in \cref{def:planted}. We index its rows and columns by a set $V$ with $|V|=n$, and we set $\mathsf E=\{\{x,y\}:S_{xy}>0\}$. We do not assume that $S$ is positive semidefinite; see \cref{rem:nd-indefinite}. For $w\in\R^V$ we write $|w|$ for the Euclidean norm, $\|w\|_n=n^{-1/2}|w|$ for the normalized Euclidean norm, and $\operatorname{dist}_n(w,\mathcal A)=\inf_{z\in\mathcal A}\|w-z\|_n$ for $\mathcal A\subset\R^V$. We write $\mathbf 1$ for the vector with all coordinates equal to one, $w\cdot w'$ for the Euclidean inner product, $w\odot w'$ for the coordinatewise product, and $w^{\odot2}=w\odot w$. For a set $\mathcal A\subset\R^V$ and $c\in\R$ we write $c\mathcal A=\{cw:w\in\mathcal A\}$ and $S\mathcal A=\{Sw:w\in\mathcal A\}$. If $\max_{x,y}S_{xy}\le1/D$ for some $D\ge1$, then
\begin{equation}\label{eq:nd-S}
\|S\|\le1,\qquad \tr S^2\le\frac nD,\qquad \max_{x\in V}(S^2)_{xx}\le\frac1D,\qquad n\ge D+1.
\end{equation}
Indeed, the bound on the operator norm follows from Jensen's inequality as in \eqref{eq:pre-S}, and $(S^2)_{xx}=\sum_yS_{xy}^2\le D^{-1}\sum_yS_{xy}=D^{-1}$; summing over $x$ gives the bound on the trace. Since each row of $S$ has unit sum and entries at most $1/D$, it has at least $D$ nonzero entries, none of them on the diagonal.
\section{Gaussian channels}\label{sec:nd-channels}
We begin with identities relating relative entropy and posterior moments for Gaussian channels. The first derivative formula in \cref{lem:nd-channel} is the relation of Guo, Shamai, and Verd\'u~\cite{GuoShamaiVerdu2005} between mutual information and the minimum mean-square error, written for relative entropy. The second derivatives in \cref{lem:nd-channel}(c) are a special case of the Hessian formula of Payar\'o and Palomar~\cite[Theorem~5]{PayaroPalomar2009hessian}. We include the short proofs for completeness.
% CHECK: PayaroPalomar2009hessian Theorem 5 verified in arXiv:0903.1945v1 (with H = I, P = Diag(sqrt(lambda)),
% the Hessian of the mutual information is -(1/2) E[Phi(Y) o Phi(Y)], Phi the conditional covariance).
Let $N\ge1$, let $\theta$ be a random vector in $\R^N$ that takes finitely many values, and let $Z$ be a standard Gaussian vector in $\R^N$ independent of $\theta$. For $\lambda\in[0,\infty)^N$, we observe $Y^\lambda=(\sqrt{\lambda_j}\,\theta_j+Z_j)_{j\le N}$. Let $\theta'$ be an independent copy of $\theta$, and let $\E'$ denote the expectation over $\theta'$ only. Then the law of $Y^\lambda$ has density
\begin{equation}\label{eq:nd-lr}
L_\lambda(y)=\E'\exp\Big(\sum_{j\le N}\big(\sqrt{\lambda_j}\,y_j\theta'_j-\lambda_j(\theta'_j)^2/2\big)\Big)\qquad(y\in\R^N)
\end{equation}
with respect to the standard Gaussian law $N(0,I_N)$, and we define
\[
I(\lambda)=\KL{\mathrm{Law}(Y^\lambda)}{N(0,I_N)}=\E\log L_\lambda(Y^\lambda).
\]
By Bayes's rule, the conditional law of $\theta$ given $Y^\lambda=y$ is the law of $\theta'$ reweighted by the exponential in \eqref{eq:nd-lr}. Given $Y^\lambda$, let $\theta^1,\theta^2,\dots$ be conditionally independent samples from this law, and write $\langle\cdot\rangle_\lambda$ for the average over them; for instance, $\langle\theta^1_j\rangle_\lambda=\E[\theta_j\mid Y^\lambda]$. Since $\theta$ and $\theta^1$ have the same conditional law given $Y^\lambda$, we have the Bayesian form of the Nishimori identity~\cite{Nishimori1981}
\begin{equation}\label{eq:nd-nishimori}
\E\big\langle f(\theta,\theta^1,\dots,\theta^k,Y^\lambda)\big\rangle_\lambda=\E\big\langle f(\theta^{k+1},\theta^1,\dots,\theta^k,Y^\lambda)\big\rangle_\lambda
\end{equation}
for all $k\ge0$ and all bounded measurable $f$. We use \eqref{eq:nd-nishimori} only for deterministic $\lambda$. Substituting $Y^\lambda=(\sqrt{\lambda_j}\theta_j+Z_j)_j$, we obtain
\begin{equation}\label{eq:nd-I-teacher}
I(\lambda)=\E\log\E'\exp\Big(\sum_{j\le N}\big(\sqrt{\lambda_j}\,Z_j\theta'_j+\lambda_j\theta_j\theta'_j-\lambda_j(\theta'_j)^2/2\big)\Big),
\end{equation}
and $\langle\cdot\rangle_\lambda$ is the average with weights proportional to the exponential in \eqref{eq:nd-I-teacher}, with $\theta'$ replaced by the replica. We regard it as a function of $(\theta,Z)$. Differentiating these weights, we find that for a function $f$ of one replica,
\begin{equation}\label{eq:nd-derivatives}
\begin{gathered}
\partial_{Z_j}\langle f(\theta^1)\rangle_\lambda=\sqrt{\lambda_j}\big(\langle f(\theta^1)\theta^1_j\rangle_\lambda-\langle f(\theta^1)\rangle_\lambda\langle\theta^1_j\rangle_\lambda\big),\\
\partial_{\lambda_j}\langle f(\theta^1)\rangle_\lambda=\langle f(\theta^1)A_j(\theta^1)\rangle_\lambda-\langle f(\theta^1)\rangle_\lambda\langle A_j(\theta^1)\rangle_\lambda,
\end{gathered}
\end{equation}
where the second identity holds for $\lambda_j>0$ and
\[
A_j(\theta^1)=\frac{Z_j\theta^1_j}{2\sqrt{\lambda_j}}+\theta_j\theta^1_j-\frac{(\theta^1_j)^2}2 .
\]
\begin{lemma}\label{lem:nd-channel}
Let $N$, $\theta$, and $I$ be as above, and let $c_\theta=\max_j\|\theta_j\|_\infty$.
\begin{enumerate}[label=(\alph*)]
\item The function $I$ is nonnegative and continuous on $[0,\infty)^N$, $I(0)=0$, and $I$ is nondecreasing in each coordinate.
\item Let $j\le N$. At every $\lambda\in[0,\infty)^N$ with $\lambda_j>0$, the partial derivative $\partial_{\lambda_j}I(\lambda)$ exists and
\[
\partial_{\lambda_j}I(\lambda)=\frac12\E\langle\theta^1_j\rangle_\lambda^2\in\Big[0,\frac12\E\theta_j^2\Big].
\]
The function $\lambda\mapsto\E\langle\theta^1_j\rangle_\lambda^2$ is continuous on $[0,\infty)^N$.
\item Let $j,k\le N$. At every $\lambda\in[0,\infty)^N$ with $\lambda_j>0$ and $\lambda_k>0$,
\[
\partial_{\lambda_k}\partial_{\lambda_j}I(\lambda)=\frac12\E\big(\langle\theta^1_j\theta^1_k\rangle_\lambda-\langle\theta^1_j\rangle_\lambda\langle\theta^1_k\rangle_\lambda\big)^2,
\]
and the right side is continuous on $[0,\infty)^N$.
\end{enumerate}
\end{lemma}
\begin{proof}
(a) The exponent in \eqref{eq:nd-I-teacher} is bounded in absolute value by $\sum_j(c_\theta\sqrt{\lambda_j}|Z_j|+3c_\theta^2\lambda_j/2)$, and so is the logarithm of its average over $\theta'$. Since this bound is integrable and locally uniform in $\lambda$, and the integrand is continuous in $\lambda$, the function $I$ is continuous by dominated convergence. Further, $I\ge0$ because it is a relative entropy, and $I(0)=0$. Monotonicity follows from (b) and continuity.
(b) Let $\lambda_j>0$. For $\lambda'$ in a neighborhood of $\lambda$ on which $\lambda'_j\ge\lambda_j/2$, the derivative in $\lambda_j$ of the integrand in \eqref{eq:nd-I-teacher} is $\langle A_j(\theta^1)\rangle_{\lambda'}$, which is bounded by $C(1+|Z_j|)$ with $C$ depending on $c_\theta$ and $\lambda_j$. We may therefore differentiate under the expectation, and $\partial_{\lambda_j}I=\E\langle A_j(\theta^1)\rangle_\lambda$. By \eqref{eq:nd-derivatives} and Gaussian integration by parts~\ref{F:ibp} in $Z_j$, which is independent of $\theta$ and of $(Z_k)_{k\ne j}$,
\[
\E\big[Z_j\langle\theta^1_j\rangle_\lambda\big]=\sqrt{\lambda_j}\,\E\big[\langle(\theta^1_j)^2\rangle_\lambda-\langle\theta^1_j\rangle_\lambda^2\big].
\]
Inserting this identity into the expression for the derivative above, we obtain
\[
\partial_{\lambda_j}I=\frac12\E\langle(\theta_j^1)^2\rangle_\lambda-\frac12\E\langle\theta^1_j\rangle_\lambda^2+\E\big[\theta_j\langle\theta^1_j\rangle_\lambda\big]-\frac12\E\langle(\theta^1_j)^2\rangle_\lambda=\E\big[\theta_j\langle\theta^1_j\rangle_\lambda\big]-\frac12\E\langle\theta^1_j\rangle_\lambda^2 .
\]
By \eqref{eq:nd-nishimori}, $\E[\theta_j\langle\theta^1_j\rangle_\lambda]=\E\langle\theta^2_j\theta^1_j\rangle_\lambda=\E\langle\theta^1_j\rangle_\lambda^2$, which gives the formula. Since $\langle\theta^1_j\rangle_\lambda=\E[\theta_j\mid Y^\lambda]$, we have $\E\langle\theta^1_j\rangle_\lambda^2\le\E\theta_j^2$ by Jensen's inequality. For fixed $(\theta,Z)$, the average $\langle\theta^1_j\rangle_\lambda$ is a continuous function of $\lambda\in[0,\infty)^N$ bounded by $c_\theta$, and the continuity follows by dominated convergence.
(c) Let $\lambda_j,\lambda_k>0$. By (b) and \eqref{eq:nd-nishimori}, $2\partial_{\lambda_j}I(\lambda')=\E[\theta_j\langle\theta^1_j\rangle_{\lambda'}]$ for all $\lambda'$ near $\lambda$. We differentiate the right side in $\lambda_k$, which is justified as in (b). For brevity, we write $b_l=\langle\theta^1_l\rangle_\lambda$, $c_{jk}=\langle\theta^1_j\theta^1_k\rangle_\lambda$, and $c_{jkk}=\langle\theta^1_j(\theta^1_k)^2\rangle_\lambda$. By \eqref{eq:nd-derivatives}, we have $2\partial_{\lambda_k}\partial_{\lambda_j}I=T_1+T_2+T_3$, where
\begin{gather*}
T_1=\frac{1}{2\sqrt{\lambda_k}}\E\big[\theta_jZ_k(c_{jk}-b_jb_k)\big],\qquad
T_2=\E\big[\theta_j\theta_k(c_{jk}-b_jb_k)\big],\\
T_3=-\frac12\E\big[\theta_j(c_{jkk}-b_jc_{kk})\big].
\end{gather*}
By \eqref{eq:nd-derivatives}, we have $\partial_{Z_k}(c_{jk}-b_jb_k)=\sqrt{\lambda_k}(c_{jkk}-b_jc_{kk}-2c_{jk}b_k+2b_jb_k^2)$, and Gaussian integration by parts in $Z_k$ gives
\[
T_1=\frac12\E\big[\theta_j(c_{jkk}-b_jc_{kk})\big]-\E\big[\theta_j(c_{jk}b_k-b_jb_k^2)\big].
\]
The quantities $b_l$, $c_{jk}$, and $c_{jkk}$ are functions of $Y^\lambda$, and $\E[\theta_j\mid Y^\lambda]=b_j$ and $\E[\theta_j\theta_k\mid Y^\lambda]=c_{jk}$. Conditioning on the observation, we find
\[
T_1+T_3=-\E\big[b_jc_{jk}b_k-b_j^2b_k^2\big],\qquad T_2=\E\big[c_{jk}^2-c_{jk}b_jb_k\big],
\]
and $T_1+T_2+T_3=\E(c_{jk}-b_jb_k)^2$. The continuity follows as in (b).
\end{proof}
\section{The scalar transform}\label{sec:nd-scalar}
For $m\in[-1,1]$ let $\nu_m$ be the law of $\sigma-m$, where $\sigma\in\{-1,1\}$ has mean $m$. The law $\nu_m$ gives mass $(1+m)/2$ to $1-m$ and mass $(1-m)/2$ to $-1-m$; it has mean zero and variance $\kappa(m)=1-m^2$, and it is the point mass at zero if $m=\pm1$. For $r\ge0$ the scalar transform is
\begin{equation}\label{eq:nd-psi}
\psi_m(r)=\E\log\E_{\tau\sim\nu_m}\exp\Big(\sqrt r\,Z\tau+r\tau\tau^*-\frac r2\tau^2\Big),
\end{equation}
where the outer expectation is over independent $\tau^*\sim\nu_m$ and $Z\sim N(0,1)$. By \eqref{eq:nd-I-teacher}, $\psi_m(r)$ is the function $I$ for $N=1$ and $\theta=\tau^*$; that is, $\psi_m(r)=\KL{\mathrm{Law}(\sqrt r\tau^*+Z)}{N(0,1)}$. For a probability measure $\mu$ on $[-1,1]$ we write
\[
\psi_\mu(r)=\int\psi_m(r)\,\mu(\mathrm dm),\qquad \Gamma_\mu(s)=2\Big[\psi_\mu(\beta^2s)-\frac{\beta^2s^2}4\Big]\qquad(r,s\ge0).
\]
\begin{lemma}\label{lem:nd-psi}
Let $m\in[-1,1]$, and for $r\ge0$ let $b_{m,r}=\E[\tau^*\mid\sqrt r\tau^*+Z]$, with $\tau^*$ and $Z$ as in \eqref{eq:nd-psi}.
\begin{enumerate}[label=(\alph*)]
\item We have $\psi_m\ge0$ and $\psi_m(0)=0$.
\item The function $\psi_m$ is continuously differentiable on $[0,\infty)$, with a right derivative at zero, and
\[
\psi_m'(r)=\frac12\E b_{m,r}^2\in\Big[0,\frac{\kappa(m)}2\Big].
\]
\item The function $\psi_m$ is nondecreasing and $1/2$-Lipschitz.
\item The function $(m,r)\mapsto\psi_m(r)$ is continuous on $[-1,1]\times[0,\infty)$.
\end{enumerate}
\end{lemma}
\begin{proof}
Parts (a) and (b) are \cref{lem:nd-channel}(a,b) with $N=1$ and $\theta=\tau^*$, since $\E(\tau^*)^2=\kappa(m)$; at $r=0$, the right derivative exists and equals $\lim_{r\downarrow0}\psi_m'(r)$ by the mean value theorem. Part (c) follows from (b), since $\kappa(m)\le1$.
For (d), let $r\ge0$ and
\[
c_\pm=\pm\sqrt r(1\mp m),\qquad \ell_{m,r}(w)=\frac{1+m}2\,e^{c_+w-c_+^2/2}+\frac{1-m}2\,e^{c_-w-c_-^2/2}\qquad(w\in\R).
\]
By \eqref{eq:nd-lr}, the function $\ell_{m,r}$ is the density of the law of $\sqrt r\tau^*+Z$ with respect to $N(0,1)$, and $\psi_m(r)=\E[\ell_{m,r}(Z)\log\ell_{m,r}(Z)]$. The function $(m,r,w)\mapsto\ell_{m,r}(w)$ is positive and continuous. Since $\ell_{m,r}(w)$ is an average of the numbers $e^{c_\pm w-c_\pm^2/2}$ and $|c_\pm|\le2\sqrt r$, we have $|\log\ell_{m,r}(w)|\le2\sqrt r|w|+2r$ and
\[
|\ell_{m,r}(w)\log\ell_{m,r}(w)|\le\big(2\sqrt r|w|+2r\big)e^{2\sqrt r|w|}.
\]
For $r$ in a bounded interval, the right side is bounded by a function of $w$ that does not depend on $(m,r)$ and is integrable under $N(0,1)$. Then (d) follows by dominated convergence.
\end{proof}
By \cref{lem:nd-psi}, for each probability measure $\mu$ on $[-1,1]$, the function $\psi_\mu$ is nonnegative and nondecreasing, $\psi_\mu(0)=0$, and $\psi_\mu$ is continuously differentiable on $[0,\infty)$ with
\begin{equation}\label{eq:nd-psi-mu}
\psi_\mu'(r)=\int\psi_m'(r)\,\mu(\mathrm dm)\in\Big[0,\frac12\int\kappa(m)\,\mu(\mathrm dm)\Big]\subset\Big[0,\frac12\Big].
\end{equation}
Indeed, $m\mapsto\psi'_m(u)$ is measurable, being a limit of difference quotients of continuous functions of $m$, and $\psi_\mu(r)=\int_0^r\int\psi'_m(u)\,\mu(\mathrm dm)\dd u$ by Fubini's theorem, where the inner integral is continuous in $u$ by dominated convergence. Since $\Gamma_\mu(0)=0$, we obtain
\begin{equation}\label{eq:nd-Gamma}
\Gamma_\mu(s)=\beta^2\int_0^s\big(2\psi_\mu'(\beta^2u)-u\big)\dd u\qquad(s\ge0).
\end{equation}
Since $\psi_{-m}=\psi_m$, the function $\Gamma_\mu$ depends only on the law of $|m|$ under $\mu$. For the SK model, it is the function $\Gamma_\mu$ defined by Chen, Panchenko, and Subag in~\cite[(3.43)--(3.44)]{CPSGeneralizedTAP}, applied to this law; a short computation with \eqref{eq:nd-psi-prime-tanh} below identifies the two definitions. Let $q=\int m^2\mu(\mathrm dm)$, and suppose that $q<1$. Chen, Panchenko, and Subag showed that $\Gamma_\mu\le0$ on $[0,1-q]$ if and only if the variational formula for their generalized TAP correction at the law of $|m|$ is minimized by the point mass at zero~\cite[Proposition~13]{CPSGeneralizedTAP}. In this case, the generalized TAP correction is replica symmetric and equals the classical TAP correction, and Plefka's condition $\beta^2\int(1-m^2)^2\mu(\mathrm dm)\le1$ holds~\cite[Propositions~13 and~14]{CPSGeneralizedTAP}.
% CHECK: (3.43)-(3.44), Propositions 13-14 and Remark 15 of CPSGeneralizedTAP verified in arXiv:1812.05066v3,
% and Remark 6 of CPSII in arXiv:1903.01030v1, only. The identification gamma_mu(u) = 2 psi_mu'(beta^2 u),
% with xi(s) = beta^2 s^2/2 and xi_q'(s) = beta^2 s, was checked by hand twice: the tilt
% exp(Phi(s,g_a(s)) - Phi(0,g_a(0))) in (3.43) shifts beta*sqrt(s)*g by (sigma - a)*beta^2*s with
% sigma = +-1 of mean a; see audit/credit-lower.md and audit/credit2-band.md.
\section{The scalar criterion for the Parisi center}\label{sec:nd-parisi}
Let $\beta>0$ and $h\ge0$. We use the notation of \cref{sec:pre-SK}: $\nu=\nu_{\beta,h}$ is the Parisi measure, $\nu(t)=\nu([0,t])$, $\Phi=\Phi_\nu$, $q_+=\max\supp\nu$, $X$ is the Parisi diffusion \eqref{eq:pre-diffusion}, $M_t=\partial_x\Phi(t,X_t)$, $\mu_+$ is the law of $M_{q_+}$, and $\Psi$ is the function \eqref{eq:pre-Psi}. By \ref{P:colehopf}, $\partial_x\Phi(t,x)=\tanh x$ for $t\in[q_+,1]$. We have $\nu(t)=1$ for $t\ge q_+$, and \eqref{eq:pre-diffusion} becomes
\begin{equation}\label{eq:nd-X-top}
X_t=X_{q_+}+\beta^2\int_{q_+}^t\tanh(X_u)\dd u+\beta(B_t-B_{q_+}),\qquad M_t=\tanh X_t\qquad(t\in[q_+,1]).
\end{equation}
In particular, $\mu_+$ is the law of $\tanh X_{q_+}$, and $\mu_+((-1,1))=1$. By \cref{cor:pre-optimality}, we have $q_+<1$ and $\E M_{q_+}^2=q_+$, which implies that
\begin{equation}\label{eq:nd-kappa-plus}
\int\kappa(m)\,\mu_+(\mathrm dm)=1-\E M_{q_+}^2=1-q_+.
\end{equation}
The next lemma identifies the law of the Parisi diffusion above $q_+$. On $[q_+,1]$ the drift is $\beta^2\tanh x$, and the diffusion is the Doob transform of Brownian motion with variance $\beta^2$ per unit time by the function $\cosh$.
\begin{lemma}\label{lem:nd-doob}
Let $\beta>0$ and $h\ge0$, let $s\in[0,1-q_+]$, and let $g\colon\R\to\R$ be bounded and measurable. For $y\in\R$ let
\[
k_s(y)=\sum_{\sigma\in\{-1,1\}}\frac{e^{\sigma y}}{2\cosh y}\,\E g\big(y+\beta^2s\sigma+\beta\sqrt s\,Z\big),
\]
where $Z\sim N(0,1)$. Then $\E g(X_{q_++s})=\E k_s(X_{q_+})$.
\end{lemma}
\begin{proof}
For $s=0$ there is nothing to prove. Let $s>0$ and $T'=q_++s\le1$, and let $(\mathfrak F_t)$ be the filtration generated by $B$, to which $X$ is adapted. For $t\in[q_+,T']$ let
\[
\Lambda_t=\exp\Big(-\beta\int_{q_+}^t\tanh(X_u)\dd B_u-\frac{\beta^2}2\int_{q_+}^t\tanh^2(X_u)\dd u\Big).
\]
Since the integrand $\beta\tanh(X_u)$ is bounded, Novikov's condition holds (see~\cite[Chapter~3]{KaratzasShreve1991} for this condition and for Girsanov's theorem), and $\Lambda$ is a martingale on $[q_+,T']$ with $\Lambda_{q_+}=1$. Let $\mathsf R$ be the probability measure with density $\Lambda_{T'}$ with respect to $\P$. Since $\E[\Lambda_{T'}\mid\mathfrak F_{q_+}]=1$, the measures $\mathsf R$ and $\P$ agree on $\mathfrak F_{q_+}$. By Girsanov's theorem, the process $\tilde B_t=B_t-B_{q_+}+\beta\int_{q_+}^t\tanh(X_u)\dd u$, $t\in[q_+,T']$, is a Brownian motion under $\mathsf R$, independent of $\mathfrak F_{q_+}$. By \eqref{eq:nd-X-top}, we have $X_t=X_{q_+}+\beta\tilde B_t$ for $t\in[q_+,T']$.
Next, \eqref{eq:nd-X-top} and It\^o's formula give
\begin{align*}
\log\cosh X_{T'}-\log\cosh X_{q_+}&=\beta\int_{q_+}^{T'}\tanh(X_u)\dd B_u+\frac{\beta^2}2\int_{q_+}^{T'}\big(1+\tanh^2(X_u)\big)\dd u\\
&=-\log\Lambda_{T'}+\frac{\beta^2s}2.
\end{align*}
Since $\Lambda_{T'}>0$, we obtain
\[
\E g(X_{T'})=\E_{\mathsf R}\big[g(X_{T'})\Lambda_{T'}^{-1}\big]=e^{-\beta^2s/2}\,\E_{\mathsf R}\Big[\frac{(g\cosh)(X_{q_+}+\beta\tilde B_{T'})}{\cosh X_{q_+}}\Big].
\]
Under $\mathsf R$, the variable $\tilde B_{T'}$ has law $N(0,s)$ and is independent of $X_{q_+}$, whose law under $\mathsf R$ is its law under $\P$. Since $\cosh(x+z)\le e^{|z|}\cosh x$, the expression inside the last expectation is bounded by $\|g\|_\infty e^{\beta|\tilde B_{T'}|}$. Then $\E g(X_{T'})=\E k(X_{q_+})$, where
\[
k(y)=\frac{e^{-\beta^2s/2}}{\cosh y}\,\E\big[(g\cosh)(y+\beta\sqrt s\,Z)\big].
\]
Finally, we write $\cosh$ as the average of $e^{\pm x}$ and use $\E[f(Z)e^{cZ}]=e^{c^2/2}\E f(Z+c)$ with $c=\pm\beta\sqrt s$, and we find $k=k_s$.
\end{proof}
\begin{proposition}\label{prop:nd-scalar}
Let $\beta>0$ and $h\ge0$, and let $\mu_+$ be the law of the Parisi center $M_{q_+}=\tanh X_{q_+}$. Then $\Gamma_{\mu_+}(s)\le0$ for every $s\ge0$.
\end{proposition}
\begin{proof}
Let $m=\tanh y\in(-1,1)$ and $r\ge0$, and let $\sigma^*$ be independent of $Z\sim N(0,1)$, with $\P(\sigma^*=\pm1)=(1\pm m)/2=e^{\pm y}/(2\cosh y)$. We claim that
\begin{equation}\label{eq:nd-psi-prime-tanh}
2\psi_m'(r)=\E\tanh^2\big(y+\sqrt r\,Z+r\sigma^*\big)-m^2.
\end{equation}
Indeed, $\tau^*=\sigma^*-m$ has law $\nu_m$. Let $W=\sqrt r\tau^*+Z$. For $\sigma\in\{-1,1\}$ and $\tau=\sigma-m$, we have
\[
\sqrt rW\tau-\frac r2\tau^2=\sigma\big(\sqrt rW+rm\big)-\sqrt rWm-\frac r2\big(1+m^2\big),
\]
where we used $\sigma^2=1$. Since the last two terms do not depend on $\sigma$, Bayes's rule gives $\P(\sigma^*=\sigma\mid W)\propto e^{\sigma(y+\sqrt rW+rm)}$, and $\E[\sigma^*\mid W]=\tanh(y+\sqrt rW+rm)=\tanh(y+\sqrt rZ+r\sigma^*)$. Since $b_{m,r}=\E[\sigma^*\mid W]-m$ and $\E\,\E[\sigma^*\mid W]=m$, we obtain $\E b_{m,r}^2=\E\tanh^2(y+\sqrt rZ+r\sigma^*)-m^2$, and \cref{lem:nd-psi}(b) gives \eqref{eq:nd-psi-prime-tanh}.
Let $s\in[0,1-q_+]$. We apply \cref{lem:nd-doob} with $g=\tanh^2$. By \eqref{eq:nd-psi-prime-tanh} with $r=\beta^2s$, the function $k_s$ of \cref{lem:nd-doob} is $k_s(y)=2\psi'_{\tanh y}(\beta^2s)+\tanh^2y$. By \eqref{eq:nd-X-top} and \eqref{eq:nd-psi-mu}, we obtain
\[
\E M_{q_++s}^2=\E\tanh^2X_{q_++s}=\E\big[2\psi'_{M_{q_+}}(\beta^2s)\big]+\E M_{q_+}^2=2\psi'_{\mu_+}(\beta^2s)+q_+ .
\]
Inserting this into \eqref{eq:nd-Gamma}, we find for $s\in[0,1-q_+]$ that
\begin{align*}
\Gamma_{\mu_+}(s)&=\beta^2\int_0^s\big(\E M_{q_++u}^2-q_+-u\big)\dd u=\beta^2\int_{q_+}^{q_++s}\big(\E M_t^2-t\big)\dd t\\
&=2\big(\Psi(q_+)-\Psi(q_++s)\big)\le0,
\end{align*}
where the inequality holds by \cref{thm:pre-optimality}, since $q_+\in\supp\nu$. Finally, let $s\ge1-q_+$. By \eqref{eq:nd-Gamma}, \eqref{eq:nd-psi-mu}, and \eqref{eq:nd-kappa-plus},
\[
\Gamma_{\mu_+}'(s)=\beta^2\big(2\psi_{\mu_+}'(\beta^2s)-s\big)\le\beta^2\big(1-q_+-s\big)\le0,
\]
which gives $\Gamma_{\mu_+}(s)\le\Gamma_{\mu_+}(1-q_+)\le0$.
\end{proof}
Since $\int m^2\mu_+(\mathrm dm)=q_+<1$, \cref{prop:nd-scalar} and~\cite[Proposition~13]{CPSGeneralizedTAP} imply that, for the SK model, the generalized TAP correction of Chen, Panchenko, and Subag at the law of $|M_{q_+}|$ is replica symmetric and equals the classical TAP correction. For mixed $p$-spin models, they noted that the correction is approximately classical at the generalized TAP states whose self-overlap is the largest point of the support of the Parisi measure~\cite[Remark~6]{CPSII}. By~\cite[Remark~15]{CPSGeneralizedTAP}, for some $\beta$ and $h>0$, the correction is not replica symmetric at some other laws whose second moment is $q_+$.
\section{The relative-entropy bound}\label{sec:nd-statement}
We recall the planted product model of \cref{def:planted}. For $m\in[-1,1]^V$, the planted configuration $v^*=\sigma^*-m$ has independent coordinates, and $v^*_x$ has law $\nu_{m_x}$. The law $\mathsf Q_m$ is the law of $Y=\beta a(v^*)+Z$ on $\R^{\mathsf E}$, where $a(v)_{xy}=\sqrt{S_{xy}}\,v_xv_y$ and $Z$ is a standard Gaussian vector, and $\mathcal D(m)=\KL{\mathsf Q_m}{\mathsf P_0}$. For $a\ge0$ we define the compact set
\[
\mathcal A_S(a)=a\mathbf 1+2\beta^2S[0,1]^V=\big\{a\mathbf 1+2\beta^2Sy:y\in[0,1]^V\big\}.
\]
Since $S$ has nonnegative entries and unit row sums, $\mathcal A_S(a)\subset[a,a+2\beta^2]^V$. For a probability measure $\mu$ on $[-1,1]$ and $m\in[-1,1]^V$, we define
\begin{equation}\label{eq:nd-disc}
\Delta_{S,a}(m;\mu)=\sup_{r\in\mathcal A_S(a)}\frac1n\sum_{x\in V}\big(\psi_{m_x}(r_x)-\psi_\mu(r_x)\big).
\end{equation}
Since the function inside the supremum is continuous in $(m,r)$ by \cref{lem:nd-psi}(d), and $\mathcal A_S(a)$ is compact, the map $\Delta_{S,a}(\cdot\,;\mu)$ is continuous on $[-1,1]^V$, and the same holds with the absolute value of the function inside the supremum.
\begin{theorem}\label{thm:nd}
Let $\beta\ge0$.
\begin{enumerate}[label=(\alph*)]
\item For every $a>0$ there exists a nonincreasing function $e_{\beta,a}\colon[1,\infty)\to[0,\infty)$, depending only on $\beta$ and $a$, with $\lim_{D\to\infty}e_{\beta,a}(D)=0$, such that the following holds. Let $D\ge1$, let $S$ be as in \cref{def:planted} with $\max_{x,y}S_{xy}\le1/D$, let $m\in[-1,1]^V$, and let $\mu$ be a probability measure on $[-1,1]$. Then
\[
\frac1n\mathcal D(m)\le\frac12\sup_{0\le s\le1}\Gamma_\mu(s)+2a+\Delta_{S,a}(m;\mu)+e_{\beta,a}(D).
\]
\item For each $k\ge1$, let $D_k\ge1$, let $S_k$ be an $n_k\times n_k$ matrix as in \cref{def:planted} with $\max_{x,y}(S_k)_{xy}\le1/D_k$, and let $m^{(k)}$ be a random vector in $[-1,1]^{n_k}$. Let $\mu$ be a probability measure on $[-1,1]$. Suppose that $D_k\to\infty$ and that for every $a>0$,
\begin{equation}\label{eq:nd-hom}
\lim_{k\to\infty}\E\sup_{r\in\mathcal A_{S_k}(a)}\bigg|\frac1{n_k}\sum_{x}\psi_{m^{(k)}_x}(r_x)-\frac1{n_k}\sum_x\psi_\mu(r_x)\bigg|=0.
\end{equation}
Then
\[
\limsup_{k\to\infty}\frac1{n_k}\E\,\mathcal D\big(m^{(k)}\big)\le\frac12\sup_{0\le s\le1}\Gamma_\mu(s).
\]
\end{enumerate}
\end{theorem}
\begin{corollary}\label{cor:nd}
Let $\beta>0$ and $h>0$, and let $(\mu_k)$ be probability measures on $[-1,1]$ that converge weakly to $\mu_+$. Then $\limsup_{k\to\infty}\sup_{0\le s\le1}\Gamma_{\mu_k}(s)\le0$.
\end{corollary}
\begin{remark}\label{rem:nd-indefinite}
The matrix $S$ need not be positive semidefinite. For example, if $G\in\cG_D$ is bipartite, as are the tori $\T^d_L$ with $L$ even and the hypercubes $Q_d$, then $-1$ is an eigenvalue of $A_G/D$. Besides symmetry, the proof of \cref{thm:nd} uses three properties of $S$. First, $S\preceq I$, which is used for the separable comparison function in \cref{sec:nd-proof}. Second, $S$ has nonnegative entries and unit row sums, which gives $S[0,1]^V\subset[0,1]^V$; this controls the set of velocities in \cref{lem:nd-barrier}. Third, $\|S\|\le1$, which enters the proof of \cref{prop:nd-contact} through the bound $|w\cdot Sw|\le|w||Sw|$; there, the quadratic form $w\cdot Sw$ is not assumed to have a sign.
\end{remark}
\section{Interpolating channels}\label{sec:nd-interpolation}
In this subsection and the next three, we fix $\beta>0$ and $m\in[-1,1]^V$. Let $v^*=(v^*_x)_{x\in V}$ have independent coordinates with laws $\nu_{m_x}$, and let $Z=(Z_e)_{e\in\mathsf E}$ and $\zeta=(\zeta_x)_{x\in V}$ be standard Gaussian vectors, with $v^*$, $Z$, and $\zeta$ independent. For $t\ge0$ and $q\in[0,\infty)^V$, we observe
\begin{equation}\label{eq:nd-channels}
Y_e=\beta(tS_{xy})^{1/2}v^*_xv^*_y+Z_e\quad(e=\{x,y\}\in\mathsf E),\qquad Y'_x=q_x^{1/2}v^*_x+\zeta_x\quad(x\in V).
\end{equation}
This is the setting of \cref{sec:nd-channels} with $N=|\mathsf E|+n$, with $\theta=((v^*_xv^*_y)_{\{x,y\}\in\mathsf E},(v^*_x)_{x\in V})$, and with $\lambda=\lambda(t,q)=((\beta^2tS_e)_{e\in\mathsf E},(q_x)_{x\in V})$, where $S_e=S_{xy}$ for $e=\{x,y\}$. We define
\[
F(t,q)=\frac1nI\big(\lambda(t,q)\big).
\]
Let $\nu_m=\bigotimes_x\nu_{m_x}$ be the law of $v^*$. For $v$ in its support, let
\begin{align*}
\mathcal E_{t,q}(v)={}&\sum_{\{x,y\}\in\mathsf E}\Big(\beta(tS_{xy})^{1/2}Z_{xy}v_xv_y+\beta^2tS_{xy}v^*_xv^*_yv_xv_y-\frac{\beta^2tS_{xy}}2v_x^2v_y^2\Big)\\
&+\sum_{x\in V}\Big(q_x^{1/2}\zeta_xv_x+q_xv^*_xv_x-\frac{q_x}2v_x^2\Big),
\end{align*}
and let
\[
\widehat F(t,q)=\frac1n\log\sum_v\nu_m(v)\,e^{\mathcal E_{t,q}(v)} .
\]
By \eqref{eq:nd-I-teacher}, we have $F(t,q)=\E\widehat F(t,q)$, where $\E$ is the expectation over $(v^*,Z,\zeta)$. The posterior average $\langle\cdot\rangle_{t,q}$ of \cref{sec:nd-channels} is the average over independent samples $v^1,v^2,\dots$ from the probability measure proportional to $\nu_m(v)e^{\mathcal E_{t,q}(v)}$; inside $\langle\cdot\rangle_{t,q}$ we write $v$ for $v^1$. All profiles $(t,q)$ are coupled through the same $(v^*,Z,\zeta)$. We write
\[
b_x=\langle v_x\rangle_{t,q},\qquad \Sigma_{xy}=\langle v_xv_y\rangle_{t,q}-b_xb_y,\qquad p_x=\E b_x^2 .
\]
These depend on $(t,q)$, which will be clear from the context. Since $|v_x|\le1+|m_x|\le2$ and $|v^*_x|\le2$, we have $|b_x|\le2$ and $|\Sigma_{xy}|\le8$.
\begin{lemma}\label{lem:nd-F}
Let $\beta>0$ and $m\in[-1,1]^V$.
\begin{enumerate}[label=(\alph*)]
\item We have $F(1,0)=n^{-1}\mathcal D(m)$ and $F(0,q)=n^{-1}\sum_x\psi_{m_x}(q_x)$ for $q\in[0,\infty)^V$.
\item The function $F$ is continuous on $[0,\infty)\times[0,\infty)^V$, nondecreasing in $t$ and in each coordinate $q_x$, and $|F(t,q)-F(t,q')|\le\|q-q'\|_n/2$.
\item Let $q\in(0,\infty)^V$. For every $t\ge0$, the function $F(t,\cdot)$ is twice continuously differentiable in a neighborhood of $q$, and
\[
2n\,\partial_{q_x}F(t,q)=p_x\in[0,\kappa_x],\qquad 2n\,\partial_{q_x}\partial_{q_y}F(t,q)=\E\Sigma_{xy}^2 .
\]
For every $t>0$, the function $F(\cdot,q)$ is differentiable at $t$, and
\[
\partial_tF(t,q)=\frac{\beta^2}{4n}\sum_{x,y\in V}S_{xy}\,\E\langle v_xv_y\rangle_{t,q}^2 .
\]
\end{enumerate}
\end{lemma}
\begin{proof}
(a) At $q=0$ the site observations are independent of everything else, and since $\beta S_{xy}^{1/2}v^*_xv^*_y=\beta a(v^*)_{xy}$, the law of the observation \eqref{eq:nd-channels} at $(1,0)$ is $\mathsf Q_m\otimes N(0,I_V)$. Its relative entropy with respect to $\mathsf P_0\otimes N(0,I_V)$ is $\mathcal D(m)$. At $t=0$ the edge observations are pure noise, and the site observations are independent over $x$; the law of the observation is then $N(0,I_{\mathsf E})\otimes\bigotimes_x\mathrm{Law}(q_x^{1/2}v^*_x+\zeta_x)$. Relative entropy is additive over products, and the relative entropy of $\mathrm{Law}(q_x^{1/2}v_x^*+\zeta_x)$ with respect to $N(0,1)$ is $\psi_{m_x}(q_x)$.
(b) and (c) The map $(t,q)\mapsto\lambda(t,q)$ is linear and has nonnegative coefficients. Continuity and monotonicity follow from \cref{lem:nd-channel}(a). If $t>0$ and $q\in(0,\infty)^V$, then all coordinates of $\lambda(t,q)$ are positive, and by \cref{lem:nd-channel}(b,c) the function $I$ has continuous first and second partial derivatives near $\lambda(t,q)$; since the coordinates of $\lambda$ indexed by $\mathsf E$ play no role in the derivatives in $q$, the same holds in $q$ for $t=0$. The formulas follow from \cref{lem:nd-channel}(b,c) and the chain rule, together with $\E(v^*_x)^2=\kappa_x$ and
\[
\sum_{e\in\mathsf E}\beta^2S_e\cdot\frac12\E\langle v_xv_y\rangle_{t,q}^2=\frac{\beta^2}4\sum_{x,y\in V}S_{xy}\E\langle v_xv_y\rangle_{t,q}^2,
\]
which holds because $S$ is symmetric with zero diagonal and $S_{xy}=0$ if $\{x,y\}\notin\mathsf E$. Finally, since $0\le\partial_{q_x}F\le1/(2n)$ on $(0,\infty)^V$, we have $|F(t,q)-F(t,q')|\le(2n)^{-1}\sum_x|q_x-q'_x|\le\|q-q'\|_n/2$ for $q,q'\in(0,\infty)^V$, and by continuity for $q,q'\in[0,\infty)^V$.
\end{proof}
\section{The score covariance identity}\label{sec:nd-score}
Fix $t\ge0$ and $q\in(0,\infty)^V$. Differentiating $\widehat F(t,\cdot)$ as in \eqref{eq:nd-derivatives}, we find that $\widehat F(t,\cdot)$ is smooth near $q$ and $2n\nabla_q\widehat F(t,q)=\chi$, where
\begin{equation}\label{eq:nd-score}
\chi_x=\frac{\zeta_xb_x}{\sqrt{q_x}}+2v^*_xb_x-\langle v_x^2\rangle_{t,q} .
\end{equation}
Differentiating once more, we find
\begin{equation}\label{eq:nd-quenched-hessian}
n\,\partial_{q_x}\partial_{q_y}\widehat F(t,q)=\langle A_xA_y\rangle_{t,q}-\langle A_x\rangle_{t,q}\langle A_y\rangle_{t,q}-\mathbf 1_{x=y}\frac{\zeta_xb_x}{4q_x^{3/2}},\qquad A_x(v)=\frac{\zeta_xv_x}{2\sqrt{q_x}}+v^*_xv_x-\frac{v_x^2}2 .
\end{equation}
The first two terms form a covariance matrix, which is positive semidefinite. Since $|b_x|\le2$, it follows that
\begin{equation}\label{eq:nd-hessian-lower}
\nabla_q^2\widehat F(t,q)\succeq-\frac1n\operatorname{diag}\Big(\frac{|\zeta_x|}{2q_x^{3/2}}\Big)_{x\in V}.
\end{equation}
\begin{lemma}\label{lem:nd-score}
Let $t\ge0$ and $q\in(0,\infty)^V$, and let $U=\chi-b^{\odot2}$. Then $\E\chi=p$, $\E U=0$, and
\[
\E\big[U_xU_y\big]=\mathbf 1_{x=y}\frac{p_x}{q_x}+\E\Sigma_{xy}^2\qquad(x,y\in V).
\]
\end{lemma}
\begin{proof}
By \cref{lem:nd-F}(c), $\E\chi_x=2n\partial_{q_x}F(t,q)=p_x$, and $\E U=0$. Since $\langle v_x^2\rangle_{t,q}=\Sigma_{xx}+b_x^2$, we have $U=U^1+U^2$ with
\[
U^1_x=2b_x(v^*_x-b_x),\qquad U^2_x=\frac{\zeta_xb_x}{\sqrt{q_x}}-\Sigma_{xx}.
\]
We first compute $\E[U^2_xU^2_y]$, conditionally on $(v^*,Z)$. Then $b$ is a smooth function of $\zeta$, and by \eqref{eq:nd-derivatives}, $\partial_{\zeta_y}b_x=\sqrt{q_y}\,\Sigma_{xy}$. Let $f_x=b_x/\sqrt{q_x}$ and write $\partial_y=\partial_{\zeta_y}$. Then $\partial_yf_x=(q_y/q_x)^{1/2}\Sigma_{xy}$ and $U^2_x=\zeta_xf_x-\partial_xf_x$. Since the functions $f_x$ have bounded derivatives of all orders in $\zeta$, Gaussian integration by parts~\ref{F:ibp} in $\zeta$ gives
\begin{align*}
\E[\zeta_xf_x\zeta_yf_y]&=\mathbf 1_{x=y}\E f_x^2+\E\big[\partial_y\partial_xf_x\,f_y+\partial_xf_x\,\partial_yf_y+\partial_yf_x\,\partial_xf_y+f_x\,\partial_x\partial_yf_y\big],\\
\E[\zeta_xf_x\,\partial_yf_y]&=\E\big[\partial_xf_x\,\partial_yf_y+f_x\,\partial_x\partial_yf_y\big],\qquad
\E[\partial_xf_x\,\zeta_yf_y]=\E\big[\partial_y\partial_xf_x\,f_y+\partial_xf_x\,\partial_yf_y\big],
\end{align*}
where the expectations are conditional on $(v^*,Z)$. Expanding the product $U^2_xU^2_y$ and combining these identities, we obtain
\[
\E\big[U^2_xU^2_y\big]=\mathbf 1_{x=y}\E f_x^2+\E\big[\partial_yf_x\,\partial_xf_y\big]=\mathbf 1_{x=y}\frac{\E b_x^2}{q_x}+\E\Sigma_{xy}^2,
\]
first conditionally on $(v^*,Z)$ and then after taking expectations.
Next, let $\mathcal Y$ be the $\sigma$-algebra generated by the observations \eqref{eq:nd-channels}. The quantities $b$ and $\Sigma$ are $\mathcal Y$-measurable, and by Bayes's rule the conditional law of $v^*$ given $\mathcal Y$ is the posterior. In particular, $\E[v^*_x-b_x\mid\mathcal Y]=0$ and $\E[(v^*_x-b_x)(v^*_y-b_y)\mid\mathcal Y]=\Sigma_{xy}$, which gives $\E[U^1_xU^1_y]=4\E[b_xb_y\Sigma_{xy}]$. Since $\zeta_y=Y'_y-\sqrt{q_y}v^*_y$, we have
\[
U^2_y=\frac{Y'_yb_y}{\sqrt{q_y}}-b_y^2-\Sigma_{yy}-b_y(v^*_y-b_y),
\]
and the first three terms are $\mathcal Y$-measurable. Hence $\E[U^1_xU^2_y]=-2\E[b_xb_y\Sigma_{xy}]$, and by symmetry $\E[U^2_xU^1_y]=-2\E[b_xb_y\Sigma_{xy}]$. When we sum the four contributions, the terms involving $b_xb_y\Sigma_{xy}$ cancel, and $\E[U_xU_y]=\E[U^2_xU^2_y]$.
\end{proof}
\section{Concentration}\label{sec:nd-concentration}
% CHECK: McDiarmid1989 is cited without a lemma number; the inequality is stated in the form used.
We use the bounded-differences inequality of McDiarmid~\cite{McDiarmid1989}: if $\xi_1,\dots,\xi_n$ are independent and $f$ changes by at most $c_x$ when the $x$th argument is changed, then $\P(|f(\xi)-\E f(\xi)|\ge\epsilon)\le2\exp(-2\epsilon^2/\sum_xc_x^2)$ for all $\epsilon\ge0$.
\begin{lemma}\label{lem:nd-conc}
Let $\beta>0$, $\Lambda>0$, $m\in[-1,1]^V$, $t\in[0,1]$, and $r\in[0,\Lambda]^V$. Then for every $\epsilon>0$,
\[
\P\big(|\widehat F(t,r)-F(t,r)|>\epsilon\big)\le4\exp\Big(-\frac{n\epsilon^2}{C_\Lambda}\Big),\qquad C_\Lambda=32\big(4\beta^2+\Lambda+1\big)^2.
\]
\end{lemma}
\begin{proof}
Fix $v^*$. By \eqref{eq:nd-derivatives}, $\partial_{Z_e}\widehat F=n^{-1}\beta(tS_{xy})^{1/2}\langle v_xv_y\rangle_{t,r}$ for $e=\{x,y\}$, and $\partial_{\zeta_x}\widehat F=n^{-1}r_x^{1/2}b_x$. Since $|\langle v_xv_y\rangle|\le4$, $|b_x|\le2$, and $\sum_{e\in\mathsf E}S_e=n/2$, the squared norm of the gradient of $\widehat F$ in $(Z,\zeta)$ is at most $n^{-2}(8\beta^2n+4\Lambda n)$. Therefore, $\widehat F$ is a Lipschitz function of $(Z,\zeta)$ with constant $((8\beta^2+4\Lambda)/n)^{1/2}$, and \ref{F:concentration} gives
\[
\P\big(|\widehat F-\Phi_0(v^*)|>\epsilon/2\;\big|\;v^*\big)\le2\exp\Big(-\frac{n\epsilon^2}{8(8\beta^2+4\Lambda)}\Big),\qquad \Phi_0(v^*)=\E\big[\widehat F\mid v^*\big].
\]
Next, changing the coordinate $v^*_x$ to the other point of the support of $\nu_{m_x}$ changes it by at most $2$ and changes $\mathcal E_{t,r}(v)$ by at most $2|v_x|(\beta^2t\sum_yS_{xy}|v^*_yv_y|+r_x)\le4(4\beta^2+\Lambda)$, for every $v$. It follows that $\widehat F$ changes by at most $4(4\beta^2+\Lambda)/n$, and so does $\Phi_0$. Since the coordinates of $v^*$ are independent and $\E\Phi_0(v^*)=F(t,r)$, McDiarmid's inequality implies that
\[
\P\big(|\Phi_0(v^*)-F(t,r)|>\epsilon/2\big)\le2\exp\Big(-\frac{n\epsilon^2}{32(4\beta^2+\Lambda)^2}\Big).
\]
Since $8(8\beta^2+4\Lambda)\le C_\Lambda$ and $32(4\beta^2+\Lambda)^2\le C_\Lambda$, the claim follows from the two bounds.
\end{proof}
The next lemma controls $\widehat F-F$ uniformly over a family of profiles of the form $q+\eta S^2y$. This family has small metric entropy, because most eigenvalues of $S$ are small: since $\tr S^2\le n/D$, at most $n/(D\alpha^2)$ eigenvalues of $S$ have absolute value larger than $\alpha$.
\begin{lemma}\label{lem:nd-net}
Let $\beta>0$ and $0<\vartheta\le\Lambda$. There exists $C_0$, depending only on $\beta$, $\vartheta$, and $\Lambda$, such that the following holds. Let $D\ge1$, let $S$ be as in \cref{def:planted} with $\max_{x,y}S_{xy}\le1/D$, and let $m\in[-1,1]^V$, $t\in[0,1]$, and $q\in[\vartheta,\Lambda]^V$. Let $A\ge1$, $\eta\in(0,\vartheta/(2A)]$, and $\alpha\in(0,1)$, and let $\mathcal M=\{q+\eta S^2y:y\in[-A,A]^V\}$. Then
\[
\E\sup_{r\in\mathcal M}\big|\widehat F(t,r)-F(t,r)\big|\le C_0\bigg(\Big(\frac{\log(3/\alpha^2)}{D\alpha^2}\Big)^{1/2}+\eta A\alpha^2\bigg).
\]
\end{lemma}
\begin{proof}
Since $S^2$ has nonnegative entries and unit row sums, $S^2y\in[-A,A]^V$ for $y\in[-A,A]^V$, and $\mathcal M$ is a compact convex subset of $[\vartheta/2,\Lambda+\vartheta/2]^V\subset[\vartheta/2,2\Lambda]^V$. The supremum is measurable, since it may be taken over a countable dense subset of $\mathcal M$ by the continuity of $\widehat F(t,\cdot)$ and $F(t,\cdot)$.
We first construct a net. Let $\Pi_\alpha$ be the orthogonal projection onto the span of the eigenvectors of $S$ whose eigenvalues have absolute value larger than $\alpha$, and let $k_\alpha$ be its rank. By \eqref{eq:nd-S}, $k_\alpha\alpha^2\le\tr S^2\le n/D$, and $|(I-\Pi_\alpha)S^2w|\le\alpha^2|w|$ for $w\in\R^V$. Let $\varrho=\alpha^2A\sqrt n$. For $y\in[-A,A]^V$, the vector $\Pi_\alpha S^2y$ lies in the ball of radius $A\sqrt n$ in the range of $\Pi_\alpha$, since $\|S\|\le1$. A maximal $\varrho$-separated subset $\mathcal C$ of this ball satisfies $|\mathcal C|\le(1+2A\sqrt n/\varrho)^{k_\alpha}\le(3/\alpha^2)^{k_\alpha}$, by comparing the volumes of the disjoint balls of radius $\varrho/2$ around its points with the ball of radius $A\sqrt n+\varrho/2$; and every point of the ball lies within distance $\varrho$ of $\mathcal C$. For each $c\in\mathcal C$ for which there exists $y\in[-A,A]^V$ with $|\Pi_\alpha S^2y-c|\le\varrho$, we choose one such $y_c$, and we let $\mathcal N$ be the set of the corresponding profiles $q+\eta S^2y_c\in\mathcal M$. Then $\mathcal N$ is deterministic and $|\mathcal N|\le(3/\alpha^2)^{k_\alpha}$. Let $y\in[-A,A]^V$, and choose $c\in\mathcal C$ with $|\Pi_\alpha S^2y-c|\le\varrho$. Then
\[
|S^2(y-y_c)|\le|\Pi_\alpha S^2(y-y_c)|+|(I-\Pi_\alpha)S^2(y-y_c)|\le2\varrho+\alpha^2|y-y_c|\le4\varrho .
\]
In particular, every $r\in\mathcal M$ lies within normalized distance $4\eta\alpha^2A$ of $\mathcal N$.
Next, by \cref{lem:nd-conc} with $\Lambda$ replaced by $2\Lambda$ and a union bound, $\P(\max_{r\in\mathcal N}|\widehat F(t,r)-F(t,r)|>\epsilon)\le4|\mathcal N|e^{-n\epsilon^2/C}$, where $C=C_{2\Lambda}$. Let $\epsilon_0=(C\log(4|\mathcal N|)/n)^{1/2}$. For $\epsilon\ge\epsilon_0$ we have $4|\mathcal N|e^{-n\epsilon^2/C}\le e^{-n(\epsilon-\epsilon_0)^2/C}$, since $\epsilon^2-\epsilon_0^2\ge(\epsilon-\epsilon_0)^2$. Integrating the tail bound, we obtain
\[
\E\max_{r\in\mathcal N}\big|\widehat F(t,r)-F(t,r)\big|\le\epsilon_0+\int_{\epsilon_0}^\infty e^{-n(\epsilon-\epsilon_0)^2/C}\dd\epsilon\le\Big(\frac{C\log(4|\mathcal N|)}n\Big)^{1/2}+\Big(\frac Cn\Big)^{1/2}.
\]
Further, $\log(4|\mathcal N|)\le\log4+n\log(3/\alpha^2)/(D\alpha^2)$. Since $n\ge D$, $\alpha<1$, and $\log(3/\alpha^2)\ge1$, both $\log 4/n$ and $1/n$ are at most $2\log(3/\alpha^2)/(D\alpha^2)$, and the right side is at most $4C^{1/2}(\log(3/\alpha^2)/(D\alpha^2))^{1/2}$.
Finally, we compare $\mathcal M$ with $\mathcal N$. Let $r'\in\mathcal M$. Since the coordinates of $r'$ are at least $\vartheta/2$, \eqref{eq:nd-score} gives $2n|\partial_{r_x}\widehat F(t,r')|\le2^{3/2}\vartheta^{-1/2}|\zeta_x|+12$, where we used $|b_x|\le2$, $|v^*_x|\le2$, and $\langle v_x^2\rangle\le4$. Since $\mathcal M$ is convex, for $r,r'\in\mathcal M$ we obtain
\[
|\widehat F(t,r)-\widehat F(t,r')|\le\frac1{2n}\sum_x\big(2^{3/2}\vartheta^{-1/2}|\zeta_x|+12\big)|r_x-r'_x|\le\big(2^{1/2}\vartheta^{-1/2}\|\zeta\|_n+6\big)\|r-r'\|_n,
\]
and $|F(t,r)-F(t,r')|\le\|r-r'\|_n/2$ by \cref{lem:nd-F}(b). Using $\E\|\zeta\|_n\le1$, we conclude that
\[
\E\sup_{r\in\mathcal M}\big|\widehat F(t,r)-F(t,r)\big|\le4C^{1/2}\Big(\frac{\log(3/\alpha^2)}{D\alpha^2}\Big)^{1/2}+4\eta\alpha^2A\big(2^{1/2}\vartheta^{-1/2}+7\big).
\]
This proves the lemma with $C_0=\max(4C_{2\Lambda}^{1/2},4(2^{1/2}\vartheta^{-1/2}+7))$.
\end{proof}
\section{The time derivative at contact points}\label{sec:nd-contact}
We now show that at every point where $F(t,\cdot)$ is touched from above by a function with Hessian at most $K/n$, the time derivative of $F$ is close to $\beta^2(4n)^{-1}p\cdot Sp$, where $p=2n\nabla_qF$.
\begin{definition}\label{def:nd-contact}
Let $\beta>0$, $m\in[-1,1]^V$, $0<\vartheta\le\Lambda$, and $K>0$. A point $(t,q)\in(0,1]\times[\vartheta,\Lambda]^V$ is a \emph{$(\vartheta,\Lambda,K)$-contact point} of $F$ if
\begin{equation}\label{eq:nd-contact}
F(t,q+h)\le F(t,q)+\nabla_qF(t,q)\cdot h+\frac K{2n}|h|^2\qquad\text{for every }h\in[-\vartheta/2,\vartheta/2]^V.
\end{equation}
\end{definition}
\begin{proposition}\label{prop:nd-contact}
Let $\beta>0$, $0<\vartheta\le\Lambda$, and $K>0$. There exists a nonincreasing function $\epsilon_{\beta,\vartheta,\Lambda,K}\colon[1,\infty)\to[0,\infty)$, depending only on $\beta$, $\vartheta$, $\Lambda$, and $K$, with $\lim_{D\to\infty}\epsilon_{\beta,\vartheta,\Lambda,K}(D)=0$, such that the following holds. Let $D\ge1$, let $S$ be as in \cref{def:planted} with $\max_{x,y}S_{xy}\le1/D$, let $m\in[-1,1]^V$, and let $(t,q)$ be a $(\vartheta,\Lambda,K)$-contact point of $F$. Then, with $p_x=\E b_x^2=2n\partial_{q_x}F(t,q)$,
\[
\partial_tF(t,q)\le\frac{\beta^2}{4n}\,p\cdot Sp+\epsilon_{\beta,\vartheta,\Lambda,K}(D).
\]
\end{proposition}
\begin{proof}
All posterior quantities are taken at the deterministic point $(t,q)$, and all uses of \eqref{eq:nd-nishimori}, through \cref{lem:nd-F,lem:nd-score}, are at this point. The perturbed profiles in Step 4 below are random, and they enter only through a pointwise Taylor bound and through \cref{lem:nd-net}.
\emph{Step 1: the Hessian.} By \cref{lem:nd-F}(c), $F(t,\cdot)$ is twice continuously differentiable near $q$. For $h\in\R^V$ and small $s>0$, Taylor's formula and \eqref{eq:nd-contact} with $sh$ in place of $h$ give $s^2h\cdot\nabla_q^2F(t,q)h/2+o(s^2)\le Ks^2|h|^2/(2n)$. This gives $\nabla^2_qF(t,q)\preceq(K/n)I$, and by \cref{lem:nd-F}(c) the matrix $\Xi=(\E\Sigma_{xy}^2)_{x,y\in V}$ satisfies
\begin{equation}\label{eq:nd-Xi}
\Xi\preceq2KI,\qquad \E\sum_{x,y}\Sigma_{xy}^2=\mathbf 1\cdot\Xi\mathbf 1\le2Kn .
\end{equation}
\emph{Step 2: reduction to the filtered fluctuation of $b^{\odot2}$.} Let $w=b^{\odot2}-p$, so that $\E w=0$ and $|w_x|\le4$. By \cref{lem:nd-F}(c) and $\langle v_xv_y\rangle=\Sigma_{xy}+b_xb_y$,
\[
\frac{4n}{\beta^2}\partial_tF(t,q)-p\cdot Sp=\sum_{x,y}S_{xy}\E\Sigma_{xy}^2+2\sum_{x,y}S_{xy}\E\big[\Sigma_{xy}b_xb_y\big]+\E\big[w\cdot Sw\big],
\]
where we used $\E[b^{\odot2}\cdot Sb^{\odot2}]-p\cdot Sp=\E[w\cdot Sw]$. By \eqref{eq:nd-Xi} and $S_{xy}\le1/D$, the first term is at most $2Kn/D$. By the Cauchy--Schwarz inequality, $|b_x|\le2$, \eqref{eq:nd-S}, and \eqref{eq:nd-Xi}, the second term is at most
\[
8\Big(\sum_{x,y}S_{xy}^2\Big)^{1/2}\Big(\E\sum_{x,y}\Sigma_{xy}^2\Big)^{1/2}\le8\Big(\frac nD\Big)^{1/2}(2Kn)^{1/2}.
\]
For the third term we only use $\|S\|\le1$ and $|w|\le4\sqrt n$: we have $|w\cdot Sw|\le|w||Sw|\le4\sqrt n|Sw|$, and the third term is at most $4n(\E\|Sw\|_n^2)^{1/2}$. Combining the three bounds, we obtain
\begin{equation}\label{eq:nd-reduction}
\partial_tF(t,q)-\frac{\beta^2}{4n}p\cdot Sp\le\frac{\beta^2}4\Big(\frac{2K}D+8\Big(\frac{2K}D\Big)^{1/2}+4\big(\E\|Sw\|_n^2\big)^{1/2}\Big).
\end{equation}
\emph{Step 3: the term $U$.} Let $\xi=\chi-p$, with $\chi$ as in \eqref{eq:nd-score}, and let $U=\chi-b^{\odot2}$ as in \cref{lem:nd-score}. Then $w=\xi-U$. By \cref{lem:nd-score}, \eqref{eq:nd-S}, and \eqref{eq:nd-Xi},
\[
\E|SU|^2=\tr\big(S\,\E[UU^{\mathsf T}]\,S\big)=\sum_x(S^2)_{xx}\frac{p_x}{q_x}+\tr(S\Xi S)\le\frac n{D\vartheta}+2K\tr S^2\le\Big(\frac1\vartheta+2K\Big)\frac nD,
\]
where we used $p_x\le1$, $q_x\ge\vartheta$, and $S\Xi S\preceq2KS^2$. This implies
\begin{equation}\label{eq:nd-w}
\E\|Sw\|_n^2\le2\E\|S\xi\|_n^2+2\Big(\frac1\vartheta+2K\Big)\frac1D .
\end{equation}
\emph{Step 4: the term $\xi$.} Fix $A\ge1$, $\eta\in(0,\vartheta/(2A)]$, and $\alpha\in(0,1)$. Let $\xi^A$ be the vector with coordinates $\max(-A,\min(A,\xi_x))$, and let $h=\eta S^2\xi^A$. Then $h\in[-\eta A,\eta A]^V\subset[-\vartheta/2,\vartheta/2]^V$, and $q+h$ lies in the set $\mathcal M$ of \cref{lem:nd-net}. The coordinates of the points of the segment from $q$ to $q+h$ are at least $\vartheta/2$, and \eqref{eq:nd-hessian-lower} gives $\nabla_q^2\widehat F\succeq-n^{-1}\operatorname{diag}(2^{1/2}\vartheta^{-3/2}|\zeta_x|)$ on this segment. Taylor's formula with integral remainder and $2n\nabla_q\widehat F(t,q)=\chi$ give
\[
\widehat F(t,q+h)-\widehat F(t,q)\ge\frac1{2n}\chi\cdot h-\frac{\eta^2A^2}{\vartheta^{3/2}}\cdot\frac1n\sum_x|\zeta_x|.
\]
On the other hand, \eqref{eq:nd-contact} and $2n\nabla_qF(t,q)=p$ give
\[
F(t,q+h)-F(t,q)\le\frac1{2n}p\cdot h+\frac K2\eta^2A^2 .
\]
Subtracting the second inequality from the first, we have
\[
\frac\eta{2n}\,\xi\cdot S^2\xi^A\le2\sup_{r\in\mathcal M}\big|\widehat F(t,r)-F(t,r)\big|+\eta^2A^2\Big(\vartheta^{-3/2}\frac1n\sum_x|\zeta_x|+\frac K2\Big).
\]
Taking expectations, using $\E|\zeta_x|\le1$ and applying \cref{lem:nd-net}, we obtain
\begin{equation}\label{eq:nd-clip}
\E\Big[\frac1n\xi\cdot S^2\xi^A\Big]\le\frac{4C_0}\eta\Big(\frac{\log(3/\alpha^2)}{D\alpha^2}\Big)^{1/2}+4C_0A\alpha^2+\eta A^2\big(2\vartheta^{-3/2}+K\big),
\end{equation}
where $C_0$ is the constant of \cref{lem:nd-net}. Next, by \eqref{eq:nd-score} and $q_x\ge\vartheta$, we have $|\xi_x|\le\varphi(\zeta_x)$ with $\varphi(z)=2\vartheta^{-1/2}|z|+13$. Let $G\sim N(0,1)$ and $\varpi(A)=\E[\varphi(G)^2\mathbf 1_{\varphi(G)>A}]$, which tends to zero as $A\to\infty$. Then $\E\|\xi\|_n^2\le\E\varphi(G)^2$ and $\E\|\xi-\xi^A\|_n^2\le\varpi(A)$. Using $\|S\|\le1$ and the Cauchy--Schwarz inequality, we find
\[
\E\|S\xi\|_n^2=\E\Big[\frac1n\xi\cdot S^2\xi^A\Big]+\E\Big[\frac1n\xi\cdot S^2(\xi-\xi^A)\Big]\le\E\Big[\frac1n\xi\cdot S^2\xi^A\Big]+\big(\E\varphi(G)^2\big)^{1/2}\varpi(A)^{1/2}.
\]
Let $e(D)$ be the infimum over $A\ge1$, $\eta\in(0,\vartheta/(2A)]$, and $\alpha\in(0,1)$ of the sum of the right side of \eqref{eq:nd-clip} and $(\E\varphi(G)^2)^{1/2}\varpi(A)^{1/2}$. Then $\E\|S\xi\|_n^2\le e(D)$, and $e$ depends only on $\beta$, $\vartheta$, $\Lambda$, and $K$, and is nonincreasing in $D$. Further, $e(D)\to0$ as $D\to\infty$. Indeed, given $\iota>0$, we first choose $A$ such that $(\E\varphi(G)^2)^{1/2}\varpi(A)^{1/2}\le\iota$, then $\eta\le\vartheta/(2A)$ such that $\eta A^2(2\vartheta^{-3/2}+K)\le\iota$, then $\alpha$ such that $4C_0A\alpha^2\le\iota$, and finally $D_0$ such that the first term in \eqref{eq:nd-clip} is at most $\iota$ for $D\ge D_0$. Then $e(D)\le4\iota$ for $D\ge D_0$.
\emph{Step 5: conclusion.} Combining \eqref{eq:nd-reduction}, \eqref{eq:nd-w}, and Step 4, we obtain the proposition with
\[
\epsilon_{\beta,\vartheta,\Lambda,K}(D)=\frac{\beta^2}4\bigg(\frac{2K}D+8\Big(\frac{2K}D\Big)^{1/2}+4\Big(2e(D)+2\Big(\frac1\vartheta+2K\Big)\frac1D\Big)^{1/2}\bigg).
\]
This function is nonincreasing and tends to zero as $D\to\infty$.
\end{proof}
\section{The scalar Hopf--Lax formula}\label{sec:nd-hopf}
\begin{lemma}\label{lem:nd-hopf}
Let $\beta\ge0$, let $\mu$ be a probability measure on $[-1,1]$, and for $t,r\ge0$ let
\[
f_\mu(t,r)=\max_{0\le s\le1}\Big(\psi_\mu(r+\beta^2ts)-\frac{\beta^2ts^2}4\Big).
\]
\begin{enumerate}[label=(\alph*)]
\item We have $f_\mu(0,r)=\psi_\mu(r)$ and $f_\mu(1,0)=\max_{0\le s\le1}\Gamma_\mu(s)/2$.
\item The function $f_\mu$ is nondecreasing in each variable, and $|f_\mu(t,r)-f_\mu(t',r')|\le\beta^2|t-t'|/4+|r-r'|/2$.
\item At every point $(t,r)\in(0,\infty)^2$ at which $f_\mu$ is differentiable, $\partial_tf_\mu=\beta^2(\partial_rf_\mu)^2$ and $0\le\partial_rf_\mu\le1/2$.
\end{enumerate}
\end{lemma}
\begin{proof}
We write $f=f_\mu$. The maximum exists because $\psi_\mu$ is continuous, and (a) follows from the definitions.
(b) For each $s$, the function $r\mapsto\psi_\mu(r+\beta^2ts)-\beta^2ts^2/4$ is nondecreasing and $1/2$-Lipschitz by \eqref{eq:nd-psi-mu}. Taking the maximum over $s$, we find that $f$ is nondecreasing and $1/2$-Lipschitz in $r$. Let $0\le t0$, let $(t,r)\in(0,\infty)^2$ be a point of differentiability of $f$, and let $s^*$ be a maximizer for $(t,r)$. The function $\tilde f(t',r')=\psi_\mu(r'+\beta^2t's^*)-\beta^2t's^{*2}/4$ satisfies $\tilde f\le f$, with equality at $(t,r)$, and it is differentiable at $(t,r)$, since $\psi_\mu$ is continuously differentiable and $r+\beta^2ts^*>0$. Since $f-\tilde f$ has a minimum at $(t,r)$, we have $\nabla f(t,r)=\nabla\tilde f(t,r)$, that is,
\[
\partial_rf(t,r)=\psi_\mu'(r+\beta^2ts^*),\qquad\partial_tf(t,r)=\beta^2s^*\psi_\mu'(r+\beta^2ts^*)-\frac{\beta^2s^{*2}}4 .
\]
The derivative of $s\mapsto\psi_\mu(r+\beta^2ts)-\beta^2ts^2/4$ is $\beta^2t(\psi_\mu'(r+\beta^2ts)-s/2)$. If $s^*=0$, it is nonpositive at $s=0$, and then $\psi'_\mu(r)=0$ by \eqref{eq:nd-psi-mu} and $\partial_rf=\partial_tf=0$. If $s^*\in(0,1)$, it vanishes at $s^*$. If $s^*=1$, it is nonnegative at $s=1$, and since $\psi'_\mu\le1/2$ by \eqref{eq:nd-psi-mu}, we have $\psi'_\mu(r+\beta^2t)=1/2$. In the last two cases $\psi'_\mu(r+\beta^2ts^*)=s^*/2$, and the formulas above give $\partial_rf=s^*/2$ and $\partial_tf=\beta^2s^{*2}/4=\beta^2(\partial_rf)^2$. In all cases $\partial_rf\in[0,1/2]$.
\end{proof}
The function $f_\mu$ is only Lipschitz, and we smooth it by averaging over small shifts toward larger times and larger profiles. We fix once and for all a smooth probability density $k$ on $\R^2$ whose support is a compact subset of $(1/2,1)^2$, and we let $c_k=\int|\partial_2k(u,v)|\dd u\dd v$.
\begin{lemma}\label{lem:nd-smooth}
Let $\beta\ge0$, let $\mu$ be a probability measure on $[-1,1]$, and let $\gamma>0$. For $(t,r)\in(-\gamma/2,\infty)^2$ let
\[
w(t,r)=\int_{\R^2}f_\mu(t+\gamma u,r+\gamma v)\,k(u,v)\dd u\dd v .
\]
Then $w$ is smooth, and for all $t,r>-\gamma/2$:
\begin{enumerate}[label=(\alph*)]
\item $0\le\partial_rw(t,r)\le1/2$ and $|\partial_r^2w(t,r)|\le c_k/(2\gamma)$;
\item $\partial_tw(t,r)\ge\beta^2(\partial_rw(t,r))^2$;
\item $w(0,r)\ge\psi_\mu(r)$ if $r\ge0$;
\item $w(1,r)\le f_\mu(1,0)+r/2+(\beta^2+2)\gamma/4$ if $r\ge0$.
\end{enumerate}
\end{lemma}
\begin{proof}
Since $k$ is supported in $(1/2,1)^2$, the arguments of $f_\mu$ in the definition of $w$ lie in $(0,\infty)^2$. Writing $w(t,r)=\gamma^{-2}\int f_\mu(t',r')k((t'-t)/\gamma,(r'-r)/\gamma)\dd t'\dd r'$ and differentiating under the integral shows that $w$ is smooth. By \cref{lem:nd-hopf}(b) and Rademacher's theorem, $f_\mu$ is differentiable almost everywhere in $(0,\infty)^2$, and its difference quotients are bounded. By dominated convergence,
\begin{gather*}
\partial_tw(t,r)=\int\partial_tf_\mu(t+\gamma u,r+\gamma v)\,k(u,v)\dd u\dd v,\\
\partial_rw(t,r)=\int\partial_rf_\mu(t+\gamma u,r+\gamma v)\,k(u,v)\dd u\dd v .
\end{gather*}
The second formula and \cref{lem:nd-hopf}(c) give $0\le\partial_rw\le1/2$. Writing the second formula as $\gamma^{-2}\int\partial_rf_\mu(t',r')k((t'-t)/\gamma,(r'-r)/\gamma)\dd t'\dd r'$ and differentiating in $r$, we obtain $|\partial_r^2w|\le(1/2)\gamma^{-1}c_k$, which proves (a). By \cref{lem:nd-hopf}(c) and Jensen's inequality,
\[
\partial_tw=\beta^2\int(\partial_rf_\mu)^2(t+\gamma u,r+\gamma v)\,k(u,v)\dd u\dd v\ge\beta^2(\partial_rw)^2,
\]
which proves (b). By \cref{lem:nd-hopf}(a,b), $f_\mu(\gamma u,r+\gamma v)\ge f_\mu(0,r)=\psi_\mu(r)$ for $u,v\ge0$, which gives (c). By \cref{lem:nd-hopf}(b), $f_\mu(1+\gamma u,r+\gamma v)\le f_\mu(1+\gamma,r+\gamma)\le f_\mu(1,0)+\beta^2\gamma/4+(r+\gamma)/2$ for $u,v\le1$, which gives (d).
\end{proof}
\section{Proof of the relative-entropy bound}\label{sec:nd-proof}
The comparison function in the proof of \cref{thm:nd} contains a penalty for the distance from a moving convex set of profiles. Since only the initial profiles in this set are used, the initial discrepancy is needed only on the set $\mathcal A_S(a)$.
\begin{lemma}\label{lem:nd-barrier}
Let $\beta>0$ and $a>0$, let $\mathcal V=\beta^2S[0,1]^V$, and for $t\in[0,1]$ let $\mathcal A_t=a\mathbf 1+(2-t)\mathcal V$. For $(t,q)\in[0,1]\times\R^V$, let $z(t,q)$ be the point of $\mathcal A_t$ closest to $q$, and let
\[
\mathcal B(t,q)=\frac1{2a}\operatorname{dist}_n(q,\mathcal A_t)^2=\frac1{2na}|q-z(t,q)|^2 .
\]
\begin{enumerate}[label=(\alph*)]
\item Each $\mathcal A_t$ is compact and convex, $a\mathbf 1\in\mathcal A_t$, $\mathcal A_t\subset[a,a+2\beta^2]^V$, and $\mathcal A_0=\mathcal A_S(a)$.
\item For each $t$, the function $\mathcal B(t,\cdot)$ is differentiable, $\nabla_q\mathcal B(t,q)=(q-z(t,q))/(na)$, and
\[
\mathcal B(t,q+h)\le\mathcal B(t,q)+\nabla_q\mathcal B(t,q)\cdot h+\frac{|h|^2}{2na}\qquad(q,h\in\R^V).
\]
\item For each $q$, the function $\mathcal B(\cdot,q)$ is differentiable on $[0,1]$, with one-sided derivatives at the endpoints, and
\[
\partial_t\mathcal B(t,q)=\max_{v\in\mathcal V}\nabla_q\mathcal B(t,q)\cdot v .
\]
\item The function $\mathcal B$ is continuous on $[0,1]\times\R^V$.
\end{enumerate}
\end{lemma}
\begin{proof}
(a) The set $\mathcal V$ is compact and convex, since it is the image of a cube under a linear map. It contains $0$, and $\mathcal V\subset[0,\beta^2]^V$ because $S$ has nonnegative entries and unit row sums. This gives (a).
(b) Fix $t$, and write $z=z(t,q)$ and $z'=z(t,q+h)$. Since $z\in\mathcal A_t$, we have $2na\mathcal B(t,q+h)\le|q+h-z|^2=|q-z|^2+2(q-z)\cdot h+|h|^2$, which is the claimed inequality. Since the projection onto a closed convex set is $1$-Lipschitz, $|z-z'|\le|h|$, and
\[
2na\mathcal B(t,q+h)=|q-z'|^2+2(q-z')\cdot h+|h|^2\ge|q-z|^2+2(q-z)\cdot h-2|h|^2 .
\]
From the two bounds, we see that $\mathcal B(t,\cdot)$ is differentiable at $q$ with the stated gradient.
(c) Fix $q$. For $s\in[1,2]$, let $v_s\in\mathcal V$ be such that $z(2-s,q)=a\mathbf 1+sv_s$; it is unique since $s>0$. Let $\phi(s)=|q-a\mathbf 1-sv_s|^2=\min_{v\in\mathcal V}|q-a\mathbf 1-sv|^2$, so that $\mathcal B(t,q)=\phi(2-t)/(2na)$. For $s,s'\in[1,2]$,
\[
\phi(s')\le|q-a\mathbf 1-s'v_s|^2=\phi(s)-2(s'-s)(q-a\mathbf 1-sv_s)\cdot v_s+(s'-s)^2|v_s|^2,
\]
and the same holds with $s$ and $s'$ exchanged. The map $s\mapsto v_s$ is continuous: if $s_j\to s$ and $\bar v$ is a limit point of $(v_{s_j})$, then $\bar v\in\mathcal V$ and $|q-a\mathbf 1-s\bar v|^2=\lim_j\phi(s_j)\le\lim_j|q-a\mathbf 1-s_jv_s|^2=\phi(s)$ along the corresponding subsequence, which forces $\bar v=v_s$. Combining this with the two inequalities, we see that $\phi$ is differentiable on $[1,2]$, with one-sided derivatives at the endpoints, and $\phi'(s)=-2(q-z)\cdot v_s$, where $z=z(2-s,q)$. Since $t=2-s$, this gives $\partial_t\mathcal B(t,q)=\nabla_q\mathcal B(t,q)\cdot v_s$. Finally, for $v\in\mathcal V$ the point $a\mathbf 1+sv$ lies in $\mathcal A_t$, and the variational characterization of the projection gives $(q-z)\cdot(a\mathbf 1+sv-z)\le0$, that is, $s(q-z)\cdot(v-v_s)\le0$. Hence $\nabla_q\mathcal B(t,q)\cdot v\le\nabla_q\mathcal B(t,q)\cdot v_s$ for every $v\in\mathcal V$.
(d) Every point $a\mathbf 1+(2-t)v$ of $\mathcal A_t$, with $v\in\mathcal V$, is within normalized distance $|t-t'|\|v\|_n\le\beta^2|t-t'|$ of the point $a\mathbf 1+(2-t')v$ of $\mathcal A_{t'}$, and vice versa. This gives $|\operatorname{dist}_n(q,\mathcal A_t)-\operatorname{dist}_n(q',\mathcal A_{t'})|\le\|q-q'\|_n+\beta^2|t-t'|$.
\end{proof}
\begin{proof}[Proof of \cref{thm:nd}]
(a) If $\beta=0$, then $\mathsf Q_m=\mathsf P_0$, $\mathcal D(m)=0$, $\Gamma_\mu=0$, $\mathcal A_S(a)=\{a\mathbf 1\}$, and $\Delta_{S,a}(m;\mu)\ge-\psi_\mu(a)\ge-a/2$, so (a) holds with $e_{0,a}=0$.
Let $\beta>0$ and $a>0$. Let $\gamma=a/(1+\beta^2)$, $\vartheta=a/2$, $\Lambda=3a/2+2\beta^2$, and $K=c_k/(2\gamma)+1/a$, and let $e_{\beta,a}=\epsilon_{\beta,\vartheta,\Lambda,K}$ be the function of \cref{prop:nd-contact}. Fix $D$, $S$, $m$, and $\mu$ as in the statement. Let $w$ be the function of \cref{lem:nd-smooth} for $\mu$ and $\gamma$, and let
\[
\Omega(t,q)=\frac1n\sum_{x\in V}w(t,q_x)\qquad(t\in[0,1],\ q\in[0,\infty)^V).
\]
Let $\mathcal V$, $\mathcal A_t$, $z(t,q)$, and $\mathcal B$ be as in \cref{lem:nd-barrier}. Fix $\rho>e_{\beta,a}(D)$, and let
\[
\mathcal L(t,q)=F(t,q)-\Omega(t,q)-\mathcal B(t,q)-\rho t\qquad\big((t,q)\in[0,1]\times[0,\infty)^V\big).
\]
\emph{Step 1: $\mathcal L$ attains its maximum.} By \cref{lem:nd-F}(b), $F(t,q)\le F(1,0)+\|q\|_n/2$. Further, $\Omega\ge0$, because $f_\mu\ge\psi_\mu\ge0$. By \cref{lem:nd-barrier}(a), $\operatorname{dist}_n(q,\mathcal A_t)\ge\|q\|_n-a-2\beta^2$. Combining these bounds, we see that $\mathcal L(t,q)\to-\infty$ as $\|q\|_n\to\infty$, uniformly in $t$. Since $\mathcal L$ is continuous by \cref{lem:nd-F}(b), \cref{lem:nd-smooth}, and \cref{lem:nd-barrier}(d), it attains its maximum on $[0,1]\times[0,\infty)^V$ at some point $(t_0,q_0)$.
\emph{Step 2: $q_0\in(0,\infty)^V$.} Suppose that $(q_0)_x=0$ for some $x$, let $e_x$ be the corresponding coordinate vector, and let $h\in(0,a)$. By \cref{lem:nd-F}(b), $F(t_0,q_0+he_x)\ge F(t_0,q_0)$. By \cref{lem:nd-smooth}(a), $\Omega(t_0,q_0+he_x)-\Omega(t_0,q_0)\le h/(2n)$. Let $z=z(t_0,q_0)$. Then $z_x\ge a$ by \cref{lem:nd-barrier}(a), and \cref{lem:nd-barrier}(b) gives
\[
\mathcal B(t_0,q_0+he_x)-\mathcal B(t_0,q_0)\le\frac{h((q_0)_x-z_x)}{na}+\frac{h^2}{2na}\le-\frac hn+\frac{h^2}{2na}.
\]
Hence $\mathcal L(t_0,q_0+he_x)-\mathcal L(t_0,q_0)\ge(h/n)(1/2-h/(2a))>0$, which contradicts the maximality of $(t_0,q_0)$.
\emph{Step 3: location of $q_0$.} By Step 2, \cref{lem:nd-F}(c), \cref{lem:nd-smooth}, and \cref{lem:nd-barrier}(b), the function $\mathcal L(t_0,\cdot)$ is differentiable at $q_0$, and $\nabla_q\mathcal L(t_0,q_0)=0$ by maximality. Let
\[
p=2n\nabla_qF(t_0,q_0),\qquad p'=2n\nabla_q\Omega(t_0,q_0)=\big(2\,\partial_rw(t_0,(q_0)_x)\big)_{x\in V}.
\]
By \cref{lem:nd-F}(c) and \cref{lem:nd-smooth}(a), $p,p'\in[0,1]^V$. Let $z=z(t_0,q_0)$. By \cref{lem:nd-barrier}(b),
\[
\frac{q_0-z}a=n\nabla_q\mathcal B(t_0,q_0)=\frac{p-p'}2\in\Big[-\frac12,\frac12\Big]^V.
\]
Hence $|(q_0)_x-z_x|\le a/2$ for all $x$, and \cref{lem:nd-barrier}(a) gives $q_0\in[a/2,3a/2+2\beta^2]^V=[\vartheta,\Lambda]^V$ and $\operatorname{dist}_n(q_0,\mathcal A_{t_0})\le a/2$.
\emph{Step 4: $t_0=0$.} Suppose that $t_0>0$. We first show that $(t_0,q_0)$ is a $(\vartheta,\Lambda,K)$-contact point of $F$. Let $h\in[-\vartheta/2,\vartheta/2]^V$. Then $q_0+h\in[0,\infty)^V$, and $\mathcal L(t_0,q_0+h)\le\mathcal L(t_0,q_0)$. By Taylor's formula and \cref{lem:nd-smooth}(a), and by \cref{lem:nd-barrier}(b), we obtain
\begin{align*}
F(t_0,q_0+h)-F(t_0,q_0)&\le\Omega(t_0,q_0+h)-\Omega(t_0,q_0)+\mathcal B(t_0,q_0+h)-\mathcal B(t_0,q_0)\\
&\le\nabla_q(\Omega+\mathcal B)(t_0,q_0)\cdot h+\Big(\frac{c_k}{4\gamma}+\frac1{2a}\Big)\frac{|h|^2}n .
\end{align*}
Since $\nabla_q(\Omega+\mathcal B)(t_0,q_0)=\nabla_qF(t_0,q_0)$ and $c_k/(4\gamma)+1/(2a)=K/2$, this is \eqref{eq:nd-contact}. By \cref{prop:nd-contact},
\[
\partial_tF(t_0,q_0)\le\frac{\beta^2}{4n}p\cdot Sp+e_{\beta,a}(D).
\]
Next, by \cref{lem:nd-smooth}(b) and $S\preceq I$,
\[
\partial_t\Omega(t_0,q_0)=\frac1n\sum_x\partial_tw(t_0,(q_0)_x)\ge\frac{\beta^2}n\sum_x\big(\partial_rw(t_0,(q_0)_x)\big)^2=\frac{\beta^2}{4n}|p'|^2\ge\frac{\beta^2}{4n}p'\cdot Sp' .
\]
Further, since $S$ is symmetric and $p-p'=2n\nabla_q\mathcal B(t_0,q_0)$ by Step 3,
\[
\frac{\beta^2}{4n}\big(p\cdot Sp-p'\cdot Sp'\big)=\frac{\beta^2}{4n}S(p+p')\cdot(p-p')=\tilde v\cdot\nabla_q\mathcal B(t_0,q_0),\qquad\tilde v=\beta^2S\Big(\frac{p+p'}2\Big).
\]
Since $(p+p')/2\in[0,1]^V$, we have $\tilde v\in\mathcal V$, and \cref{lem:nd-barrier}(c) gives $\tilde v\cdot\nabla_q\mathcal B(t_0,q_0)\le\partial_t\mathcal B(t_0,q_0)$. Combining the last three displays, we find
\[
\partial_tF(t_0,q_0)\le\partial_t\Omega(t_0,q_0)+\partial_t\mathcal B(t_0,q_0)+e_{\beta,a}(D)<\partial_t\Omega(t_0,q_0)+\partial_t\mathcal B(t_0,q_0)+\rho,
\]
that is, $\partial_t\mathcal L(t_0,q_0)<0$, where at $t_0=1$ all time derivatives are left derivatives. However, $t\mapsto\mathcal L(t,q_0)$ is differentiable at $t_0$ by \cref{lem:nd-F}(c), \cref{lem:nd-smooth}, and \cref{lem:nd-barrier}(c), and since it attains its maximum over $[0,1]$ at $t_0\in(0,1]$, we have $\partial_t\mathcal L(t_0,q_0)\ge0$. This contradiction shows that $t_0=0$.
\emph{Step 5: conclusion.} Since $a\mathbf 1\in\mathcal A_1$, we have $\mathcal B(1,a\mathbf 1)=0$, and $\Omega(1,a\mathbf 1)=w(1,a)$. Since $\mathcal B\ge0$ and $t_0=0$, the maximality of $(0,q_0)$ gives
\begin{align*}
F(1,a\mathbf 1)-w(1,a)-\rho&=\mathcal L(1,a\mathbf 1)\le\mathcal L(0,q_0)\le F(0,q_0)-\Omega(0,q_0)\\
&\le\frac1n\sum_x\big(\psi_{m_x}((q_0)_x)-\psi_\mu((q_0)_x)\big),
\end{align*}
where in the last step we used \cref{lem:nd-F}(a) and \cref{lem:nd-smooth}(c). By Step 3 and \cref{lem:nd-barrier}(a), there exists $z\in\mathcal A_0=\mathcal A_S(a)$ with $\|q_0-z\|_n\le a/2$. Since $\psi_{m_x}$ and $\psi_\mu$ are $1/2$-Lipschitz, the right side is at most
\[
\frac1n\sum_x\big(\psi_{m_x}(z_x)-\psi_\mu(z_x)\big)+\frac1n\sum_x|(q_0)_x-z_x|\le\Delta_{S,a}(m;\mu)+\|q_0-z\|_n\le\Delta_{S,a}(m;\mu)+\frac a2 .
\]
By \cref{lem:nd-F}(a,b), \cref{lem:nd-smooth}(d), and \cref{lem:nd-hopf}(a), and since $(\beta^2+2)\gamma/4\le a/2$, we have
\begin{align*}
\frac1n\mathcal D(m)=F(1,0)\le F(1,a\mathbf 1)&\le w(1,a)+\rho+\Delta_{S,a}(m;\mu)+\frac a2\\
&\le\frac12\sup_{0\le s\le1}\Gamma_\mu(s)+\frac{3a}2+\Delta_{S,a}(m;\mu)+\rho .
\end{align*}
Letting $\rho\downarrow e_{\beta,a}(D)$ proves (a).
(b) Fix $a>0$. For each $k$, we apply (a) with $S=S_k$ and $D=D_k$ to each value of $m^{(k)}$. The function $\mathcal D$ is continuous on $[-1,1]^{n_k}$, by \cref{lem:nd-F}(a) and the dominated convergence theorem applied to \eqref{eq:nd-I-teacher} (see also \cref{sec:band}), and $\Delta_{S_k,a}(\cdot\,;\mu)$ is continuous. Taking expectations, we obtain
\[
\frac1{n_k}\E\,\mathcal D\big(m^{(k)}\big)\le\frac12\sup_{0\le s\le1}\Gamma_\mu(s)+2a+\E\sup_{r\in\mathcal A_{S_k}(a)}\bigg|\frac1{n_k}\sum_x\big(\psi_{m^{(k)}_x}(r_x)-\psi_\mu(r_x)\big)\bigg|+e_{\beta,a}(D_k).
\]
By \eqref{eq:nd-hom} and $D_k\to\infty$, the limit superior of the right side as $k\to\infty$ is at most $\sup_{0\le s\le1}\Gamma_\mu(s)/2+2a$. Letting $a\downarrow0$ proves (b).
\end{proof}
\begin{proof}[Proof of \cref{cor:nd}]
Let $r\ge0$. By \cref{lem:nd-psi}(d), the function $m\mapsto\psi_m(r)$ is bounded and continuous on $[-1,1]$, and we have $\psi_{\mu_k}(r)\to\psi_{\mu_+}(r)$ by the weak convergence. By \eqref{eq:nd-psi-mu}, the functions $\psi_{\mu_k}$ and $\psi_{\mu_+}$ are $1/2$-Lipschitz, and a finite net of $[0,\beta^2]$ shows that the convergence is uniform on $[0,\beta^2]$. Then $\sup_{0\le s\le1}|\Gamma_{\mu_k}(s)-\Gamma_{\mu_+}(s)|=2\sup_{0\le r\le\beta^2}|\psi_{\mu_k}(r)-\psi_{\mu_+}(r)|\to0$, and \cref{prop:nd-scalar} completes the proof.
\end{proof}
% ---------- 08-proofs.tex
\chapter{Proofs of the main theorems}\label{sec:proofs}
This section proves \cref{thm:main,thm:hypercube,cor:ground}. \Cref{sec:geo-fixed} combines the results of \cref{sec:amp,sec:band,sec:nondetection} into a lower bound for tori of a fixed side length and for hypercubes. \Cref{sec:geo-side} shows that the pressure of the torus $\T^d_L$ differs from a limit that does not depend on $L$ by at most $\beta^2/(4L)$, uniformly in $d$. \Cref{sec:geo-main} completes the proofs, using \cref{prop:upper} for the upper bound.
\section{The lower bound at a fixed side length}\label{sec:geo-fixed}
\begin{proposition}\label{prop:geo-fixed}
Let $\beta\ge0$ and $h\in\R$. For every integer $L\ge3$,
\[
\liminf_{d\to\infty}p_{d,L}(\beta,h)\ge\PSK(\beta,h),
\qquad\text{and}\qquad
\liminf_{d\to\infty}p_{Q_d}(\beta,h)\ge\PSK(\beta,h).
\]
\end{proposition}
\begin{proof}
At $\beta=0$ both sides equal $\log2\cosh h$, by \cref{lem:pre-basic}(a) for $p_G$ and for $p^{\mathrm{SK}}_N$. Since $p_G$ and $\PSK$ are even in $h$, we may assume that $\beta>0$ and $h\ge0$. Let $(G_d)_{d\ge1}$ be either the tori $\T^d_L$, with $D_d=2d$, or the hypercubes $Q_d$, with $D_d=d$. In both cases $G_d\in\cG_{D_d}$, $D_d\to\infty$, and $n=|V(G_d)|\to\infty$.
We first reduce to $h>0$. By \cref{lem:pre-basic}(d), which also holds for $\PSK$, we have $|p_{G_d}(\beta,0)-p_{G_d}(\beta,\eta)|\le\eta$ and $|\PSK(\beta,0)-\PSK(\beta,\eta)|\le\eta$ for every $\eta>0$. If the claim holds at $(\beta,\eta)$, then
\[
\liminf_{d\to\infty}p_{G_d}(\beta,0)\ge\liminf_{d\to\infty}p_{G_d}(\beta,\eta)-\eta\ge\PSK(\beta,\eta)-\eta\ge\PSK(\beta,0)-2\eta,
\]
and letting $\eta\to0$ gives the claim at $(\beta,0)$. It therefore suffices to treat $\beta>0$ and $h>0$. We write $S=A_{G_d}/D_d$, which satisfies the hypotheses of \cref{def:planted} and $\max_{x,y}S_{xy}=1/D_d$.
Fix $\varepsilon\in(0,1)$. Let $T$, $\mu_\varepsilon$, and the query histories of depth $T$ on $G_d$ with centers $m=m^{(d)}$ be given by \cref{prop:amp-centre}. We first bound the relative entropy of the planted product model at $m$. For $m'\in[-1,1]$ and $r\ge0$ let $\Upsilon(m',r)=\psi_{m'}(r)$. By \cref{lem:nd-psi}, $\Upsilon$ is continuous on $[-1,1]\times[0,\infty)$ and $1/2$-Lipschitz in its second variable, and by the definition of $\psi_\mu$, we have $\int\Upsilon(m',r)\mu_\varepsilon(\mathrm dm')=\psi_{\mu_\varepsilon}(r)$. For every $a>0$, \cref{prop:amp-centre}(c), applied with $c=2\beta^2$ and $B=1$, is the hypothesis \eqref{eq:nd-hom} of \cref{thm:nd}(b) with $\mu=\mu_\varepsilon$. Then \cref{thm:nd}(b) implies that
\begin{equation}\label{eq:geo-nd}
\limsup_{d\to\infty}\frac1n\E\,\mathcal D(m)\le\omega(\varepsilon),\qquad\omega(\varepsilon)=\frac12\sup_{0\le s\le1}\Gamma_{\mu_\varepsilon}(s).
\end{equation}
We have $\omega(\varepsilon)\ge0$, because $\Gamma_{\mu_\varepsilon}(0)=0$. Further, $\mu_\varepsilon\to\mu_+$ weakly as $\varepsilon\to0$, since $\mathrm W_1(\mu_\varepsilon,\mu_+)\le\varepsilon$ by \cref{prop:amp-centre}(b), and \cref{cor:nd} gives $\limsup_{\varepsilon\to0}\omega(\varepsilon)\le0$.
By \eqref{eq:geo-nd}, \cref{thm:band} applies with $\delta=\omega(\varepsilon)$. Combining it with \cref{prop:amp-centre}(a), we obtain
\begin{align*}
\liminf_{d\to\infty}p_{G_d}(\beta,h)&\ge\liminf_{d\to\infty}\frac1n\E\Big[\beta H_{G_d}(m)+h\sum_xm_x+\sum_x\ent(m_x)+\frac{\beta^2}4\langle\kappa,S\kappa\rangle\Big]-4\beta\sqrt{\omega(\varepsilon)}\\
&\ge\PSK(\beta,h)-\varepsilon-4\beta\sqrt{\omega(\varepsilon)}.
\end{align*}
The left side does not depend on $\varepsilon$, and letting $\varepsilon\to0$ completes the proof.
\end{proof}
\section{The side length}\label{sec:geo-side}
The proof of \cref{thm:main} reduces the uniformity in $L$ to the fixed side lengths of \cref{prop:geo-fixed}. For this we compare tori with boxes. It is convenient to allow graphs that are not regular. For a finite simple graph $\Gamma=(V_\Gamma,E_\Gamma)$ and a number $D\ge1$, we define
\[
p^D_\Gamma(\beta,h)=\frac1{|V_\Gamma|}\E\log\sum_{\sigma\in\{-1,1\}^{V_\Gamma}}\exp\Big(\frac{\beta}{\sqrt D}\sum_{xy\in E_\Gamma}g_{xy}\sigma_x\sigma_y+h\sum_{x\in V_\Gamma}\sigma_x\Big),
\]
where $(g_e)_{e\in E_\Gamma}$ are independent standard Gaussian variables. If $\Gamma\in\cG_D$, then $p^D_\Gamma=p_\Gamma$.
The next lemma states that the pressure increases when Gaussian couplings are added, and bounds the increase. The monotonicity was proved by Contucci and Graffi for short-range Gaussian spin glasses, in the form of the superadditivity of the pressure~\cite[Theorem~2]{ContucciGraffi2004monotonicity}, using the interpolation of Guerra and Toninelli~\cite{GuerraToninelli2002Thermodynamic}. Contucci and Lebowitz extended it to couplings with symmetric distributions~\cite[Theorem~1]{ContucciLebowitz2007correlation}. We include the short proof, since we also need the upper bound.
% CHECK: ContucciGraffi2004monotonicity Theorem 2 (arXiv:math-ph/0302013v2) and ContucciLebowitz2007correlation Theorem 1
% (arXiv:cond-mat/0612371v1) verified against the arXiv versions only.
\begin{lemma}\label{lem:geo-edge}
Let $D\ge1$, $\beta\ge0$, and $h\in\R$, and let $\Gamma$ and $\Gamma'$ be finite simple graphs on the same set of $n$ vertices such that $E_\Gamma\subset E_{\Gamma'}$ and $E_{\Gamma'}\setminus E_\Gamma$ has $k$ elements. Then
\[
0\le p^D_{\Gamma'}(\beta,h)-p^D_\Gamma(\beta,h)\le\frac{\beta^2k}{2Dn}.
\]
Further, $\log2\cosh h\le p^D_\Gamma(\beta,h)\le\log2\cosh h+\beta^2|E_\Gamma|/(2Dn)$, and if $\Gamma$ is the disjoint union of graphs $\Gamma_1,\dots,\Gamma_r$, then $p^D_\Gamma=\sum_{j=1}^r(|V_{\Gamma_j}|/n)\,p^D_{\Gamma_j}$.
\end{lemma}
\begin{proof}
Let $F=E_{\Gamma'}\setminus E_\Gamma$. For $s\in[0,1]$ let $\langle\cdot\rangle_s$ be the Gibbs average for the exponent
\[
\frac{\beta}{\sqrt D}\sum_{xy\in E_\Gamma}g_{xy}\sigma_x\sigma_y+\frac{\beta\sqrt s}{\sqrt D}\sum_{xy\in F}g_{xy}\sigma_x\sigma_y+h\sum_{x}\sigma_x,
\]
and let $\varphi(s)$ be $n^{-1}$ times the expected logarithm of the corresponding partition function. Then $\varphi(0)=p^D_\Gamma$ and $\varphi(1)=p^D_{\Gamma'}$. For $s\in(0,1]$, Gaussian integration by parts~\ref{F:ibp} in $g_{xy}$, $xy\in F$, gives
\[
\varphi'(s)=\frac{\beta}{2n\sqrt{sD}}\sum_{xy\in F}\E\big[g_{xy}\langle\sigma_x\sigma_y\rangle_s\big]=\frac{\beta^2}{2nD}\sum_{xy\in F}\big(1-\E\langle\sigma_x\sigma_y\rangle_s^2\big)\in\Big[0,\frac{\beta^2k}{2nD}\Big].
\]
Since $\varphi$ is continuous on $[0,1]$, integrating gives the first claim. The bounds on $p^D_\Gamma$ follow from the first claim applied to the graph with vertex set $V_\Gamma$ and no edges, whose pressure is $\log2\cosh h$, and to $\Gamma$. For a disjoint union, the partition function is the product of the partition functions of the components, which are independent, and the last claim follows.
\end{proof}
For integers $d\ge1$ and $N\ge1$ let $\Lambda^d_N=\{0,\dots,N-1\}^d$ be the box in which two vertices are adjacent if they differ by $1$ in exactly one coordinate, and write $b_{d,N}=p^{2d}_{\Lambda^d_N}$ for its pressure with couplings of variance $1/(2d)$. More generally, a box is a product of $d$ intervals of integers, with the same adjacency.
The existence of the thermodynamic limit of short-range spin glasses is classical~\cite{Vuillermot1977thermodynamics,KhaninSinai1979existence,vanEntervanHemmen1983thermodynamic}; see also~\cite{ContucciGraffi2004monotonicity,ContucciGiardinaPule2004thermodynamic,ContucciLebowitz2007correlation}. Contucci and Graffi compared free and periodic boundary conditions for the EA model with the interpolation of \cref{lem:geo-edge}, and they bounded the difference of the pressures by a multiple of the size of the boundary~\cite[Theorems~1 and~2]{ContucciGraffi2004surface}. For the torus, the existence of the limit and a bound of order $1/L$ on the difference with the free-boundary pressure were recently proved by Wang, Zhu, and Chueng under weak moment assumptions~\cite{WangZhuChueng2026torus}. The next lemma is a form of these bounds that is uniform in $d$.
% CHECK: ContucciGraffi2004surface Theorems 1 and 2 verified against arXiv:math-ph/0306006v3 only.
\begin{lemma}\label{lem:geo-side}
Let $d\ge1$, $\beta\ge0$, and $h\in\R$. The limit $p^\infty_d(\beta,h)=\lim_{N\to\infty}b_{d,N}(\beta,h)$ exists, and for every integer $L\ge3$,
\[
\big|p_{d,L}(\beta,h)-p^\infty_d(\beta,h)\big|\le\frac{\beta^2}{4L}.
\]
\end{lemma}
\begin{proof}
We first compare the torus with the box. Let $L\ge3$. The torus $\T^d_L$ is obtained from $\Lambda^d_L$ by adding, for each coordinate $j\le d$ and each choice of the other $d-1$ coordinates, the edge between the two vertices whose $j$th coordinates are $L-1$ and $0$. Since $L\ge3$, these $dL^{d-1}$ edges are not edges of $\Lambda^d_L$. Since $\T^d_L\in\cG_{2d}$ has $L^d$ vertices, \cref{lem:geo-edge} with $D=2d$ gives
\begin{equation}\label{eq:geo-torus-box}
0\le p_{d,L}-b_{d,L}\le\frac{\beta^2dL^{d-1}}{2\cdot2d\cdot L^d}=\frac{\beta^2}{4L}.
\end{equation}
Next, we compare boxes of different sizes. Fix integers $m\ge1$ and $N\ge m$, and write $N=km+r$ with $k=\lfloor N/m\rfloor$ and $0\le r0$, one interval of length $r$. Taking products, we obtain a partition of $\Lambda^d_N$ into boxes, of which $k^d$ are translates of $\Lambda^d_m$. Let $\Gamma$ be the graph on $\Lambda^d_N$ whose edges are the edges of $\Lambda^d_N$ with both endpoints in the same box of the partition. For each of the $d$ coordinate directions, there are $N^{d-1}$ lines in $\Lambda^d_N$ parallel to it, and on each of them at most $k$ edges join different intervals. In total, at most $dkN^{d-1}\le dN^d/m$ edges of $\Lambda^d_N$ are not edges of $\Gamma$. By \cref{lem:geo-edge} with $D=2d$, we have $0\le b_{d,N}-p^{2d}_\Gamma\le\beta^2/(4m)$. Further, $p^{2d}_\Gamma$ is the average of the pressures of the boxes of the partition, weighted by their numbers of vertices. Each translate of $\Lambda^d_m$ has pressure $b_{d,m}$. Every box with $n'$ vertices has at most $dn'$ edges, and by \cref{lem:geo-edge} its pressure, like $b_{d,m}$, lies in $[\log2\cosh h,\log2\cosh h+\beta^2/4]$. Let $f_N=(km/N)^d$ be the fraction of vertices of $\Lambda^d_N$ that lie in translates of $\Lambda^d_m$. We have shown that
\[
b_{d,m}-(1-f_N)\frac{\beta^2}4\le b_{d,N}\le b_{d,m}+(1-f_N)\frac{\beta^2}4+\frac{\beta^2}{4m}.
\]
Since $f_N=(1-r/N)^d\to1$ as $N\to\infty$, with $d$ and $m$ fixed, we obtain
\begin{equation}\label{eq:geo-block}
b_{d,m}\le\liminf_{N\to\infty}b_{d,N}\le\limsup_{N\to\infty}b_{d,N}\le b_{d,m}+\frac{\beta^2}{4m}.
\end{equation}
Taking the limit inferior in $m$ on the right side of \eqref{eq:geo-block}, we obtain $\limsup_Nb_{d,N}\le\liminf_mb_{d,m}$, so the limit $p^\infty_d$ exists. By \eqref{eq:geo-block}, it satisfies $b_{d,m}\le p^\infty_d\le b_{d,m}+\beta^2/(4m)$ for every $m\ge1$. For $m=L$, combining this with \eqref{eq:geo-torus-box} shows that $p_{d,L}$ and $p^\infty_d$ both lie in the interval $[b_{d,L},b_{d,L}+\beta^2/(4L)]$, which completes the proof.
\end{proof}
\begin{remark}\label{rem:geo-side}
\Cref{lem:geo-side} is the only place where the geometry of the torus enters the passage from fixed to arbitrary side lengths. It shows that the limits $N\to\infty$ and $L\to\infty$ of the box and torus pressures in a fixed dimension coincide. Combined with \cref{thm:main}, it gives $\lim_{d\to\infty}p^\infty_d(\beta,h)=\PSK(\beta,h)$, so the convergence also holds when the thermodynamic limit is taken first.
\end{remark}
\section{Proofs of the main results}\label{sec:geo-main}
\begin{proof}[Proof of \cref{thm:main}]
Fix $\beta\ge0$ and $h\in\R$. Since $\T^d_L\in\cG_{2d}$ for every $L\ge3$, \cref{prop:upper} implies that
\[
\limsup_{d\to\infty}\sup_{L\ge3}p_{d,L}(\beta,h)\le\limsup_{D\to\infty}\sup_{G\in\cG_D}p_G(\beta,h)\le\PSK(\beta,h).
\]
For the lower bound, fix an integer $m\ge3$. For $L\ge m$, applying \cref{lem:geo-side} at the side lengths $L$ and $m$, we obtain
\[
p_{d,L}\ge p^\infty_d-\frac{\beta^2}{4L}\ge p_{d,m}-\frac{\beta^2}{4m}-\frac{\beta^2}{4L}\ge p_{d,m}-\frac{\beta^2}{2m}.
\]
Hence $\inf_{L\ge3}p_{d,L}\ge\min_{3\le L\le m}p_{d,L}-\beta^2/(2m)$. Since the minimum is over finitely many side lengths, \cref{prop:geo-fixed} implies that
\[
\liminf_{d\to\infty}\inf_{L\ge3}p_{d,L}(\beta,h)\ge\PSK(\beta,h)-\frac{\beta^2}{2m}.
\]
Letting $m\to\infty$ and combining with the upper bound, we obtain \cref{thm:main}.
\end{proof}
\begin{proof}[Proof of \cref{thm:hypercube}]
Since $Q_d\in\cG_d$, we have $\limsup_dp_{Q_d}(\beta,h)\le\PSK(\beta,h)$ by \cref{prop:upper}, and \cref{prop:geo-fixed} gives the matching lower bound.
\end{proof}
\begin{proof}[Proof of \cref{cor:ground}]
Let $G\in\cG_D$, $h\in\R$, and $\beta>0$. By the bound after \eqref{eq:pre-SK-sandwich}, $e(G;h)$ lies in the interval $[\beta^{-1}p_G(\beta,\beta h)-\beta^{-1}\log2,\ \beta^{-1}p_G(\beta,\beta h)]$, and by \eqref{eq:pre-PSK-ge-Pstar}, $\Pstar(h)$ lies in $[\beta^{-1}\PSK(\beta,\beta h)-\beta^{-1}\log2,\ \beta^{-1}\PSK(\beta,\beta h)]$. Comparing the two intervals, we obtain
\[
|e(G;h)-\Pstar(h)|\le\frac{|p_G(\beta,\beta h)-\PSK(\beta,\beta h)|}{\beta}+\frac{\log2}\beta.
\]
Taking the supremum over $L\ge3$ for $G=\T^d_L$, letting $d\to\infty$, and using \cref{thm:main} at $(\beta,\beta h)$, we obtain $\limsup_d\sup_{L\ge3}|e(\T^d_L;h)-\Pstar(h)|\le(\log2)/\beta$. Letting $\beta\to\infty$ proves the claim for tori, and the same argument with \cref{thm:hypercube} proves it for hypercubes.
\end{proof}
Finally, we justify \cref{rem:uniform}. By \cref{lem:pre-basic}(d), for each $B\ge0$ the functions $p_G$, $G\in\bigcup_D\cG_D$, and $p_N^{\mathrm{SK}}$, $N\ge1$, satisfy $|f(\beta,h)-f(\beta',h')|\le B|\beta-\beta'|/2+|h-h'|$ for $\beta,\beta'\in[0,B]$ and $h,h'\in\R$, and so does their pointwise limit $\PSK$. Let $K\subset[0,B]\times\R$ be compact and $\eta>0$, and let $K'$ be a finite subset of $K$ such that every point of $K$ lies within distance $\eta$ of $K'$ in each coordinate. Then
\[
\sup_{(\beta,h)\in K}\sup_{L\ge3}|p_{d,L}(\beta,h)-\PSK(\beta,h)|\le\max_{(\beta,h)\in K'}\sup_{L\ge3}|p_{d,L}(\beta,h)-\PSK(\beta,h)|+(B+2)\eta,
\]
and the maximum tends to zero as $d\to\infty$ by \cref{thm:main}. Since $\eta$ is arbitrary, the convergence in \cref{thm:main} is uniform on $K$. The same argument applies to hypercubes, and to the ground-state energy, which is $1$-Lipschitz in $h$ by its definition. Its SK limit $\Pstar(h)$ is also $1$-Lipschitz, as the pointwise limit of the $1$-Lipschitz functions $m_N(h)$.
We also prove the limit in \cref{rem:intro-bipartite}. A similar interpolation, with independent SK models on the two parts, is in~\cite[Section~3.1]{BarraGenoveseGuerra2011}. Let $D\ge1$ and $n=2D$, let $G=K_{D,D}$ have parts $V_1$ and $V_2$ of size $D$, and let $v\in\{-1,1\}^V$ be equal to $1$ on $V_1$ and to $-1$ on $V_2$. For $\sigma^1,\sigma^2\in\{-1,1\}^V$ let $u=\sigma^1\odot\sigma^2$. Since $\sum_{x\in V_1}u_x$ and $\sum_{y\in V_2}u_y$ are $(\langle u,\mathbf 1\rangle\pm\langle u,v\rangle)/2$, and every vertex of $V_1$ is adjacent to every vertex of $V_2$, we have
\[
\E H_G(\sigma^1)H_G(\sigma^2)=\frac1D\sum_{x\in V_1}\sum_{y\in V_2}u_xu_y=\frac1{2n}\big(\langle u,\mathbf 1\rangle^2-\langle u,v\rangle^2\big).
\]
We identify $V$ with $\{1,\dots,n\}$ and let $H_n^{\mathrm{SK}}$ be an SK Hamiltonian on $\{-1,1\}^V$, independent of $(g_e)_{e\in E}$. Then $\E H_n^{\mathrm{SK}}(\sigma^1)H_n^{\mathrm{SK}}(\sigma^2)=(\langle u,\mathbf 1\rangle^2-n)/(2n)$, and the difference of the two covariances is
\[
C(\sigma^1,\sigma^2)=\E H_n^{\mathrm{SK}}(\sigma^1)H_n^{\mathrm{SK}}(\sigma^2)-\E H_G(\sigma^1)H_G(\sigma^2)=\frac{\langle u,v\rangle^2-n}{2n}.
\]
In particular, $C(\sigma,\sigma)=-1/2$ for every $\sigma$, since $\langle\mathbf 1,v\rangle=0$. For $\theta\in[0,1]$, let $\langle\cdot\rangle_\theta$ be the Gibbs average for the exponent $\beta(\sqrt{1-\theta}\,H_G(\sigma)+\sqrt\theta\,H_n^{\mathrm{SK}}(\sigma))+h\sum_x\sigma_x$, and let $\varphi(\theta)$ be $n^{-1}$ times the expected logarithm of the corresponding partition function. Then $\varphi(0)=p_G(\beta,h)$ and $\varphi(1)=p_n^{\mathrm{SK}}(\beta,h)$. For $\theta\in(0,1)$, Gaussian integration by parts~\ref{F:ibp} gives, as in~\cite{GuerraToninelli2002Thermodynamic},
\[
\varphi'(\theta)=\frac{\beta^2}{2n}\E\big\langle C(\sigma^1,\sigma^1)-C(\sigma^1,\sigma^2)\big\rangle_\theta=-\frac{\beta^2}{4n^2}\E\big\langle\langle\sigma^1\odot\sigma^2,v\rangle^2\big\rangle_\theta\le0,
\]
where $\sigma^1$ and $\sigma^2$ are independent samples from the Gibbs measure. Since $\varphi$ is continuous on $[0,1]$, we obtain $p_{K_{D,D}}(\beta,h)\ge p_{2D}^{\mathrm{SK}}(\beta,h)$, and the right side tends to $\PSK(\beta,h)$ as $D\to\infty$. Since $K_{D,D}\in\cG_D$, \cref{prop:upper} gives the matching upper bound.
\backmatter
\setlength{\bibsep}{3pt plus 1pt minus 1pt}
\bibliographystyle{plainnat}
\bibliography{refs-hexagon}
\end{document}