<?xml version="1.0" encoding="utf-8"?><feed xmlns="http://www.w3.org/2005/Atom" ><generator uri="https://jekyllrb.com/" version="3.10.0">Jekyll</generator><link href="/feed.xml" rel="self" type="application/atom+xml" /><link href="/" rel="alternate" type="text/html" /><updated>2026-07-25T13:49:52+00:00</updated><id>/feed.xml</id><title type="html">Ludwig Winkler</title><subtitle>Jekyll version of the Massively theme by HTML5UP</subtitle><entry><title type="html">Korea and Switzerland</title><link href="/blog/KoreaSwitzerland26/" rel="alternate" type="text/html" title="Korea and Switzerland" /><published>2026-07-25T00:00:00+00:00</published><updated>2026-07-25T00:00:00+00:00</updated><id>/blog/KoreaSwitzerland26</id><content type="html" xml:base="/blog/KoreaSwitzerland26/"><![CDATA[<!-- ## Berlin Over The Years -->

<style>
    .image-gallery {
        overflow: auto;
        margin-left: -1% !important;
    }

    .image-gallery li {
        float: left;
        float: top;
        display: block;
        margin: 0 0 1% 1%;
        width: 99%;
    }

    .image-gallery li a {
        text-align: top;
        text-decoration: none !important;
        color: #777;
    }

    .image-gallery li a span {
        display: block;
        text-overflow: ellipsis;
        overflow: hidden;
        white-space: nowrap;
        padding: 3px 0;
    }

    .image-gallery li a img {
        width: 100%;
        height: 100%;
        display: flex;
        vertical-align: top;
    }
</style>

<ul class="image-gallery">
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    <li><a href="/photo_gallery/korea_switzerland26/01.jpg"><img src="/photo_gallery/korea_switzerland26/01.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/korea_switzerland26/03.jpg"><img src="/photo_gallery/korea_switzerland26/03.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/korea_switzerland26/04.jpg"><img src="/photo_gallery/korea_switzerland26/04.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/korea_switzerland26/05.jpg"><img src="/photo_gallery/korea_switzerland26/05.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/korea_switzerland26/06.jpg"><img src="/photo_gallery/korea_switzerland26/06.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/korea_switzerland26/10.jpg"><img src="/photo_gallery/korea_switzerland26/10.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/korea_switzerland26/11.jpg"><img src="/photo_gallery/korea_switzerland26/11.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/korea_switzerland26/12.jpg"><img src="/photo_gallery/korea_switzerland26/12.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/korea_switzerland26/13.jpg"><img src="/photo_gallery/korea_switzerland26/13.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/korea_switzerland26/14.jpg"><img src="/photo_gallery/korea_switzerland26/14.jpg" /></a></li>
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
</ul>

<!--  -->

<!--  -->
<p><!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 --></p>]]></content><author><name></name></author><category term="photography" /><summary type="html"><![CDATA[Of Peninsulas and Mountains]]></summary><media:thumbnail xmlns:media="http://search.yahoo.com/mrss/" url="/photo_gallery/korea_switzerland26/06.jpg" /><media:content medium="image" url="/photo_gallery/korea_switzerland26/06.jpg" xmlns:media="http://search.yahoo.com/mrss/" /></entry><entry><title type="html">Monte Carlo (git) Tree Search</title><link href="/blog/MCgitS/" rel="alternate" type="text/html" title="Monte Carlo (git) Tree Search" /><published>2026-05-10T00:00:00+00:00</published><updated>2026-05-10T00:00:00+00:00</updated><id>/blog/MCgitS</id><content type="html" xml:base="/blog/MCgitS/"><![CDATA[<script>
MathJax = {
  tex: {
    inlineMath: [['$','$'], ['\\(','\\)']],
    displayMath: [['$$','$$'], ['\\[','\\]']],
    processEscapes: true,
    tags: 'all'
  },
  options: {
    skipHtmlTags: ['script', 'noscript', 'style', 'textarea', 'pre']
  }
};
</script>

<script src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-mml-chtml.js" async=""></script>

<style>
.demo-card {
  background: #ffffff;
  border: 1px solid rgba(0,0,0,0.12);
  border-radius: 10px;
  padding: 20px;
  margin: 2em 0;
  color: #1a1a1a;
  font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', sans-serif;
}
.demo-card h4 {
  margin: 0 0 12px;
  color: #1e6fc4;
  font-size: 0.85em;
  letter-spacing: 0.06em;
  text-transform: uppercase;
}
.demo-controls {
  display: flex;
  gap: 10px;
  flex-wrap: wrap;
  margin-bottom: 12px;
  align-items: center;
}
.demo-btn {
  background: rgba(100,180,255,0.12);
  border: 1px solid rgba(100,180,255,0.6);
  box-shadow: none;
  color: #1e6fc4 !important;
  padding: 6px 18px;
  height: auto;
  line-height: 1.5;
  border-radius: 5px;
  cursor: pointer;
  font-size: 0.83em;
  font-weight: 600;
  letter-spacing: 0.02em;
  text-transform: none;
  transition: background 0.15s, color 0.15s;
}
.demo-btn:hover {
  background: rgba(100,180,255,0.24);
  color: #0a4a8c !important;
}
.demo-status {
  font-size: 0.78em;
  color: rgba(0,0,0,0.65);
  font-family: monospace;
  padding: 5px 10px;
  background: rgba(0,0,0,0.05);
  border-radius: 4px;
  min-height: 1.4em;
  margin-bottom: 8px;
}
.demo-stats {
  display: flex;
  gap: 20px;
  margin-top: 8px;
  font-size: 0.8em;
  color: rgba(0,0,0,0.65);
}
.demo-stats b { color: #b8860b; }
canvas.demo-canvas {
  display: block;
  width: 100%;
  border-radius: 6px;
  background: #ffffff;
}
.mc-readout {
  display: flex;
  gap: 18px;
  margin-top: 8px;
}
.mc-readout-left {
  flex: 1 1 auto;
  min-width: 0;
  display: flex;
  flex-direction: column;
  justify-content: space-between;
}
.mc-readout-left .demo-stats,
.mc-readout-left .legend-row { margin-top: 0; }
.mc-key {
  flex: 0 0 auto;
  align-self: flex-start;
  pointer-events: none;
}
.legend-row {
  display: flex;
  gap: 14px;
  flex-wrap: wrap;
  align-items: center;
  margin-top: 10px;
  font-size: 0.75em;
  color: rgba(0,0,0,0.65);
}
.ldot {
  display: inline-block;
  width: 9px; height: 9px;
  border-radius: 50%;
  margin-right: 3px;
  vertical-align: middle;
}
/* PUCT explorer */
.puct-grid {
  display: grid;
  grid-template-columns: 1fr 1fr;
  gap: 22px;
}
@media (max-width: 600px) { .puct-grid { grid-template-columns: 1fr; } }
.slider-grp { margin-bottom: 11px; }
.slider-grp label {
  display: flex;
  justify-content: space-between;
  font-size: 0.8em;
  color: rgba(0,0,0,0.7);
  margin-bottom: 3px;
}
.slider-grp label span { color: #b8860b; font-family: monospace; }
.slider-grp input[type=range] { width: 100%; accent-color: #1e6fc4; }
.puct-formula {
  font-size: 0.82em;
  background: rgba(0,0,0,0.04);
  border-left: 3px solid #1e6fc4;
  border-radius: 0 6px 6px 0;
  padding: 12px 14px;
  margin-bottom: 12px;
  line-height: 2;
  font-family: monospace;
  color: #1a1a1a;
}
.puct-score-big {
  text-align: center;
  font-size: 1.5em;
  font-family: monospace;
  color: #2a7d2a;
  margin-top: 6px;
}
/* Callout */
.callout {
  background: rgba(100,180,255,0.08);
  border-left: 4px solid #1e6fc4;
  border-radius: 0 8px 8px 0;
  padding: 14px 18px;
  margin: 1.8em 0;
  color: #1a2238;
  font-size: 0.95em;
  line-height: 1.7;
}
</style>

<p>There has been a tremendous interest in doing autoresearch with LLM agents.
The core idea is that since LLM’s have intelligence (the amount for real world problem solving is still being debated), have infinite stamina and have access to the abundance of information on the web including all recent research, then they could automate parts of the research pipeline that we all do.
Evidently, the writing was on the wall and there is a whole swath of start ups that try to capitalize on this idea with the idea of an AI scientist (Periodic Labs, FutureHouse etc).</p>

<p>LLM’s rely on the iterative filling of their context (to have context at all) and you can add a RAG system, but then you’re still dependent on what’s in your context.
So whatever a LLM does to solve a problem is highly influenced by what’s in it’s context.
Of course there is a lot of baseline intelligence in LLM’s, but the context is still a huge factor in what it can do.</p>

<p>But research is not a linear process in the same way that we linearly and monotonically fill up a context.
Research consist of many rabbit holes, dead ends and resets.
This resetting is something that LLM’s are not ideally placed to do due to the their dependence on the context when trying to find solutions to problems by trial and error.</p>

<p>That begs the question: can we wrap a harness around a LLM to make it more like a human researcher that’s able to deal with setbacks, rabbit hole failures and resetting the context when needed?</p>

<h3 id="graduate-student-descent">Graduate student descent</h3>

<p>There’s an old joke in machine learning that the real optimization algorithm behind most papers isn’t Adam or SGD.
It’s <em>graduate student descent</em>: a PhD student sits in front of a screen, semi-informed, and tweaks things. Bump the learning rate. 
Swap out a loss term. Try a slightly different architecture. Rerun. Squint at the validation number. Try again.</p>

<p>Most of this is busywork — hyperparameter fiddling, “what if I change this one thing,” waiting for the run to finish. Only a handful of the tries actually matter, and the student usually can’t tell in advance which ones, which is exactly why they have to keep trying.</p>

<p>Having done a PhD in machine learing I’m all too familiar with this process.
I have spent countless hours doing this exact loop, it’s unfortunately necessary to get a good result, but it’s also mind numbing and not very fun when you haven’t found the magic combination yet.
I doubt we’ll be able to do away with graduate student descent as machine learning is intrinsically a latent model problem where try to explain/model/predict data and generalize from it.
If you’ve done machine learning research you’ll know th Bermuda triangle of hope between model architecture, parameter optimization and data quirks.</p>

<p>Your graduate student of choice would</p>
<ul>
  <li><strong>proposes</strong> a change, based on a hunch about what might help,</li>
  <li><strong>evaluates</strong> it by running a training job and reading off a number,</li>
  <li>and then <strong>decides what to try next</strong> — usually by leaning into whatever’s been working, but every so often taking a flyer on a direction they’ve barely touched.</li>
</ul>

<p>It’s a semi-informed search over the tree of possible experiments, steered by intuition and metrics.
Intuition is hard to come by if you haven’t done a lot of research before, but LLM agents offer boundless stamina and will never stop trying stuff out.
This has been epitomized in the Ralph loop where a semi-intelligent actor just hammers the problem over and over again.
Or it’s your new grad student is still optimistic about the next experiment.</p>

<h3 id="branching-and-resetting">Branching and Resetting</h3>

<p>The other modus operandi we usually face in research is that we often have to reset our context and start over.
That’s the humiliating act of realizing that your ideas have failed and we need to figuratively and literally clean the white board.</p>

<p>Here’s my biggest gripe with using LLM’s for autoresearch: They’re hard at resetting their context and starting over in a principled way.
The original idea of autoresearch by Karpathy was to use git to reset to the previous state if something failed but it still left the context unchanged.
So fundamentally, while the code was reset the context wasn’t.
It essentially set the autoresearch agent on a linear trajectory of git states with no way of completely changing track and pursuing a different line of research.</p>

<p>But git allows branching such that we could try different ideas from the same root node.
We can try experiments and log the code state in the git tree and then branch off to try a different idea.
Importantly, this allows us to keep a ledge or diary of sorts of the many experiments which can develop in a multitude of different ways.
If we want to revisit a prior experiment, we simply have to checkout the git commit and we have the exact code state that was used to run that experiment.
That allows us to jump around in the tree and pursue different lines of research.</p>

<h3 id="monte-carlo-tree-search">Monte Carlo Tree Search</h3>

<p>So up to now we’ve concluded that LLM are useful testing out ideas with unwavering enthusiasm and that git is a perfect way to log the experiments and keep track of the many different lines of research.
The last missing piece of the puzzle is to combine the two in a principled way such that we can explore the tree of experiments in a way that balances exploration and exploitation.</p>

<p>A long while ago in 2017, I gave a talk on AlphaGo and the Monte Carlo Tree Search algorithm that they used as a fast rollout policy to explore the state space tree of a game of Go.
Mind you, Go is a game with a huge state space and the number of possible moves is astronomical.
So they used a combination of a neural network to evaluate the board state and a Monte Carlo Tree Search to explore the tree of possible moves.</p>

<p>In the Monte Carlo Tree Search algorithm we work fundamentally on a tree structure where each node is a state and each edge is a possible action that can be taken from that state.</p>

<p>MCTS is a way to explore the tree of possible states in a principled way that balances exploration and exploitation.
In the case of AlphaGo we simply counted the number of times a node was visited and the number of wins that were achieved from that node.</p>

<p>As the saying goes “An image is worth a thousand words” so let’s look at the MCTS algorithm in a visual way:</p>

<style>
.mcts-flip {
  margin: 1.6em auto;
  max-width: 100%;
  border: 1px solid rgba(0,0,0,0.12);
  border-radius: 10px;
  padding: 14px 14px 12px;
  background: #ffffff;
  font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', sans-serif;
}
.mcts-flip-stage {
  position: relative;
  background: #ffffff;
  border-radius: 6px;
  overflow: hidden;
  text-align: center;
  min-height: 200px;
  display: flex;
  align-items: center;
  justify-content: center;
}
.mcts-flip-stage img {
  max-width: 100%;
  height: auto;
  display: block;
  margin: 0 auto;
}
.mcts-flip-stage img:not(.is-active) { display: none; }
.mcts-flip-controls {
  display: flex;
  align-items: center;
  justify-content: space-between;
  gap: 12px;
  margin-top: 10px;
}
.mcts-flip-btn {
  background: rgba(100,180,255,0.12);
  border: 1px solid rgba(100,180,255,0.6);
  color: #1e6fc4;
  padding: 5px 14px;
  border-radius: 5px;
  cursor: pointer;
  font-size: 0.83em;
  font-weight: 600;
  font-family: inherit;
  transition: background 0.15s, color 0.15s;
}
.mcts-flip-btn:hover { background: rgba(100,180,255,0.24); color: #0a4a8c; }
.mcts-flip-btn:disabled { opacity: 0.35; cursor: default; }
.mcts-flip-label {
  flex: 1 1 auto;
  text-align: center;
  font-size: 0.85em;
  color: #1a1a1a;
  letter-spacing: 0.02em;
}
.mcts-flip-label b { color: #b8860b; font-family: monospace; margin-right: 6px; }
.mcts-flip-dots {
  display: flex;
  justify-content: center;
  gap: 7px;
  margin-top: 9px;
}
.mcts-flip-dot {
  width: 8px; height: 8px;
  border-radius: 50%;
  background: rgba(0,0,0,0.2);
  border: none;
  padding: 0;
  cursor: pointer;
  transition: background 0.15s;
}
.mcts-flip-dot:hover { background: rgba(0,0,0,0.4); }
.mcts-flip-dot.is-active { background: #1e6fc4; }
</style>

<div class="mcts-flip" id="mcts-flip" tabindex="0" aria-label="MCTS step-by-step flipchart">
  <div class="mcts-flip-stage">
    <img src="/blog/MCgitS/MCTS_GameTree.png" alt="Game tree" data-step="Game tree" class="is-active" />
    <img src="/blog/MCgitS/MCTS_Selection.png" alt="Selection" data-step="Selection" />
    <img src="/blog/MCgitS/MCTS_Expansion.png" alt="Expansion" data-step="Expansion" />
    <img src="/blog/MCgitS/MCTS_Simulation.png" alt="Simulation" data-step="Simulation" />
    <img src="/blog/MCgitS/MCTS_Backpropagation.png" alt="Backpropagation" data-step="Backpropagation" />
    <img src="/blog/MCgitS/MCTS_Action.png" alt="Action" data-step="Action" />
  </div>
  <div class="mcts-flip-controls">
    <button class="mcts-flip-btn" id="mcts-flip-prev" aria-label="Previous">← Prev</button>
    <div class="mcts-flip-label"><b id="mcts-flip-count">1 / 6</b><span id="mcts-flip-name">Game tree</span></div>
    <button class="mcts-flip-btn" id="mcts-flip-next" aria-label="Next">Next →</button>
  </div>
  <div class="mcts-flip-dots" id="mcts-flip-dots"></div>
</div>

<script>
(function() {
  var root  = document.getElementById('mcts-flip');
  if (!root) return;
  var imgs  = root.querySelectorAll('.mcts-flip-stage img');
  var prev  = document.getElementById('mcts-flip-prev');
  var next  = document.getElementById('mcts-flip-next');
  var count = document.getElementById('mcts-flip-count');
  var name  = document.getElementById('mcts-flip-name');
  var dots  = document.getElementById('mcts-flip-dots');
  var i = 0, n = imgs.length;

  var dotEls = [];
  for (var k = 0; k < n; k++) {
    (function(idx) {
      var d = document.createElement('button');
      d.className = 'mcts-flip-dot';
      d.setAttribute('aria-label', 'Step ' + (idx + 1));
      d.addEventListener('click', function() { go(idx); });
      dots.appendChild(d);
      dotEls.push(d);
    })(k);
  }

  function go(j) {
    i = (j + n) % n;
    for (var k = 0; k < n; k++) {
      imgs[k].classList.toggle('is-active', k === i);
      dotEls[k].classList.toggle('is-active', k === i);
    }
    count.textContent = (i + 1) + ' / ' + n;
    name.textContent  = imgs[i].getAttribute('data-step') || '';
    prev.disabled = (i === 0);
    next.disabled = (i === n - 1);
  }

  prev.addEventListener('click', function() { go(i - 1); });
  next.addEventListener('click', function() { go(i + 1); });
  root.addEventListener('keydown', function(e) {
    if (e.key === 'ArrowLeft')  { e.preventDefault(); go(i - 1); }
    if (e.key === 'ArrowRight') { e.preventDefault(); go(i + 1); }
  });

  go(0);
})();
</script>

<p>Initially we start out with a state tree with all the moves with their wins and visits.</p>

<ul style="list-style: none; padding-left: 0; margin-left: 0;">
  <li><strong>Select</strong>: We use a formula to compare all non-terminal child nodes in the tree and decide to which child node to go first. In this simple case we simply compare the number of wins in the subtree to all visits of the subtree. </li>
  <li><strong>Expand</strong>: Reaching a terminal leaf node, we chose an action on how to change the state. In Go we would place a stone on the game board.</li>
  <li><strong>Simulate</strong>: We then simulate the game from the new state to see the outcome. This could be a cheap and fast rollout policy until we win or loose.</li>
  <li><strong>Backpropagate</strong>: We log the win (or loss) and backpropagate the ratio up the tree, automatically increasing the number of wins of any node between the leaf node and the root. This increases the percentage of wins all of those recursive subtrees.</li>
  <li><strong>Action</strong>: With one win or loss recorded, we can choose the best action based on the updated statistics. We use a policy that does a bit of exploration and chooses the middle subtree.</li>
</ul>

<p>Below is an animation of a MCTS search for a simple analytical case to visualize the search character.
The reward is a simple function of a binary action whether to add a 0 or a 1 to a string of length 5. 
The reward is the number of 1’s in the string. The search starts at the root node and explores the tree of possible strings until it finds the optimal string of all 1’s.
After having found the gren and dark green combinations the algorithms keeps on exploring the lesser valuable paths to make sure that it doesn’t miss a better solution.</p>

<p><img src="/blog/MCgitS/k5_tree.gif" alt="MCTS tree visualization" width="100%" style="max-width: 800px;" /></p>

<h3 id="translating-mcts-to-autoresearch">Translating MCTS to Autoresearch</h3>

<p>MCTS gives us a principled way to explore a tree of possible states all originating from a common root node and searching the tree for the best state.</p>

<p>In order to get the performance of a particular node, we define a validation script that isn’t allowed to change and returns a single metric that gauges the performance of the state. This metric is then used to compare the performance of different nodes in the tree.</p>

<p>MCTS was originally formulated for games which have a binary outcome: win or loss.
In machine learning research we often have a continuous outcome (loss) and we want to minimize that loss.
The simple switch is to set an upper and lower (preferabbly zero but negative loglikelihoods can go lower) bound and then just rescaling the validated performance in those bounds to a [0,1] range.</p>

<p>The selection is a formula that compares the performance of all child nodes and chooses the best one to explore further.
A particular node can have children which are already validated or (pre-generated) proposals which still require an expansion.</p>

<p>An expansion is the code change that we did that changes the code base of a particular node in the tree.
This could be a change in the model architecture, a change in the loss function, a change in the optimizer or a change in the data set.</p>

<p>After each expansion we run the validation script to see how good our change performed.
This is the simulation step in MCTS.
Of course our proposal can be so bad that we run immediatley into NaN’s or other problems.
In that case we simply ignore the proposal and mark it as failed.
Failed nodes have no repercussion and are simply skipped in the selection step and represent a dead end.</p>

<p>Once the simulation/validation ran, we backpropagate the [0,1] value up the path between the leaf node and the root node, updating the performance of all nodes in between.</p>

<p>Here’s an interactive example with random best values. You can play it a couple of times ot see how</p>

<div class="demo-card">
<h4>Live MCTS tree simulation</h4>
<div class="demo-controls">
  <button class="demo-btn" id="mc-step">Step →</button>
  <button class="demo-btn" id="mc-auto">Auto</button>
  <button class="demo-btn" id="mc-reset">Reset</button>
  <span style="font-size:0.78em;color:rgba(0,0,0,0.5)">30-iteration budget</span>
</div>
<div class="demo-status" id="mc-status">Ready — press Step to begin.</div>
<canvas class="demo-canvas" id="mc-canvas" width="700" height="160"></canvas>
<div class="mc-readout">
<div class="mc-readout-left">
<div class="demo-stats">
  <span>Iter <b id="mc-iter">0</b> / 30</span>
  <span>Best loss <b id="mc-best">0.0740</b></span>
  <span>Pending proposals <b id="mc-open">4</b></span>
</div>
<div class="legend-row">
  <span><span class="ldot" style="background:hsl(120,65%,42%)"></span>low loss (good)</span>
  <span><span class="ldot" style="background:hsl(55,70%,47%)"></span>medium</span>
  <span><span class="ldot" style="background:hsl(0,65%,45%)"></span>high loss</span>
  <span><span class="ldot" style="background:#2a2a3a;border:1px solid #555"></span>failed verify</span>
  <span><span class="ldot" style="background:#e8a030"></span>queued proposal</span>
  <span><span class="ldot" style="background:#64b4ff"></span>selected path</span>
  <span><span class="ldot" style="background:transparent;border:2px solid #ffd700"></span>newly expanded</span>
</div>
</div>
<svg class="mc-key" width="178" height="88" viewBox="0 0 178 88" aria-label="What each node shows">
  <rect x="0.5" y="0.5" width="177" height="87" rx="8" fill="#ffffff" stroke="#000000" stroke-opacity="0.2" />
  <!-- sample node -->
  <circle cx="30" cy="44" r="14" fill="#97b927" stroke="#000000" stroke-opacity="0.45" stroke-width="2" />
  <text x="30" y="44" fill="#ffffff" font-family="monospace" font-size="8" text-anchor="middle" dominant-baseline="central">0.06</text>
  <text x="38.5" y="35" fill="#1e5ab4" font-family="monospace" font-size="8" font-weight="bold">N9</text>
  <circle cx="19" cy="69" r="4" fill="#e8a030" stroke="#b8730e" />
  <circle cx="30" cy="69" r="4" fill="#e8a030" stroke="#b8730e" />
  <circle cx="41" cy="69" r="4" fill="#e8a030" stroke="#b8730e" />
  <!-- leaders + labels -->
  <g stroke="#000000" stroke-opacity="0.35" stroke-width="1">
    <line x1="40" y1="34" x2="58" y2="18" />
    <line x1="44" y1="44" x2="58" y2="44" />
    <line x1="30" y1="73" x2="58" y2="72" />
  </g>
  <g fill="#000000" fill-opacity="0.75" font-family="-apple-system,BlinkMacSystemFont,'Segoe UI',sans-serif" font-size="9" dominant-baseline="central">
    <text x="61" y="18">subtree size</text>
    <text x="61" y="44">loss</text>
    <text x="61" y="72">open proposals</text>
  </g>
</svg>
</div>
</div>

<p>Notice how the search keeps coming back to the best branch — but never <em>only</em> that branch. Every so often it wanders off to try something it has mostly ignored. That tension, between milking what already works and checking on what you’ve neglected, is the whole game. Hold onto it; in a moment we’ll see the exact formula that produces it.</p>

<hr />

<h3 id="keeping-things-simple">Keeping things simple</h3>

<p>The idea is all nice but the devil’s in the detail as we all know.</p>

<p>For the implementation I opted for a some very hard boundaries.</p>
<ul style="list-style: none; padding-left: 0; margin-left: 0;">
  <li><strong>Simple</strong>: The whole setup should consist of a single functional file and maybe an additional utils file</li>
  <li><strong>A Single Contract</strong>: There is no try/except (looking at you coding agents), defaults, if/else and catch mechanism. There is a single way to run this simple set up, if you don't it will fail.</li>
  <li><strong>YAGNI (Recycling git)</strong>: Git provides everything we need via inheritance, branching, checking out commits in a super convenient package. We will use as much of git as possible.</li>
  <li><strong>Extendable Base Case</strong>: With the three paradigms above in a single functional python script, it can easily be extended for more complex scenarios to your liking.</li>
</ul>

<p>Wait, how do we inject “intelligence” into the whole setup?
It’s simple: we use the Python SDK for Copilot/Claude.
With everything we’ve talked about so far, every step of the way is a simple prompt to the LLM with a very specific task and a very specific context.
Essentially, we have a very clear idea of what we want to do and we can prompt the LLM to do it for us.
The thing is that we can actually isolate the LLM to whatever it’s supposed to do and not let it wander off into the weeds of the code base.
Since everything runs on git, we can run git commands manually from within the research loop, and start a coding session with the LLM to implement a particular proposal.
After the part with the intelligence is done, we can switch back to normal programming and run the validation script from within our loop.
Since the agents get a fresh context and a clear cut, localized and fence off task to do, we manage everything else around them with a mini harness.</p>

<p>Basically, the MCTS algorithm, the validation script, the git plumbing, the proposals calls, the implementation calls are all simply python commands run from a single script.
The agents might screw up, but we have the harness around them to make sure that they don’t screw up the whole system and can simply pursue a different line of research through the MCTS formalism.</p>

<hr />

<h3 id="the-plumbing-under-the-hood">The Plumbing under the hood</h3>

<p>Once we’ve expanded the tree to a new node, it gets validated and we run a proposer loop that attaches five distinct proposals in text format to the node.
Each proposal has a promise score and a rationale for why it might work.</p>

<p>Then we run a selection algorithm on the node and its children to decide which child node to explore next.
The most basic form would be to run PUCTS (Predictor + Upper Confidence bound applied to Trees) which is a simple formula that balances exploration and exploitation:</p>

<div style="overflow-x: auto;">
\begin{align*}
\text{score}(s, a) = \underbrace{Q(s,a)}_{\text{exploitation}} + \underbrace{c \cdot \pi(a \mid s) \cdot \frac{\sqrt{N(s)}}{1 + N(s,a)}}_{\text{exploration bonus } U}
\end{align*}
</div>

<p>Here, $Q(s,a)$ is the average value of the child node $a$ from state $s$, $N(s)$ is the number of times state $s$ has been visited (includes the subtree attached to that node), and $N(s,a)$ is the number of times action $a$ has been taken from state $s$. The prior $\pi(a \mid s)$ is the promise score of the proposal that created this child node, and $c$ is a constant that controls how much we favor exploration over exploitation.</p>

<p>It looks like a mouthful, but it’s just those two instincts sitting side by side:</p>

<ul>
  <li>$Q(s,a)$ is <strong>how good this looks</strong> — the best score seen so far down this branch (and for a brand-new idea, we optimistically borrow the parent’s score, so it isn’t punished for being untried).</li>
  <li>The second term, $U$, is <strong>how under-explored it is</strong> — large when we’ve barely visited this branch and when our gut feeling about it, the prior $\pi$, was high. It shrinks a little every time we come back.</li>
  <li>$c = 0.5$ is just a knob for how adventurous we are overall.</li>
</ul>

<p>The lovely part is how the balance shifts on its own. Early on, $U$ dominates and the search fans out, sampling lots of directions. As a branch piles up visits, its $U$ collapses and its hard-won $Q$ takes over. Exploration quietly hands off to exploitation, with no schedule to tune by hand.</p>

<p>With a small probability $\varepsilon$ we can also flat-sample a node and try a proposal from there, which is a simple way to keep the search from getting stuck in a local optimum.
So essentially from time to time, we’ll look at all $N(s)$ and pick a node at random to explore, instead of always picking the best $Q(s,a) + U$.
That’s a little bit of jitter to make the search jump out of a local optimum and explore other branches which might be blocked due to a series of bad validations.</p>

<p><strong>What gets stapled to each commit.</strong> Earlier I waved vaguely at git “notes.” Concretely, each live commit carries a small JSON note recording its score, the proposal that created it, and the proposals still waiting to be tried from here:</p>

<div class="language-json highlighter-rouge"><div class="highlight"><pre class="highlight"><code><span class="p">{</span><span class="w">
  </span><span class="nl">"metrics"</span><span class="p">:</span><span class="w"> </span><span class="p">{</span><span class="nl">"loss"</span><span class="p">:</span><span class="w"> </span><span class="mf">0.0623</span><span class="p">},</span><span class="w">
  </span><span class="nl">"winner"</span><span class="p">:</span><span class="w">  </span><span class="p">{</span><span class="nl">"plan"</span><span class="p">:</span><span class="w"> </span><span class="s2">"CODE. Edit losses.py…"</span><span class="p">,</span><span class="w"> </span><span class="nl">"promise"</span><span class="p">:</span><span class="w"> </span><span class="mf">0.72</span><span class="p">,</span><span class="w"> </span><span class="nl">"rationale"</span><span class="p">:</span><span class="w"> </span><span class="s2">"…"</span><span class="p">},</span><span class="w">
  </span><span class="nl">"open"</span><span class="p">:</span><span class="w">    </span><span class="p">[{</span><span class="nl">"plan"</span><span class="p">:</span><span class="w"> </span><span class="s2">"HYPERPARAM…"</span><span class="p">,</span><span class="w"> </span><span class="nl">"promise"</span><span class="p">:</span><span class="w"> </span><span class="mf">0.45</span><span class="p">,</span><span class="w"> </span><span class="nl">"rationale"</span><span class="p">:</span><span class="w"> </span><span class="s2">"…"</span><span class="p">}],</span><span class="w">
  </span><span class="nl">"state"</span><span class="p">:</span><span class="w">   </span><span class="s2">"evaluated"</span><span class="w">
</span><span class="p">}</span><span class="w">
</span></code></pre></div></div>

<p>The nice thing about this is that the entire tree is self-contained in git. Git does all the tracking. The <code class="language-plaintext highlighter-rouge">metrics</code> are the result of the validation script.
I also opted for the script to parse in the entire tree into a nested tree data structure and do all the calculations in memory, then write the notes back to git at the end of each iteration. That way, if the machine dies mid-run, the tree is still consistent and we can pick up where we left off.
The requires us to parse the whole git tree once per iteration, but the tree is small enough that it doesn’t matter. Most of our time will be spent anyways on GPU compute.</p>

<p><code class="language-plaintext highlighter-rouge">winner</code> is the proposal that produced <em>this</em> commit (it lives on the child, not the parent); <code class="language-plaintext highlighter-rouge">open</code> is the queue of ideas not yet tried from here. The entire tree is rebuilt from these notes on every wake-up, so there’s no second copy of the state to fall out of sync. Each run namespaces its notes under an ID like <code class="language-plaintext highlighter-rouge">YYYYMMDD_HHMMSS_µs-happy-otter-abc12345</code>, so several searches can branch off the same starting commit without colliding.</p>

<p><strong>The verify contract.</strong> After each commit the orchestrator runs your evaluation command and reads back a JSON file that has to contain at least <code class="language-plaintext highlighter-rouge">{"loss": &lt;number&gt;}</code>. To stop the agents from “improving” the metric by quietly rewriting the scorer, the verify script is <em>hash-locked</em> on the first run: its checksum is pinned, and if it ever changes, the run aborts on the spot.</p>

<p><strong>The three states a node can be in.</strong> Not every experiment succeeds, and the tree has to cope. A node is one of:</p>

<p><strong><code class="language-plaintext highlighter-rouge">evaluated</code></strong> — verify succeeded and a loss was recorded. The search can keep descending from here.</p>

<p><strong><code class="language-plaintext highlighter-rouge">failed</code></strong> — verify errored or timed out. The search can <em>still</em> descend from here, since a fix might live one commit below, but the node is scored pessimistically so the search routes around it without giving up on the branch.</p>

<p><strong><code class="language-plaintext highlighter-rouge">terminal</code></strong> — a genuine dead end, either declared by the verifier or because every one of its children is itself terminal. The search will not descend here again.</p>

<p>That last rule is worth dwelling on, because <em>terminal</em> deadness bubbles upward. The moment a node has no ideas left in its queue and every one of its children has gone terminal, it turns terminal too — so the exhaustion of an entire subtree propagates quietly back toward the root, one dead node at a time, and the search never wastes another thought on it.</p>

<p><strong>One full iteration</strong>, end to end:</p>

<ol>
  <li><strong>Rebuild</strong> the tree from git notes (pure reads)</li>
  <li><strong>Backpropagate</strong> once — post-order walk filling <code class="language-plaintext highlighter-rouge">n</code>, <code class="language-plaintext highlighter-rouge">subtree_value</code>, cascading <code class="language-plaintext highlighter-rouge">terminal</code> states</li>
  <li>With prob. $\varepsilon$: <strong>flat-sample</strong> a node and append one proposal → go to 7<br />Otherwise: <strong>PUCT-descend</strong> to pick <code class="language-plaintext highlighter-rouge">(node, proposal_index)</code></li>
  <li><strong>Checkout</strong> that node’s commit</li>
  <li><strong>Implement</strong>: run the implementer agent, verify the diff, commit the child</li>
  <li><strong>Evaluate</strong>: run the verify script, record <code class="language-plaintext highlighter-rouge">loss</code> and <code class="language-plaintext highlighter-rouge">state</code></li>
  <li><strong>Seed</strong>: run the proposer at the new node to populate its <code class="language-plaintext highlighter-rouge">open</code> list</li>
  <li><strong>Save</strong>: write notes for child and parent</li>
</ol>

<p>The order is deliberately crash-proof: if the machine dies anywhere in the middle, the tree on disk is still consistent — either the new commit’s note exists and we carry on, or it doesn’t and the parent simply re-tries that idea next time. The entire footprint on disk is a couple of git refs, a checksum, and a log file. No database, no cache, nothing to corrupt.</p>

<p>Every time you start a new run, a random string is created with the root hash in it.
This serves as a marker that we’ll only consider any <code class="language-plaintext highlighter-rouge">git notes</code> that are marked under this ID. It also automatically denotes our root such that we can keep running multiple runs from the same root in parallel and from different machines without any collisions.
This also enables the script to run iteratively such that you can generate new runs on the fly based off previous autoresearch runs. If you know git, you don’t have to learn new concepts.</p>

<p><strong>The whole thing is a single Python file of around 1700 lines, standard library only. Drop it into a git repo, point it at your task and your evaluation command, and let it climb overnight.</strong></p>

<hr />

<p>The thing I keep coming back to is <em>what</em> got automated and what stubbornly didn’t. The busywork — the “let me just bump this and see,” the overnight grind of running and re-running and squinting at numbers — turns out to be genuinely mechanizable. A tree, a metric, and a formula will happily do all of it while you sleep. But the <em>taste</em> — knowing which direction is worth a hunch in the first place, recognizing that a boring-looking number is secretly the interesting result.</p>

<p>So we’ve automated graduate student descent. We have not, it seems, automated the graduate student.</p>

<p>You can find the code <a href="https://github.com/ludwigwinkler/MCgitS">here</a>.</p>

<hr />

<script>
(function() {
'use strict';

/* ── Constants ── */
var C = 0.5, LO = 0.0, HI = 0.1, TAU = Math.PI * 2;
var NODE_R = 16, LEVEL_H = 84, MARGIN = 34, BOTTOM = 30;
var PIP_R = 4.2, PIP_GAP = 11, PIP_DROP = 14;   /* proposal pips, fanned below a node */
var ANIM_MS = 520;
var TYPES = ['HYPERPARAM','CODE','NEW','BLEND'];
var nid = 0;

/* ── Node ── */
function mkNode(parent, loss, proposal, state) {
  return {
    id: nid++, parent: parent, children: [],
    loss: loss, proposal: proposal,
    openProposals: [],
    state: state || (loss !== null ? 'evaluated' : null),
    n: 0, sv: 0, px: 0, py: 0, lx: 0, depth: 0
  };
}

function l2v(loss) {
  return (HI - Math.max(LO, Math.min(HI, loss))) / (HI - LO);
}

function v2c(v) {
  var h = Math.round(v * 120);
  return 'hsl(' + h + ',65%,44%)';
}

function randProps(n) {
  var out = [];
  for (var i = 0; i < n; i++) {
    out.push({
      label: TYPES[Math.floor(Math.random() * TYPES.length)],
      promise: Math.round((0.15 + Math.random() * 0.80) * 100) / 100
    });
  }
  return out;
}

/* ── Backprop ── */
function walkBP(node, parentSV) {
  node.sv = parentSV;
  for (var i = 0; i < node.children.length; i++) walkBP(node.children[i], node.sv);
  var cnt = (node.loss !== null && node.state !== 'failed') ? 1 : 0;
  for (var j = 0; j < node.children.length; j++) cnt += node.children[j].n;
  node.n = cnt;
  var vals = [];
  if (node.loss !== null && node.state !== 'failed') vals.push(l2v(node.loss));
  for (var k = 0; k < node.children.length; k++) {
    var c = node.children[k];
    if (c.state !== 'failed' && c.n > 0) vals.push(c.sv);
  }
  if (vals.length > 0) {
    var mx = vals[0];
    for (var m = 1; m < vals.length; m++) if (vals[m] > mx) mx = vals[m];
    node.sv = mx;
  }
}

function backprop(root) { walkBP(root, 1.0); }

/* ── Select ── */
function selectFrom(node, path) {
  if (!node.openProposals.length && !node.children.length) return {node: node, pi: null, path: path};
  var live = [];
  for (var i = 0; i < node.children.length; i++) {
    if (node.children[i].state !== 'terminal') live.push(node.children[i]);
  }
  var totalP = 0;
  for (var j = 0; j < live.length; j++) totalP += live[j].proposal.promise;
  for (var k = 0; k < node.openProposals.length; k++) totalP += node.openProposals[k].promise;
  if (totalP <= 0) return {node: node, pi: null, path: path};

  var best = -Infinity, bestType = null, bestIdx = -1;
  for (var ci = 0; ci < live.length; ci++) {
    var ch = live[ci];
    var pr = ch.proposal.promise / totalP;
    var sc = ch.sv + C * pr * Math.sqrt(Math.max(node.n, 1)) / (1 + ch.n);
    if (sc > best) { best = sc; bestType = 'child'; bestIdx = ci; }
  }
  for (var oi = 0; oi < node.openProposals.length; oi++) {
    var op = node.openProposals[oi];
    var opr = op.promise / totalP;
    var osc = node.sv + C * opr * Math.sqrt(Math.max(node.n, 1)) / 1;
    if (osc > best) { best = osc; bestType = 'open'; bestIdx = oi; }
  }
  if (bestType === 'child') {
    var child = live[bestIdx];
    return selectFrom(child, path.concat([child]));
  }
  return {node: node, pi: bestIdx, path: path};
}

/* ── Layout ── */
function assignLeaves(node, counter) {
  if (node.children.length === 0) {
    node.lx = counter.v++;
  } else {
    for (var i = 0; i < node.children.length; i++) assignLeaves(node.children[i], counter);
    var sum = 0;
    for (var j = 0; j < node.children.length; j++) sum += node.children[j].lx;
    node.lx = sum / node.children.length;
  }
}

function applyPos(node, depth, sx, offX) {
  node.px = offX + node.lx * sx;
  node.py = MARGIN + depth * LEVEL_H;
  node.depth = depth;
  for (var i = 0; i < node.children.length; i++) applyPos(node.children[i], depth + 1, sx, offX);
}

function maxDepth(node, d) {
  var mx = d;
  for (var i = 0; i < node.children.length; i++) {
    var v = maxDepth(node.children[i], d + 1);
    if (v > mx) mx = v;
  }
  return mx;
}

function allNodes(node, out) {
  out.push(node);
  for (var i = 0; i < node.children.length; i++) allNodes(node.children[i], out);
  return out;
}

/* ── Small helpers ── */
function lerp(a, b, t) { return a + (b - a) * t; }
function easeOut(t) { return 1 - (1 - t) * (1 - t); }
function pipX(node, i, count) { return node.px - (count - 1) * PIP_GAP / 2 + i * PIP_GAP; }
function pipY(node) { return node.py + NODE_R + PIP_DROP; }

/* ── Render ── */
function render(canvas, root, selSet, lastNew, anim, now) {
  var ctr = {v: 0};
  assignLeaves(root, ctr);
  var nLeaves = Math.max(ctr.v, 1);

  var dpr = window.devicePixelRatio || 1;
  var cssW = Math.max(canvas.clientWidth || 700, 280);
  var avail = cssW - 2 * (MARGIN + NODE_R);
  var sx = Math.min(112, Math.max(34, nLeaves > 1 ? avail / (nLeaves - 1) : avail));
  var usedW = (nLeaves - 1) * sx;
  var offX = (cssW - usedW) / 2;
  applyPos(root, 0, sx, offX);

  var md = maxDepth(root, 0);
  var logicalH = MARGIN + md * LEVEL_H + NODE_R + BOTTOM;
  var logicalW = cssW;

  canvas.width = Math.round(logicalW * dpr);
  canvas.height = Math.round(logicalH * dpr);
  canvas.style.height = logicalH + 'px';

  var ctx = canvas.getContext('2d');
  ctx.setTransform(dpr, 0, 0, dpr, 0, 0);
  ctx.clearRect(0, 0, logicalW, logicalH);
  var bg = ctx.createLinearGradient(0, 0, 0, logicalH);
  bg.addColorStop(0, '#ffffff');
  bg.addColorStop(1, '#f4f6fa');
  ctx.fillStyle = bg;
  ctx.fillRect(0, 0, logicalW, logicalH);

  var nodes = allNodes(root, []);

  /* animation interpolation (a proposal pip flying down into the new child) */
  var ap = 1, ix = 0, iy = 0, ir = NODE_R;
  if (anim) {
    ap = easeOut(Math.max(0, Math.min(1, (now - anim.t0) / anim.dur)));
    ix = lerp(anim.parent.px, anim.child.px, ap);
    iy = lerp(pipY(anim.parent), anim.child.py, ap);
    ir = lerp(PIP_R, NODE_R, ap);
  }

  /* edges */
  ctx.lineCap = 'round';
  for (var ei = 0; ei < nodes.length; ei++) {
    var nd = nodes[ei];
    for (var ci = 0; ci < nd.children.length; ci++) {
      var ch = nd.children[ci];
      var ex = ch.px, ey = ch.py;
      var animEdge = anim && ch === anim.child;
      if (animEdge) { ex = ix; ey = iy; }
      var hi = (selSet[nd.id] && selSet[ch.id]) || animEdge;
      ctx.beginPath();
      ctx.moveTo(nd.px, nd.py);
      ctx.lineTo(ex, ey);
      ctx.strokeStyle = hi ? 'rgba(40,110,210,0.95)' : 'rgba(0,0,0,0.2)';
      ctx.lineWidth = hi ? 2.6 : 1.1;
      ctx.stroke();
    }
  }

  /* proposal pips beneath each node (the queue of things still to try) */
  for (var pq = 0; pq < nodes.length; pq++) {
    var pn = nodes[pq];
    var k = pn.openProposals.length;
    for (var pp = 0; pp < k; pp++) {
      var prop = pn.openProposals[pp];
      ctx.beginPath();
      ctx.arc(pipX(pn, pp, k), pipY(pn), PIP_R, 0, TAU);
      ctx.fillStyle = 'rgba(232,160,48,' + (0.4 + 0.55 * prop.promise).toFixed(3) + ')';
      ctx.fill();
      ctx.lineWidth = 1;
      ctx.strokeStyle = 'rgba(245,180,70,0.85)';
      ctx.stroke();
    }
  }

  /* nodes */
  ctx.textAlign = 'center';
  ctx.textBaseline = 'middle';
  for (var ni = 0; ni < nodes.length; ni++) {
    var n = nodes[ni];
    var isNew = (n === lastNew);
    var inSel = !!selSet[n.id];
    var isRoot = !n.parent;
    var animating = anim && n === anim.child;
    var x = animating ? ix : n.px;
    var y = animating ? iy : n.py;
    var r = animating ? ir : NODE_R;

    /* blue glow trailing the expansion, so the eye follows it down */
    if (animating) {
      var halo = ctx.createRadialGradient(x, y, r, x, y, r + 14);
      halo.addColorStop(0, 'rgba(100,180,255,0.5)');
      halo.addColorStop(1, 'rgba(100,180,255,0)');
      ctx.beginPath();
      ctx.arc(x, y, r + 14, 0, TAU);
      ctx.fillStyle = halo;
      ctx.fill();
    }

    ctx.beginPath();
    ctx.arc(x, y, r, 0, TAU);
    ctx.fillStyle = n.state === 'failed' ? '#2a2a3a'
                  : n.loss !== null ? v2c(l2v(n.loss))
                  : '#2c3050';
    ctx.fill();
    ctx.strokeStyle = animating ? '#1e7ad6' : isNew ? '#d99b00' : inSel ? '#1e7ad6' : isRoot ? 'rgba(0,0,0,0.55)' : 'rgba(0,0,0,0.3)';
    ctx.lineWidth = animating ? 3.2 : isNew ? 3 : inSel ? 2.5 : isRoot ? 2 : 1;
    ctx.stroke();

    /* label (fades in as an animating child grows) */
    ctx.textAlign = 'center';
    ctx.textBaseline = 'middle';
    ctx.globalAlpha = animating ? ap : 1;
    ctx.fillStyle = '#fff';
    if (n.state === 'failed') {
      ctx.font = 'bold 13px sans-serif';
      ctx.fillText('×', x, y);
    } else if (n.loss !== null) {
      ctx.font = '8.5px monospace';
      ctx.fillText(n.loss.toFixed(3), x, y);
    } else {
      ctx.font = '11px sans-serif';
      ctx.fillText('…', x, y);
    }
    ctx.globalAlpha = 1;

    /* visit counter N, tucked at the top-right corner; grows as visits accrue */
    if (!animating && n.n > 0) {
      ctx.font = 'bold 8px monospace';
      ctx.textAlign = 'left';
      ctx.textBaseline = 'alphabetic';
      ctx.shadowColor = 'rgba(255,255,255,0.9)';
      ctx.shadowBlur = 2;
      ctx.fillStyle = 'rgba(30,90,180,0.95)';
      ctx.fillText('N' + n.n, x + NODE_R * 0.6, y - NODE_R * 0.58);
      ctx.shadowBlur = 0;
    }
  }
}

/* ── Simulation state ── */
var root, selPath, lastNew, lastAction, iteration, bestLoss, autoTimer, anim, rafId;

function init() {
  nid = 0;
  root = mkNode(null, 0.074, null, 'evaluated');
  root.openProposals = randProps(4);
  backprop(root);
  selPath = [root];
  lastNew = null;
  anim = null;
  lastAction = 'Initialized — baseline loss 0.0740, 4 proposals queued.';
  iteration = 0;
  bestLoss = 0.074;
}

function openCount() {
  var ns = allNodes(root, []), s = 0;
  for (var i = 0; i < ns.length; i++) s += ns[i].openProposals.length;
  return s;
}

function step() {
  var res = selectFrom(root, [root]);
  selPath = res.path;

  if (res.pi === null) {
    res.node.openProposals = randProps(3);
    backprop(root);
    lastAction = 'Node ' + res.node.id + ' was out of ideas — seeded 3 fresh proposals.';
    lastNew = null;
    anim = null;
    return;
  }

  var node = res.node;
  var prop = node.openProposals.splice(res.pi, 1)[0];   /* one proposal leaves the queue */
  var fail = Math.random() < 0.10;
  var pLoss = node.loss !== null ? node.loss : 0.074;
  var raw = pLoss + (Math.random() - 0.48) * 0.022;
  var loss = Math.max(0.012, Math.min(0.115, raw));

  var child = mkNode(node, fail ? null : loss, prop, fail ? 'failed' : 'evaluated');
  if (!fail) child.openProposals = randProps(3);
  node.children.push(child);
  backprop(root);

  selPath = res.path.concat([child]);   /* keep the blue path running all the way to the new node */
  lastNew = child;
  iteration++;
  if (!fail && loss < bestLoss) bestLoss = loss;
  anim = { parent: node, child: child, t0: (window.performance || Date).now(), dur: ANIM_MS };

  lastAction = fail
    ? '[' + prop.label + '] promise ' + prop.promise + ' → verify FAILED'
    : '[' + prop.label + '] promise ' + prop.promise + ' → loss ' + loss.toFixed(4) + (loss <= bestLoss + 0.0001 ? '  ★ new best' : '');
}

/* ── Wire up ── */
document.addEventListener('DOMContentLoaded', function() {
  var canvas = document.getElementById('mc-canvas');
  var stepBtn = document.getElementById('mc-step');
  var autoBtn = document.getElementById('mc-auto');
  var resetBtn = document.getElementById('mc-reset');
  var statusEl = document.getElementById('mc-status');
  var iterEl = document.getElementById('mc-iter');
  var bestEl = document.getElementById('mc-best');
  var openEl = document.getElementById('mc-open');

  if (!canvas) return;

  function now() { return (window.performance || Date).now(); }

  function selSet() {
    var s = {};
    for (var i = 0; i < selPath.length; i++) s[selPath[i].id] = true;
    return s;
  }

  function paint(t) { render(canvas, root, selSet(), lastNew, anim, t); }

  function refreshStats() {
    statusEl.textContent = lastAction;
    iterEl.textContent = iteration;
    bestEl.textContent = bestLoss.toFixed(4);
    if (openEl) openEl.textContent = openCount();
  }

  function stopRaf() { if (rafId) { cancelAnimationFrame(rafId); rafId = null; } }

  function runAnim() {
    function frame(t) {
      paint(t);
      if (anim && (t - anim.t0) < anim.dur) {
        rafId = requestAnimationFrame(frame);
      } else {
        anim = null; rafId = null; paint(now());
      }
    }
    rafId = requestAnimationFrame(frame);
  }

  function doStep() {
    stopRaf();
    anim = null;          /* settle any in-flight animation to its final frame */
    step();
    refreshStats();
    if (anim) runAnim(); else paint(now());
  }

  init();
  paint(now());
  refreshStats();

  stepBtn.addEventListener('click', function() {
    if (iteration >= 30) { statusEl.textContent = 'Budget of 30 iterations exhausted — press Reset.'; return; }
    doStep();
  });

  autoBtn.addEventListener('click', function() {
    if (autoTimer) {
      clearInterval(autoTimer); autoTimer = null; autoBtn.textContent = 'Auto';
    } else {
      autoTimer = setInterval(function() {
        if (iteration >= 30) { clearInterval(autoTimer); autoTimer = null; autoBtn.textContent = 'Auto'; return; }
        doStep();
      }, 900);
      autoBtn.textContent = 'Pause';
    }
  });

  resetBtn.addEventListener('click', function() {
    if (autoTimer) { clearInterval(autoTimer); autoTimer = null; autoBtn.textContent = 'Auto'; }
    stopRaf();
    init(); paint(now()); refreshStats();
  });

  window.addEventListener('resize', function() { if (!rafId) paint(now()); });
});

})();
</script>

<script>
(function() {
'use strict';

function drawPUCT() {
  var Q     = parseFloat(document.getElementById('ps-Q').value);
  var pi    = parseFloat(document.getElementById('ps-pi').value);
  var Np    = parseInt(document.getElementById('ps-Np').value);
  var Nc    = parseInt(document.getElementById('ps-Nc').value);
  var c     = 0.5;
  var U     = c * pi * Math.sqrt(Math.max(Np, 1)) / (1 + Nc);
  var score = Q + U;

  document.getElementById('pv-Q').textContent  = Q.toFixed(2);
  document.getElementById('pv-pi').textContent = pi.toFixed(2);
  document.getElementById('pv-Np').textContent = Np;
  document.getElementById('pv-Nc').textContent = Nc;
  document.getElementById('pf-Q').textContent  = Q.toFixed(3);
  document.getElementById('pf-pi').textContent = pi.toFixed(2);
  document.getElementById('pf-Np').textContent = Np;
  document.getElementById('pf-Nc').textContent = Nc;
  document.getElementById('pf-U').textContent  = U.toFixed(3);
  document.getElementById('pf-score').textContent = score.toFixed(3);

  var canvas = document.getElementById('puct-canvas');
  if (!canvas) return;
  var ctx = canvas.getContext('2d');
  var w = canvas.width, h = canvas.height;
  ctx.clearRect(0, 0, w, h);
  ctx.fillStyle = '#ffffff';
  ctx.fillRect(0, 0, w, h);

  var maxV = 1.8;
  var bw = 52, gap = 18, base = h - 20;

  function bar(x, val, color, label, valStr) {
    var bh = Math.max(0, val / maxV * (h - 28));
    ctx.fillStyle = color;
    ctx.fillRect(x, base - bh, bw, bh);
    ctx.fillStyle = 'rgba(0,0,0,0.7)';
    ctx.font = '9px monospace';
    ctx.textAlign = 'center';
    ctx.fillText(label, x + bw / 2, base + 12);
    ctx.fillStyle = color;
    ctx.font = '8px monospace';
    ctx.fillText(valStr, x + bw / 2, base - bh - 6);
  }

  bar(gap,           Q,     '#4a90d9', 'Q',   Q.toFixed(3));
  bar(gap*2 + bw,    U,     '#e8a030', 'U',   U.toFixed(3));
  bar(gap*3 + bw*2,  score, '#55cc55', 'Q+U', score.toFixed(3));
}

document.addEventListener('DOMContentLoaded', function() {
  ['ps-Q','ps-pi','ps-Np','ps-Nc'].forEach(function(id) {
    var el = document.getElementById(id);
    if (el) el.addEventListener('input', drawPUCT);
  });
  drawPUCT();
});
})();
</script>]]></content><author><name></name></author><category term="blog" /><summary type="html"><![CDATA[Autoresearch with Monte Carlo Tree Search on Git Trees]]></summary><media:thumbnail xmlns:media="http://search.yahoo.com/mrss/" url="/blog/MCgitS/git_tree.gif" /><media:content medium="image" url="/blog/MCgitS/git_tree.gif" xmlns:media="http://search.yahoo.com/mrss/" /></entry><entry><title type="html">New York</title><link href="/blog/NYC26/" rel="alternate" type="text/html" title="New York" /><published>2026-04-11T00:00:00+00:00</published><updated>2026-04-11T00:00:00+00:00</updated><id>/blog/NYC26</id><content type="html" xml:base="/blog/NYC26/"><![CDATA[<!-- ## Berlin Over The Years -->

<style>
    .image-gallery {
        overflow: auto;
        margin-left: -1% !important;
    }

    .image-gallery li {
        float: left;
        float: top;
        display: block;
        margin: 0 0 1% 1%;
        width: 99%;
    }

    .image-gallery li a {
        text-align: top;
        text-decoration: none !important;
        color: #777;
    }

    .image-gallery li a span {
        display: block;
        text-overflow: ellipsis;
        overflow: hidden;
        white-space: nowrap;
        padding: 3px 0;
    }

    .image-gallery li a img {
        width: 100%;
        height: 100%;
        display: flex;
        vertical-align: top;
    }
</style>

<ul class="image-gallery">
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC5387.jpg"><img src="/photo_gallery/nyc26/DSC5387.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC5407.jpg"><img src="/photo_gallery/nyc26/DSC5407.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC5462.jpg"><img src="/photo_gallery/nyc26/DSC5462.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC5612.jpg"><img src="/photo_gallery/nyc26/DSC5612.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC5615.jpg"><img src="/photo_gallery/nyc26/DSC5615.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC5658.jpg"><img src="/photo_gallery/nyc26/DSC5658.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC5692.jpg"><img src="/photo_gallery/nyc26/DSC5692.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC5709.jpg"><img src="/photo_gallery/nyc26/DSC5709.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC5913.jpg"><img src="/photo_gallery/nyc26/DSC5913.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC5943.jpg"><img src="/photo_gallery/nyc26/DSC5943.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC5970.jpg"><img src="/photo_gallery/nyc26/DSC5970.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC6007.jpg"><img src="/photo_gallery/nyc26/DSC6007.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC6209.jpg"><img src="/photo_gallery/nyc26/DSC6209.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC6426.jpg"><img src="/photo_gallery/nyc26/DSC6426.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC6507.jpg"><img src="/photo_gallery/nyc26/DSC6507.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC6623.jpg"><img src="/photo_gallery/nyc26/DSC6623.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC6638.jpg"><img src="/photo_gallery/nyc26/DSC6638.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC6725.jpg"><img src="/photo_gallery/nyc26/DSC6725.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC6906.jpg"><img src="/photo_gallery/nyc26/DSC6906.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/nyc26/DSC7008.jpg"><img src="/photo_gallery/nyc26/DSC7008.jpg" /></a></li>
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
</ul>

<!--  -->

<!--  -->
<p><!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 --></p>]]></content><author><name></name></author><category term="photography" /><summary type="html"><![CDATA[Up and Down the Hudson to Boston]]></summary><media:thumbnail xmlns:media="http://search.yahoo.com/mrss/" url="/photo_gallery/nyc26/DSC5658.jpg" /><media:content medium="image" url="/photo_gallery/nyc26/DSC5658.jpg" xmlns:media="http://search.yahoo.com/mrss/" /></entry><entry><title type="html">Feynman-Kac Correctors</title><link href="/blog/FKC/" rel="alternate" type="text/html" title="Feynman-Kac Correctors" /><published>2026-03-18T00:00:00+00:00</published><updated>2026-03-18T00:00:00+00:00</updated><id>/blog/FKC</id><content type="html" xml:base="/blog/FKC/"><![CDATA[<script>
MathJax = {
  tex: {
    inlineMath: [['$','$'], ['\\(','\\)']],
    displayMath: [['$$','$$'], ['\\[','\\]']],
    processEscapes: true,
    tags: 'all'
  },
  options: {
    skipHtmlTags: ['script', 'noscript', 'style', 'textarea', 'pre']
  }
};
</script>

<script src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-mml-chtml.js" async=""></script>

<h2 id="fokker-planck-equation">Fokker-Planck Equation</h2>

<p>The Fokker-Planck equation describes the time evolution of the probability density function $p(\mathbf{x}, t)$ of a stochastic process:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\frac{\partial p(x, t)}{\partial t} = -\nabla \cdot \left[ \mu(x, t) p(x, t) \right] + \Delta \left[ \sigma^2(x, t) p(x, t) \right]
\end{align*}
$$
</div>

<p>where:</p>
<ul>
  <li>$\mu(x, t)$ is the drift coefficient (deterministic force)</li>
  <li>$\sigma^2(x, t)$ is the diffusion coefficient (related to the noise strength)</li>
  <li>$\nabla$ is the gradient operator</li>
  <li>$\Delta = \nabla^2$ is the Laplacian operator</li>
</ul>

<p>What this essential equation tells us is how the probability density of a system’s state changes over time due to both deterministic forces (drift) and random fluctuations (diffusion). The first term on the right-hand side represents the effect of the drift, while the second term accounts for the diffusion.
If we compute $\partial_t p(x,t)$ for a given $p(x,t)$, we can determine how the probability distribution evolves over time, which is crucial for understanding the dynamics of stochastic systems.
The nice thing is that normalization is automatically baked into this equation. The total probability is conserved, meaning that if we integrate $p(x, t)$ over all possible states $x$, it will always equal 1 for all time $t$.</p>

<p><img src="/blog/FKC/fokker_planck_gmm.gif" alt="Fokker-Planck Equation Evolution" style="max-width: 100%; height: auto; display: block; margin: 0 auto;" /></p>

<h2 id="the-product-of-distributions">The product of distributions</h2>

<p>The product of two distributions $p(x)$ and $q(x)$ is defined as:</p>
<div style="overflow-x: auto;">
$$
(p \cdot q)(x) = \frac{p(x) q(x)}{\int p(x') q(x') dx'} = \frac{1}{Z} p(x) q(x)
$$
</div>

<p>Why does the integral in the denominator appear? It ensures that the resulting distribution is properly normalized, meaning that the total probability integrates to 1.
The product $p(x) q(x)$ represents the unnormalized joint distribution and therefore has the correct shape, and dividing by the integral $\int p(x’) q(x’) dx’$ normalizes it to become a valid probability distribution.</p>

<h2 id="reward-tilted-sde">Reward Tilted SDE</h2>

<p>We consider the product of the original distribution $q_t(x)$ and an exponential of a reward function $r_t(x)$, scaled by a factor $\beta_t$:</p>
<div style="overflow-x: auto;">
$$
p_t(x) = \frac{1}{Z} \ q_t(x) \ \exp( \ \beta_t \ r(x))
$$
</div>

<p><strong>The big question is: what is the Fokker Planck equation of $p_t(x)$ from which we can easily read off the SDE if the process evolves according to the original process dynamics of $q_t(x)$?</strong>
Said differently, what’s the probability of a sample $x$ under the tilted $p_t$ if it evolves according to the stochastic differential equation of $q_t$?</p>

<p>So we need</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\partial_t p_t(x) &amp;= - \nabla \cdot [\mu \ p_t] + \frac{\sigma_p^2}{2} \Delta p_t \\
% &amp;= - \nabla \mu \cdot p_t - \mu \cdot \nabla p_t + \frac{\sigma_p^2}{2} \nabla \cdot [ \nabla p_t] \\
\end{align*}
$$
</div>

<p>where $\mu$ and $\sigma_p^2$ are the drift and diffusion of the SDE that generates $q_t$.</p>

<p>In order to obtain the FPE for the tilted distribution $p_t(x)$, we will exploit the FPE in log space, which offers some nice algebraic advantages.
To realize that we seek to obtain the log FPE, namely not $\partial_t p_t(x)$, but</p>
<div style="overflow-x: auto;">
$$
\partial_t \log p_t(x) =  \partial_t\log q_t(x) + \partial_t \beta_t \ r(x) - \partial_t \log Z
$$
</div>

<h3 id="fpe-in-log-space">FPE in log space</h3>

<p>Let us first recall the log-derivative trick which we will use extensively in the following calculations. For any function (or distribution) $q_t(x)$, we have the following identities:</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\partial_t \log p_t = \frac{1}{p_t} \partial_t p_t \quad \rightarrow \quad \partial_t p_t &amp;= p_t \ \partial_t \log p_t \\
\nabla \log p_t = \frac{1}{p_t} \nabla p_t \quad \rightarrow \quad \nabla p_t &amp;= p_t \ \nabla \log p_t \\
\Delta p_t = \nabla \cdot \nabla p_t \quad \rightarrow \quad \Delta p_t &amp;= \nabla \cdot [p_t \nabla \log p_t]  \\
&amp;= p_t \Delta \log p_t + \underbrace{\nabla p_t}_{=p_t \nabla \log p_t} \cdot \nabla \log p_t  \\
&amp;= p_t \Delta \log p_t + p_t ||\nabla \log p_t||^2
\end{align*}
$$
</div>

<p>This allows us to substitute all the gradient calculations in the FPE of our original process $q_t$ with the log derivatives,</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\partial_t q_t(x) &amp;= - \nabla \cdot [\mu \ q_t] + \frac{\sigma_q^2}{2} \Delta q_t \\
&amp;= - \nabla \mu_q \cdot q_t - \mu_q \cdot \nabla q_t + \frac{\sigma_q^2}{2} \nabla \cdot [ \nabla q_t] \\
&amp; \downarrow \quad \text{transform into log space} \\
q_t \ \partial_t \log q_t 
&amp;= - \nabla \mu_q \cdot q_t - \mu_q \cdot q_t \ \nabla \log q_t + \frac{\sigma_q^2}{2} q_t \ \left( \Delta \log q_t + ||\nabla \log q_t||^2 \right) \\ 
&amp; \downarrow \quad \text{divide by} \ q_t \\
\partial_t \log q_t 
&amp;= - \nabla \mu_q - \mu_q \cdot \nabla \log q_t + \frac{\sigma_q^2}{2} \Delta \log q_t + \frac{\sigma_q^2}{2} ||\nabla \log q_t||^2
\end{align*}
$$
</div>

<p>The term $\partial_t \log q_t$ is log transformed Fokker Planck equation and governs the evolution of the original distribution $q_t$ in log space.</p>

<p>How $q_t$ (or $\log q_t$) evolves over time is determined by the drift $\mu_q$ and the diffusion $\sigma_q^2$ of the original process.
We’re able to model that with the FPE of the original process $q_t$, but here we’re actually interested in the dynamics of $p_t$.
So the question is whether we can somehow massage the term</p>
<div style="overflow-x: auto;">
$$
\partial_t \log p_t(x) =  \partial_t\log q_t(x) + \partial_t \beta_t \ r(x) - \partial_t \log Z
$$
</div>
<p>into the form of the FPE from which we can then read off the drift and diffusion of the SDE the tilted process $p_t$.</p>

<p>In fact, we can express the change $\log q_t$ in terms of the sought-after process $p_t$ by rearranging the terms,</p>
<div style="overflow-x: auto;">
$$
\log q_t = \log p_t - \beta_t \ r(x) + \log Z
$$
</div>

<p>The derivatives of $\log q_t$ are</p>
<div style="overflow-x: auto;">
$$
\nabla \log q_t = \nabla \log p_t - \beta_t \nabla r(x) \quad \text{and} \quad \Delta \log q_t = \Delta \log p_t - \beta_t \Delta r(x)
$$
</div>

<p>Substituting these into the log FPE of $q_t$ gives us</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\partial_t \log q_t 
&amp;= - \nabla \mu_q - \mu_q \cdot {\color{red}\nabla \log q_t} + \frac{\sigma_q^2}{2} {\color{blue}\Delta \log q_t} + \frac{\sigma_q^2}{2} {\color{green}||\nabla \log q_t||^2} \\
&amp;= - \nabla \mu_q - \mu_q \cdot {\color{red}(\nabla \log p_t - \beta_t \nabla r(x))} + \frac{\sigma_q^2}{2} {\color{blue}(\Delta \log p_t - \beta_t \Delta r(x))} + \frac{\sigma_q^2}{2} {\color{green}||\nabla \log p_t - \beta_t \nabla r(x)||^2} \\
&amp;= \underbrace{- \nabla \mu_q - \mu_q \cdot {\color{red}\nabla \log p_t} + \frac{\sigma_q^2}{2} {\color{blue}\Delta \log p_t} + \frac{\sigma_q^2}{2} {\color{green}||\nabla \log p_t||^2}}_{\text{log FPE of} \ p_t}
+ {\color{red} \mu_q \cdot \beta_t \nabla r(x)} - \frac{\sigma_q^2}{2} {\color{blue}\beta_t \Delta r(x)} - \frac{\sigma_q^2}{2} \beta_t^2 ({\color{green} 2 \nabla\log p_t \cdot \nabla r(x) +||\nabla r(x)||^2)}
\end{align*}
$$
</div>

<p>And what do we spy with our little eyes?
The first four terms are actually the log FPE of $p_t$ itself!</p>

<p>We we plug in $\partial_t \log q_t$ into $\partial_t \log p_t(x)$, we get</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\partial_t \log p_t(x)
&amp;=  \partial_t\log q_t(x) + \partial_t \beta_t \ r(x) - \partial_t \log Z \\
&amp;= \overbrace{\underbrace{- \nabla \mu_q - \mu_q \cdot {\color{red}\nabla \log p_t} + \frac{\sigma_q^2}{2} {\color{blue}\Delta \log p_t} + \frac{\sigma_q^2}{2} {\color{green}||\nabla \log p_t||^2}}_{\text{log FPE of} \ p_t}
+ {\color{red} \mu_q \cdot \beta_t \nabla r(x)} - \frac{\sigma_q^2}{2} {\color{blue}\beta_t \Delta r(x)} - \frac{\sigma_q^2}{2} \beta_t^2 ({\color{green} 2 \nabla\log p_t \cdot \nabla r(x) +||\nabla r(x)||^2)}}^{\partial_t \log q_t} + \partial_t \beta_t \ r(x) - \partial_t \log Z 
\end{align*}
$$
</div>

<p>In order to go back to the non-log FPE, we just have to multiply both sides by $p_t$ and apply the log derivative identities from above in reverse,</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
p_t \partial_t \log p_t(x)
&amp;=  p_t \left( \partial_t\log q_t(x) + \partial_t \beta_t \ r(x) - \partial_t \log Z \right) \\
\partial_t p_t &amp;= - \nabla \mu_q \cdot p_t - \mu_q \cdot p_t \nabla \log p_t + \frac{\sigma_q^2}{2} p_t \left( \Delta \log p_t + ||\nabla \log p_t||^2 \right)
+ p_t \left(\mu_q \cdot \beta_t \nabla r(x) - \frac{\sigma_q^2}{2} \beta_t \Delta r(x) - \frac{\sigma_q^2}{2} \beta_t^2 (2 \nabla\log p_t \cdot \nabla r(x) +||\nabla r(x)||^2) + \partial_t \beta_t \ r(x) - \partial_t \log Z \right) \\
\partial_t p_t &amp;= - \nabla \mu_q \cdot p_t - \mu_q \cdot \nabla p_t + \frac{\sigma_q^2}{2} \left(  p_t \nabla \cdot [ \nabla \log p_t] + p_t \nabla \log p_t \cdot \nabla \log p_t \right)
+ p_t \left(\ldots \right) \\
\partial_t p_t &amp;= - \nabla \cdot [ \mu \ p_t ] + \frac{\sigma_q^2}{2} \left( p_t \nabla \cdot [ \nabla \log p_t] + \nabla p_t  \cdot \nabla \log p_t \right)
+ p_t \left(\ldots \right) \\
\partial_t p_t &amp;= - \nabla \cdot [ \mu \ p_t ] + \frac{\sigma_q^2}{2} \nabla \cdot [ \underbrace{p_t \nabla \log p_t}_{\nabla p_t}]
+ p_t \left(\ldots \right) \\
\partial_t p_t &amp;= - \nabla \cdot [ \mu \ p_t ] + \frac{\sigma_q^2}{2} \Delta p_t
+ p_t \Big(\underbrace{\mu_q \cdot \beta_t \nabla r(x) - \frac{\sigma_q^2}{2} \beta_t \Delta r(x) - \frac{\sigma_q^2}{2} \beta_t^2 (2 \nabla\log p_t \cdot \nabla r(x) +||\nabla r(x)||^2) + \partial_t \beta_t \ r(x)}_{g(x) } - \partial_t \log Z \Big) \\
\end{align*}
$$
</div>

<p>The last thing is how we actually compute $\partial_t \log Z$.
We can compute $\partial_t \log Z$ by differentiating the definition of $Z$.
For notational simplicity, let us define $\tilde{p} = \int q_t \exp(\beta_t r(x)) dx$</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\partial_t \log Z &amp;= \partial_t \log \int q_t(x) \exp(\beta_t r(x)) dx \\
&amp;= \frac{1}{Z} \ \partial_t \int q_t(x) \exp(\beta_t r(x)) dx \\
&amp;= \frac{1}{Z} \int \partial_t \ \tilde{p}_t \ dx \\
&amp;= \frac{1}{Z} \int \tilde{p} \ \partial_t \log \tilde{p}_t \ dx \\
&amp;= \int p_t \ \partial_t \log \tilde{p}_t \ dx \\
&amp;= \int p_t \ \left[  \partial_t \ \log q_t + \partial_t \beta_t \ r(x) \ \right] dx \\
\end{align*}
$$
</div>

<p>It should immediately be obvious that the term $\partial_t \log q_t + \partial_t \beta_t \ r(x)$ is the same term into which we’ve inserted the log derivatives.
So we expand the $\log q_t$ in the same way as we did above, and then multiply it with the $p_t$ in the integral, we get</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\partial_t \log Z_t 
&amp;= \int  - \nabla \cdot [ \mu \ p_t ] + \frac{\sigma_q^2}{2} \Delta p_t + p_t g(x) \ dx \\
&amp;= \int  \partial_t p_t + p_t g(x) \ dx \\
&amp;= \underbrace{\int  \partial_t p_t dx}_{=0} + \int p_t g(x) \ dx \\
&amp;= \mathbb{E}_{p_t} \left[ g(x) \right]
\end{align*}
$$
</div>
<p>which is literally just repeating the same steps that we did above to transform the log FPE to the FPE.</p>

<p>One of the key properties of the Fokker-Planck equation is that it conserves probability, meaning that the total probability integrates to 1 for all time $t$.
So if we integrate all the outflows and inflows of probability across the entire state space, they should balance out to zero, ensuring that the total probability remains constant.
If at any time step $t$, the change in probability $\partial_t$ over the entire space $x$ would not be zero, it would imply that probability is either being created or destroyed, which violates the fundamental principle of probability conservation.
So if from one $t$ to the next $t+dt$, all the changes in probability across the state space would be $-0.1$, it would imply that the total probability has decreased by 0.1 which would render our distribution $p_t$ invalid as it would no longer integrate to 1.</p>

<p>But in fact, we can show this analytically by integrating the FPE of $p_t$ over the entire state space.
For the left hand side, we have</p>
<div style="overflow-x: auto;">
$$
\int \partial_t p_t dx = \partial_t \int p_t dx = \partial_t 1 = 0
$$
</div>
<p>where we’ve used Leibniz rule where we can interchange differentiation and integration for different variables.
For the right hand side of the FPE, we have</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
&amp;\int - \nabla \cdot [ \mu \ p_t ] + \frac{\sigma_q^2}{2} \Delta p_t \ dx \\
= &amp; \int - \nabla \cdot [ \mu \ p_t ] + \frac{\sigma_q^2}{2} \Delta p_t \ dx \\
= &amp; \int - \nabla \mu \ p_t - \nabla p_t \mu + \frac{\sigma_q^2}{2} \nabla^2 p_t  \ dx \\
\end{align*}
$$
</div>

<p>Using integration by parts,</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\int f \ \nabla g \ dx = [f \ g]^{\infty}_{-\infty} - \int g \ \nabla f \ dx
\end{align*}
$$
</div>
<p>we can exploit that fact that for any reasonable probability distribution $p_t$, the probability density at the boundaries of the state space should be zero, meaning that $\lim_{x \to \infty} p_t(x) = 0$ and $\lim_{x \to -\infty} p_t(x) = 0$.</p>

<p>Thus we have,</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
&amp; \int - \nabla \mu \ p_t - \nabla p_t \mu + \frac{\sigma_q^2}{2} \nabla^2 p_t  \ dx \\
= &amp; \int - \nabla \mu \ p_t - [ p_t \ \mu ]^{\infty}_{-\infty} + p_t \nabla \mu + \frac{\sigma_q^2}{2} \nabla^2 p_t \ dx \\
= &amp; \int \frac{\sigma_q^2}{2} \nabla \ [p_t \nabla \log p_t ] \ dx \\
= &amp; \frac{\sigma_q^2}{2} \int  \nabla \ [p_t \nabla \log p_t ] \ dx \\
= &amp; \frac{\sigma_q^2}{2} [p_t \nabla \log p_t ]^{\infty}_{-\infty} \\
= &amp; \frac{\sigma_q^2}{2} \ 0
\end{align*}
$$
</div>
<p>where we’ve used the fact integration and derivatives cancel each other out while preserving the boundary evaluation conditions of $p_t$ at $\pm \infty$.</p>

<p>We can also play the integration by parts game one more time and show that</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
&amp; \frac{\sigma_q^2}{2} \int  \nabla \ [p_t \nabla \log p_t ] \ dx \\
= &amp; \frac{\sigma_q^2}{2} \int  \nabla p_t \cdot \nabla \log p_t + p_t \nabla^2 \log p_t \ dx \\
= &amp; \frac{\sigma_q^2}{2} \int  [p_t \nabla \log p_t]^\infty_{-\infty} - p_t \nabla^2 \log p_t + p_t \nabla^2 \log p_t \ dx \\
= &amp; \frac{\sigma_q^2}{2} \ 0
\end{align*}
$$
</div>

<p>All of this let’s us rewrite the FPE of our tilted distribution $p_t$ as</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\partial_t p_t &amp;= - \nabla \cdot [ \mu \ p_t ] + \frac{\sigma_q^2}{2} \Delta p_t
+ p_t \Big(g(x) - \mathbb{E}_{p_t} \left[ g(x) \right] \Big)
\end{align*}
$$
</div>

<p>Is it in fact quite interesting that the change of probability equation that we’ve derived consists of two different parts: the original FP equation of $p_t$ and an additional term that captures the effect of the reward tilt on the probability distribution.</p>

<p>The first term consists only of the original drift $\mu_q$ and diffusion $\sigma_q^2$ of the original process $q_t$, which implies that we can model the original SDE as is.
It is only the second term that introduces the effect of the reward tilt, modifying the probability distribution according to the function $g(x)$.
Every term in the second part $g(x)$ contains in fact the reward function $r(x)$, whether it be $r(x)$ directly or its gradients $\nabla r(x)$ and $\Delta r(x)$, which means that the reward function is the only source of change in the probability distribution $p_t$.</p>

<h2 id="the-feynman-kac-equation">The Feynman-Kac Equation</h2>

<p>The Feynman-Kac equation is a fundamental result that links partial differential equations (PDEs) to expectations of stochastic processes. For a function $p$ satisfying certain boundary conditions, it defines a PDE.</p>

<p>For a SDE for a particle $x_t$ defined as</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
dx_s = \mu(x_s) ds + \sigma(x_s) dW_s
\end{align*}
$$
</div>
<p>We can define two key operators:
The first one is the the generator that acts on the test function and describes its instantaneous rate of change of the test function $\theta$ along the paths of the SDE. We’ve seen this in the Feynman Kac equation where we used it to define</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\mathcal{L}_t [\phi] = \mu(x) \cdot \nabla \phi + \frac{\sigma^2(x)}{2} \nabla^2 \phi
\end{align*}
$$
</div>
<p>where the FK equation is defined as $\partial_t \phi +\mathcal{L}_t[\phi] + g \ \phi = 0$.</p>

<p>The adjoint generator acts on the probability distribution $p$ and describes how the probability distribution evolves over time under the dynamics of the SDE. It is defined as</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\mathcal{L}^*_t [p] = - \nabla \cdot [\mu(x) p] + \frac{\sigma^2(x)}{2} \nabla^2 p
\end{align*}
$$
</div>

<p>Both generators are linked through the identity</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\int \phi \ \mathcal{L}^*_t [p] dx = \int p \ \mathcal{L}_t [\phi] dx
\end{align*}
$$
</div>

<p>Now, we’re interested in some test function on $x$ that we name $\phi(x, t)$ and we’re interested in what the distribution of $\phi(x,0)$ and the end of our reverse diffusion process is.
We have defined the probability evolution of $p$ to be governed by the Feynman-Kac PDE,</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\partial_t p_t &amp;= - \nabla \cdot [ \mu \ p_t ] + \frac{\sigma_q^2}{2} \Delta p_t
+ p_t \overbrace{\Big(g(x) - \mathbb{E}_{p_t} \left[ g(x) \right] \Big)}^{\bar{g}}
\end{align*}
$$
</div>

<p>This tells us how the probability distribution $p_t$ evolves over time under the influence of both the original dynamics of the SDE (captured by the first two terms) and the reward tilt (captured by the last term).</p>

<h2 id="connecting-it-to-the-feynman-kac-equation">Connecting it to the Feynman Kac equation</h2>

<p>In a previous <a href="https://ludwigwinkler.github.io/blog/FeynmanKac/">blog post on the Feynman-Kac equation</a>, we saw that for very particular PDE that are defined in the state space of the SDE, we can express the solution of the PDE as an expectation over the trajectories of the SDE.
Namely, we have a PDE of the form</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
-\partial_t u(x,t) = \mu(x,t) \nabla u(x,t) + \frac{1}{2}\sigma^2(x,t) \nabla^2 u(x,t) + g(x,t) u(x,t)
\end{align*}
$$
</div>
<p>subject to the terminal condition $u(x,T) = \Phi_T(x)$, then the solution of this PDE is given by the Feynman-Kac formula.
A straightforward option would be to evolve this very particular PDE backward in time form $T$ to an earlier $t$ and then read off the solution at $t=0$. This would give us $u(x,t)$.</p>

<p>If the dynamics of $x$ are governed by the Ito drift-diffusion process defined above, then the solution of the PDE is given by the Feynman-Kac formula</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
u(x,t) = \mathbb{E} \Big[ \Phi_T(x_T) \exp \left( \int_t^T g(x_s, s) ds \right) \Big| x_t = x \Big]
\end{align*}
$$
</div>
<p>where the expectation is taken over all trajectories of the SDE that start at $x$ at time $t$ and evolve according to the dynamics of the SDE until time $T$ with values $x_T$.</p>

<p>We now have two differential equations</p>
<div style="overflow-x: auto;">
$$
\begin{align*}\partial_t p_t &amp;= - \nabla \cdot [ \mu \ p_t ] + \frac{\sigma_q^2}{2} \Delta p_t
+ p_t \Big(g(x) - \mathbb{E}_{p_t} \left[ g(x) \right] \Big) \\
g(x,t) = &amp; \mu_q \cdot \beta_t \nabla r(x) - \frac{\sigma_q^2}{2} \beta_t \Delta r(x) - \frac{\sigma_q^2}{2} \beta_t^2 (2 \nabla\log p_t \cdot \nabla r(x) +||\nabla r(x)||^2) + \partial_t \beta_t \ r(x) \\
-\partial_t u(x,t) &amp;= \mu(x,t) \nabla u(x,t) + \frac{1}{2}\sigma^2(x,t) \nabla^2 u(x,t) + g(x,t) u(x,t)
\end{align*}
$$
</div>

<p>The samples $x$ evolve according to the $dx = \mu_t dt + \sigma_t dW_t$ and induce the stochastic process $q_t$ but we’re interested in the distribution $p_t(x) = \frac{1}{Z} q_t(x) \exp(\beta r(x))$ and want to know how likely it is to sample $x$ from $p_t$ if we evolve according to the SDE of $q_t$.
The function $g$ consists almost exclusively of terms that contain the reward function $r(x)$, which nicely shows that any change in the probability distribution $p_t$ is solely due to the reward tilt encapsulated in $g$.</p>

<p>Let’s assume a test function $u(x,t)$ with some terminal condition $u(x,T)$ and we evolve it backward in time according to the Feynman-Kac PDE over the trajectories of the SDE defined by $\mu$ and $\sigma$. As we’re evolving some terminal function backwards in time, this is obviously very closely linked to the Kolmogorov backward equation.</p>

<p><strong>The question the FKC correctors try to answer is: Given my samples $x$ generated from $q_t$ but tilted towards some reward function $r(x)$, what is the expectation of $u(x,T)$ if I were to evolve further from $t$ to $T$ according to the original SDE of $q_t$?</strong></p>

<p>We’re interested in the expectation</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{E}_{p_T} \Big[ u(x,T) \ \Big| \ x_t = x \Big] = \int u(x,T) \ p_T(x) dx = \int u(x,T) \ \frac{1}{Z_T} q_T(x) \exp(\beta_T r(x)) dx
\end{align*}
$$
</div>

<p>In the equation above, $q$ generates the samples $x$ according to the SDE, $r(x)$ tilts our distribution towards higher rewards such that our samples $x$ should exhibit certain characteristics favoured by $r(x)$ and then we want to know the expected value of $u(x,T)$ under this tilted distribution.</p>

<p>Now we can ask ourselves how $\int u(x,T) \ p_T(x) dx$ evolves over time. To make things a tad easier, we’ll for now just consider the unnormalized distribution $\tilde{p}_t(x) = q_t(x) \exp(\beta_t r(x))$ and then we can later on divide by (some estimate of) $Z_t$ to get the normalized distribution $p_t$.
Fortunately, every component in $\partial_t p_t$ is a function of $p_t$, so we extract $1/Z_t$ out of every component to get</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\partial_t \tilde{p}_t &amp;= - \nabla \cdot [ \mu \ \tilde{p}_t ] + \frac{\sigma_q^2}{2} \Delta \tilde{p}_t
+ \tilde{p}_t \bar{g}(x)
\end{align*}
$$
</div>
<p>The proof for this is in the FKC paper, but it is quite intuitive if you think about it. The original FPE of $p_t$ is linear in $p_t$, so if we replace $p_t$ with $\tilde{p}_t / Z_t$, we can just pull out the $1/Z_t$ from every term and get the FPE of $\tilde{p}_t$.
The authors show the consistency of this in the appendix.</p>

<p>Thus we’re want to know</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\frac{d}{dt} \int u(x,t) \ \tilde{p}_t(x) dx
\end{align*}
$$
</div>
<p>behaves.
Using Leibniz rule, we can interchange differentiation and integration for different variables, which gives us</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\frac{d}{dt} \int u(x,t) \ \tilde{p}_t(x) dx 
&amp;= \int \partial_t u(x,t) \ \tilde{p}_t(x) dx + \int u(x,t) \ \partial_t \tilde{p}_t(x) dx \\
&amp;= \int \left(-\mu(x,t) \nabla u(x,t) - \frac{1}{2}\sigma^2(x,t) \nabla^2 u(x,t) - g(x,t) u(x,t) \right) \ \tilde{p}_t(x) dx \\
&amp; \quad + \int u(x,t) \ \left(- \nabla \cdot [ \mu \ \tilde{p}_t ] + \frac{\sigma_q^2}{2} \Delta \tilde{p}_t + \tilde{p}_t \bar{g}(x) \right) dx \\
&amp;= \int \left(-\mu(x,t) \nabla u(x,t) - \frac{1}{2}\sigma^2(x,t) \nabla^2 u(x,t) - g(x,t) u(x,t) \right) \ \tilde{p}_t(x) dx \\
&amp; \quad + \int \left(\mu(x,t) \nabla u(x,t) + \frac{1}{2}\sigma^2(x,t) \nabla^2 u(x,t) + g(x,t) u(x,t) \right) \ \tilde{p}_t(x) dx \quad \quad | \quad \quad \text{integration by parts with } \tilde{p}_t(\pm \infty) = 0 \\
&amp;= 0
\end{align*} 
$$
</div>
<p>The integration by parts substitution is a pure algebraic busy work and I skipped it here. It’s just applying the product rule many times to fully separate every term, then applying integration by parts to every term by pulling in $u$ and recombining the terms.</p>

<p>Integrating the dynamics from $0$ to $T$ gives us</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\int_0^T \frac{d}{dt} \int u(x,t) \ \tilde{p}_t(x) dx dt
&amp;= \left[ \int u(x,t) \ \tilde{p}_t(x) dx \right]^T_0 \\
&amp;= \int u(x,T) \ \tilde{p}_T(x) dx - \int u(x,0) \ \tilde{p}_0(x) dx \\ 
&amp;= 0 \\
&amp; \downarrow \quad \text{rearranging} \\
\int u(x,T) \ \tilde{p}_T(x) dx &amp;= \int u(x,0) \ \tilde{p}_0(x) dx
\end{align*}
$$
</div>

<p>All that remains is substituing back the Feynman-Kac formula for $u(x,0)$ and substituting $\tilde{p}_t = p_t Z_t$, which gives us</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\int u(x,T) \ \tilde{p}_T(x) dx &amp;= \int \mathbb{E} \Big[ \Phi_T(x_T) \exp \left( \int_0^T g(x_s, s) ds \right) \Big| x_0 = x \Big] \ \tilde{p}_0(x) dx \\ 
Z_T \int u(x,T) \ p_T(x) dx &amp;= Z_0 \int \mathbb{E} \Big[ \Phi_T(x_T) \exp \left( \int_0^T g(x_s, s) ds \right) \Big| x_0 = x \Big] \ p_0(x) dx
\end{align*}
$$
</div>

<p>Since at $t=0$, we have $p_0(x) = q_0(x) \exp(\beta_0 r(x)) / Z_0$ with $\beta_0 = 0$ we get $p_0(x) = q_0(x) / Z_0$ with $Z_0 = 1$, we can simplify the right hand side to get</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
Z_T \int u(x,T) \ p_T(x) dx &amp;= \int \mathbb{E} \Big[ \Phi_T(x_T) \exp \left( \int_0^T g(x_s, s) ds \right) \Big| x_0 = x \Big] \ q_0(x) dx \\
Z_T \mathbb{E}_{p_T} [ u(x,T)] &amp;= \mathbb{E}_{q_0} \Big[ \mathbb{E} \Big[ \Phi_T(x_T) \exp \left( \int_0^T g(x_s, s) ds \right) \Big| x_0 = x \Big] \Big] \\
\mathbb{E}_{p_T} [ u(x,T)] &amp;= \frac{1}{Z_T} \mathbb{E} \Big[ \Phi_T(x_T) \exp \left( \int_0^T g(x_s, s) ds \right) \Big| x_0 = x \Big] \\
\end{align*}
$$
</div>

<p>This is pure importance sampling on a trajectory level, where we sample trajectories from the original SDE, calculate $g(x_s, s)$ along the trajectory and simply integrating it in log space as a simple sum over the time steps.
In the <a href="https://ludwigwinkler.github.io/blog/ESSSMC/">SMC blog post</a> we saw that we can use the weights from SMC to estimate $\hat{Z}_T = 1/K \sum_k^K \exp(\int_0^T g(x,s) ds)$, which gives us a practical way to compute the expectation of $u(x,T)$ under the tilted distribution $p_T$ by sampling trajectories from the original SDE and weighting them according to the reward tilt encapsulated in $g$.</p>

<p><img src="/blog/FKC/importance_weights.gif" alt="Importance Weighting of Generated Samples" style="max-width: 100%; height: auto; display: block; margin: 0 auto;" /></p>]]></content><author><name></name></author><category term="blog" /><summary type="html"><![CDATA[Correct your steps.]]></summary><media:thumbnail xmlns:media="http://search.yahoo.com/mrss/" url="/blog/FKC/steered_diffusion.gif" /><media:content medium="image" url="/blog/FKC/steered_diffusion.gif" xmlns:media="http://search.yahoo.com/mrss/" /></entry><entry><title type="html">Pacific Northwest</title><link href="/blog/Seattle26/" rel="alternate" type="text/html" title="Pacific Northwest" /><published>2026-03-16T00:00:00+00:00</published><updated>2026-03-16T00:00:00+00:00</updated><id>/blog/Seattle26</id><content type="html" xml:base="/blog/Seattle26/"><![CDATA[<!-- ## Berlin Over The Years -->

<style>
    .image-gallery {
        overflow: auto;
        margin-left: -1% !important;
    }

    .image-gallery li {
        float: left;
        float: top;
        display: block;
        margin: 0 0 1% 1%;
        width: 99%;
    }

    .image-gallery li a {
        text-align: top;
        text-decoration: none !important;
        color: #777;
    }

    .image-gallery li a span {
        display: block;
        text-overflow: ellipsis;
        overflow: hidden;
        white-space: nowrap;
        padding: 3px 0;
    }

    .image-gallery li a img {
        width: 100%;
        height: 100%;
        display: flex;
        vertical-align: top;
    }
</style>

<ul class="image-gallery">
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    <li><a href="/photo_gallery/seattle26/DSC4878.jpg"><img src="/photo_gallery/seattle26/DSC4878.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/seattle26/DSC5055.jpg"><img src="/photo_gallery/seattle26/DSC5055.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/seattle26/DSC5100.jpg"><img src="/photo_gallery/seattle26/DSC5100.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/seattle26/DSC5132-Pano.jpg"><img src="/photo_gallery/seattle26/DSC5132-Pano.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/seattle26/DSC5200.jpg"><img src="/photo_gallery/seattle26/DSC5200.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/seattle26/DSC5239.jpg"><img src="/photo_gallery/seattle26/DSC5239.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/seattle26/DSC62681.jpg"><img src="/photo_gallery/seattle26/DSC62681.jpg" /></a></li>
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
</ul>

<!--  -->

<!--  -->
<p><!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 --></p>]]></content><author><name></name></author><category term="photography" /><summary type="html"><![CDATA[Upper Left America]]></summary><media:thumbnail xmlns:media="http://search.yahoo.com/mrss/" url="/photo_gallery/seattle26/DSC5239.jpg" /><media:content medium="image" url="/photo_gallery/seattle26/DSC5239.jpg" xmlns:media="http://search.yahoo.com/mrss/" /></entry><entry><title type="html">The Weight of Sampling</title><link href="/blog/ESSSMC/" rel="alternate" type="text/html" title="The Weight of Sampling" /><published>2026-02-12T00:00:00+00:00</published><updated>2026-02-12T00:00:00+00:00</updated><id>/blog/ESSSMC</id><content type="html" xml:base="/blog/ESSSMC/"><![CDATA[<script>
MathJax = {
  tex: {
    inlineMath: [['$','$'], ['\\(','\\)']],
    displayMath: [['$$','$$'], ['\\[','\\]']],
    processEscapes: true,
    tags: 'all'
  },
  options: {
    skipHtmlTags: ['script', 'noscript', 'style', 'textarea', 'pre']
  }
};
</script>

<script src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-mml-chtml.js" async=""></script>

<h2 id="why-we-need-it">Why we need it</h2>

<p>Suppose you want to estimate $\mathbb{E}_p[f(X)]$ for some target distribution $p$. In a perfect world, you’d draw $N$ i.i.d. samples from $p$ and compute the sample mean. The variance of that estimator drops as $1/N$ — that’s what you learn in ML 101.</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\hat{\mu}_f = \frac{1}{N} \sum_{i=1}^{N} f(X_i) \quad \quad \text{where} \quad X_i \sim p \quad \quad \text{and} \quad \quad \mathbb{V}[\hat{\mu}_f] = \frac{1}{N} \sigma_f
\end{align*}
$$
</div>

<p>If we can draw i.i.d. samples from $p$, then there’s no real problem in computing the expectation.
But alas, we can’t always draw i.i.d. samples from $p$.</p>

<p>But sampling from $p$ is not always possible.
This can be for a variety of reasons, but one common reason is that the distribution $p$ is complex and high-dimensional and its normalization constant is not known.
Oftentimes, we only know $p$ up to a normalizing constant $p = \tilde{p} / Z$.
Without knowing $Z$, we can’t assign the correct probability to a point $x$ under $p(x)$.
For the same value of $\tilde{p}(x)$, the probability of $x$ under $p(x)$ is different if $Z$ is different.</p>

<p>The mathematical version of this predicament shows up everywhere.
Consider the workhorse of modern machine learning: Bayesian posterior inference.
You observe data $\mathcal{D}$ and want to reason about parameters $\theta$ under the posterior</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
p(\theta | \mathcal{D}) = \frac{p(\mathcal{D} | \theta)\, p(\theta)}{p(\mathcal{D})}, \quad \text{where} \quad p(\mathcal{D}) = \int p(\mathcal{D} | \theta)\, p(\theta)\, d\theta
\end{align*}
$$
</div>

<p>The numerator — likelihood times prior — you can evaluate pointwise, no problem.
But the denominator $p(\mathcal{D})$, the marginal likelihood, requires the probability of your entire dataset $\mathcal{D}$ under the model.</p>

<p>So either you somehow magically know the probability of every sample in your dataset $\mathcal{D}$ under the model, or you need to estimate it by defining every possible value of $\theta$ and evaluating the likelihood and prior for each of them.
Every possible value of $\theta$ for continuous parameters is an infinite number of values, so you’d need an infinite amount of computational resources to evaluate the likelihood and prior for each of them.</p>

<p>For anything beyond toy conjugate models, this integral is intractable: there is no closed-form solution and the dimensionality of $\theta$ makes brute-force numerical integration hopeless.
So you’re stuck with a <strong>distribution you can <em>evaluate</em> (up to a constant) but can’t <em>sample from</em> directly</strong>.
You know the shape of the landscape, but you have no way to generate points that are distributed according to it.</p>

<p>This is where sampling schemes come in.
But in practice, the samples these schemes produce are almost never i.i.d. from the target:</p>

<ul>
  <li><strong><a href="https://ludwigwinkler.github.io/blog/HMC/">MCMC</a></strong> produces samples that are serially correlated (each sample depends on the previous one).</li>
  <li><strong>Importance sampling</strong> draws from a proposal $q \neq p$, then corrects with weights — but those weights can be wildly uneven.</li>
  <li><strong>Particle filters</strong> resample with replacement, creating duplicate particles (see <a href="https://ludwigwinkler.github.io/blog/AISSMC/">Annealed Importance Sampling</a> for a related staged-sampling scheme).</li>
</ul>

<p>All of these schemes are ways to sample from $p$ even if you can’t sample from $p$ directly. In all these cases, your $N$ raw samples carry <em>less information</em> than $N$ truly independent samples from the OG $p$ would. You need a single number that answers: <strong>“how many i.i.d. samples is my collection of samples I proposed actually worth?”</strong> That number is the ESS.</p>

<p>Without it, you’re flying blind — you can’t meaningfully compare algorithms, diagnose poor mixing, set stopping criteria, or calibrate your confidence in Monte Carlo estimates.</p>

<h2 id="lets-get-the-equations-rolling">Let’s get the equations rolling</h2>

<h3 id="importance-sampling">Importance Sampling</h3>

<p>We want to compute:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mu = \mathbb{E}_p[f(X)] = \int f(x)\, p(x)\, dx
\end{align*}
$$
</div>

<p>We can’t sample from $p$ directly, so we draw $X_1, \dots, X_N \overset{\text{i.i.d.}}{\sim} q$ from a proposal distribution $q$, and use the importance sampling identity:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mu =\int \underbrace{\frac{q(x)}{q(x)}}_{=1} \ p(x) \ f(x) dx =  \int q(x) \frac{p(x)}{q(x)} q(x)\ f(x) dx = \mathbb{E}_q\!\left[w(X) \ f(X)\right]
\end{align*}
$$
</div>

<p>where $w(x) = p(x)/q(x)$ are the (unnormalized) importance weights.</p>

<p>Below you can see the importance weights $w(x) = p(x)/q(x)$ for a Gaussian mixture model from the cover.
If we sample from $q$ we have to take into consideration that we’re not sampling from $p$ directly.
The correction for sampling from $q$ but wanting to estimate the expectation under $p$ is given by the importance weights $w(x) = p(x)/q(x)$.
For all examples in the visualization below, $p(x) &lt; q(x)$, such that the importance weights will all be smaller than 1.
These weights $w(x) &lt; 1$, will thus downweight the contribution of the samples from $q$ in the expectation estimator,</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\hat{\mu}_f = \frac{1}{N} \sum_{i=1}^{N} w(X_i)\, f(X_i)
\end{align*}
$$
</div>

<p>whereas parts where $p(x) &gt; q(x)$, the importance weights will be greater than 1, thus upweighting the contribution of the samples from $q$ in the expectation estimator.</p>

<p><img src="/blog/ESS/weights.png" alt="Importance Sampling" style="max-width: 100%; height: auto; display: block; margin: 0 auto;" /></p>

<h4 id="deriving-the-self-normalized-estimator">Deriving the self-normalized estimator</h4>

<h5 id="the-problem-with-standard-importance-sampling">The problem with standard importance sampling</h5>

<p>If we know $p$ exactly (including its normalizing constant), life is simple. The <strong>standard IS estimator</strong> is:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\hat{\mu}_{\text{IS}} = \frac{1}{N} \sum_{i=1}^{N} w(X_i)\, f(X_i), \qquad w(x) = \frac{p(x)}{q(x)}
\end{align*}
$$
</div>

<p>This is unbiased:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{E}_q[\hat{\mu}_{\text{IS}}] = \mathbb{E}_q[w(X)f(X)] = \int f(x)\frac{p(x)}{q(x)}q(x)\,dx = \int f(x)\,p(x)\,dx = \mu
\end{align*}
$$
</div>

<p>But in practice — especially in Bayesian inference — we only know $p$ up to a normalizing constant $p = \tilde{p} / Z$. 
While we’re interested in estimating $w = p / q$, we’re left holding the bag with its unnormalized cousin where $Z$ is unknown,</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
w(x) = \frac{p(x)}{q(x)} = \frac{1}{Z}\frac{\tilde{p}(x)}{q(x)} = \frac{1}{Z} \tilde{w}(x)
\end{align*}
$$
</div>

<p>which makes us arrive again at the same spot we started from that we don’t know $Z$ in the first place.
But with unnormalized weights $\tilde{w}$, there is a trick we can pull to bootstrap ourselves out of this seemingly dead end.</p>

<p>The key observation is that the normalizing constant $Z$ is itself is an integral over the whole space $x$ of $\tilde{p}(x)$.
We can therefore use importance sampling to evaluate the integral as in</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
Z = \int \tilde{p}(x)\,dx =  \int \frac{q(x)}{q(x)} \tilde{p}(x) = \int \ q(x) \ \frac{\tilde{p}(x)}{q(x)}\,dx = \int q(x) \ \tilde{w}(x) dx = \mathbb{E}_q[\tilde{w}(X)]
\end{align*}
$$
</div>

<p>Given that we’ve generated the samples $x \sim q$ proportionally to the distribution $q$ by definition, the expectation reduces to a simple average over the samples $x$,</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\hat{Z} = \frac{1}{N}\sum_{i=1}^{N} \tilde{w}(X_i) \quad \quad ; \quad \quad X_i \sim q
\end{align*}
$$
</div>

<p>Going back to our original objective we want to identify</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\hat{\mu}_{\text{IS}} &amp;= \frac{1}{N} \sum_{i=1}^{N} w(X_i)\, f(X_i) \\
&amp;= \frac{1}{N} \sum_{i=1}^{N} \frac{p(X_i)}{q(X_i)}\, f(X_i) \\
&amp;=\frac{1}{N} \sum_{i=1}^{N} \frac{1}{Z}\frac{\tilde{p}(X_i)}{q(X_i)}\, f(X_i) \\
&amp;\approx \frac{1}{N}  \frac{1}{\hat{Z}} \sum_{i=1}^{N} \tilde{w}(X_i)\, f(X_i) \\
&amp;= \frac{1}{N}  \frac{1}{\frac{1}{N}\sum_{i=j}^{N} \tilde{w}(X_j)} \sum_{i=1}^{N} \tilde{w}(X_i)\, f(X_i) \\
&amp;= \frac{\sum_{j=i}^{N} \tilde{w}(X_i)\, f(X_i)}{\sum_{j=i}^{N} \tilde{w}(X_j)} \\
\end{align*}
$$
</div>

<p>Whereas the importance estimator \(\hat{\mu}_{\text{IS}}\) assumes access to the target distribution $p$ including its normalization constant $Z$, using the unnormalized weights \(\tilde{w}\) to estimate \(\hat{Z}\) turns this estimator it to the <em>self-normalized</em> estimator \(\hat{\mu}_{\text{SN}}\).</p>

<p>We can simplify the equation for $\hat{\mu}_{\text{SN}}$ even further by observing that the sum in the denominator is self contained (as in sum over $j$), so we can write</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\hat{\mu}_{\text{SN}} 
&amp;= \frac{\sum_{i=1}^{N} \tilde{w}(X_i)\, f(X_i)}{\sum_{j=1}^{N} \tilde{w}(X_j)} \\
&amp;= \sum_{i=1}^{N} \frac{\tilde{w}(X_i)}{\sum_{j=1}^{N} \tilde{w}(X_j)}\, f(X_i) \\
&amp;= \sum_{i=1}^{N} \bar{w}(X_i)\, f(X_i) \\
\end{align*}
$$
</div>

<p>where the normalized weights $\bar{w}$ are defined as</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\bar{w}_i = \frac{\tilde{w}_i}{\sum_{j=1}^{N} \tilde{w}_j}
\end{align*}
$$
</div>

<h4 id="how-much-does-our-estimator-actually-vary">How much does our estimator actually vary?</h4>

<p>The self-normalized estimator $\hat{\mu}_{\text{SN}}$ is a ratio of two random quantities.
This makes its variance analysis more subtle than the simple i.i.d. case.</p>

<p>The key question is: how much does our estimate vary from one set of samples to the next?
If we draw $N$ samples from $q$ repeatedly, how much will $\hat{\mu}_{\text{SN}}$ fluctuate around the true mean $\mu$?</p>

<p>Here’s a (somewhat) intuitive example: Let’s assume we want an estimate of the average height of people in Europe by sending $N$ random people a text message. Imagine you picking 100 random cell phone numbers and blasting off a text saying “Yo, how tall are you?”. These text message could arrive on a Greek island in the sun, in the eternal winter darkness of a Finnish village or in the land of the giants, Holland.</p>

<p>Problem is, you don’t know how tall nor how representative the person reading the text is for the average height of people in Europe.
Ideally, you want to representatively sample from all countries in Europe from all height groups: the small people in Holland, the tall and small people in Portugal and the average tall people in Greece.</p>

<p>The variance of your little survey depends on two factors:</p>
<ol>
  <li>The intrinsically different heights in Europe that exists regardless of who you pick. Every country has differently tall people.</li>
  <li>Who you send the text to in Europe from your 100 allowed text messages.</li>
</ol>

<p>Particularly the second source of variability is tricky in our case. 
Did you send the text message it by accident to all Dutch men measuring 2.14m? Then your average is like 2.05m. Hardly realistic.
Or maybe you send it to all Greek grandmas towering at a mighty 1.55m. Also hardly representative.</p>

<p>In more mathy terms we would say that the variance of $\hat{\mu}_{\text{SN}}$ depends on two factors:</p>
<ol>
  <li>The intrinsic variability of $f(X)$ under the target distribution $p$ captured by $\sigma_f^2 = \text{Var}_p[f(X)]$. (the natural distribution of heights in Europe)</li>
  <li>The mismatch between proposal $q$ and target $p$, reflected in the weight variability. ($q$ being the proposed people you send your height text message to)</li>
</ol>

<p>When $q$ is a poor approximation to $p$, a few samples will have very large weights while most have tiny weights.
For simplicites sake assume that $q$ is a uniform $q=1/100$.
The probability of picking the tiny Greek grandmas or the super tall male Dutchies is very small as their heights are at the far ends of the height distribution in Europe. The average height is around 1.75m, so picking a height of 1.55m or 2.14m is very unlikely.</p>

<p>It’s important to note that we’re self-normalizing the weights $\tilde{w}_i$ with all our samples, so we have \(\bar{w}_i = \tilde{w} / \sum_{j=1}^{100} \tilde{w}_j\).</p>

<p>If we ask 98 Greek grandmas and towering Dutchies and only 2 slightly less extreme samples, the contribution of those extreme 98 weights will be negligible.
Remember that in order to compute the average weight we’ll compute,</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\hat{\mu}_{\text{SN}} 
&amp;= \sum_{i=1}^{N} \frac{\tilde{w}(X_i)}{\sum_{j=1}^{N} \tilde{w}(X_j)}\, f(X_i) \\
&amp;= \overbrace{\sum_{i=1}^{98} \underbrace{\frac{\tilde{w}(X_i)}{\sum_{j=1}^{N} \tilde{w}(X_j)}}_{\approx 0}\, f(X_i)}^{\text{Unrepresentative Greek Grandmas \&amp; Tall Dutchies}} + \overbrace{\sum_{i=1}^{2} \frac{\tilde{w}(X_i)}{\sum_{j=1}^{N} \tilde{w}(X_j)}\, f(X_i)}^{\text{effective number of samples}=2} \\
\end{align*}
$$
</div>

<p>But lo and behold, due to $p$ being very small, the weights of the 98 unrepresentative samples have a negligible contribution to the overall value.
In effect, we’re trying to estimate the average European height with two samples. Not ideal when we had assumed that we used 100 samples.</p>

<p>This weight concentration in those two meaningful samples means we’re effectively using far fewer than $N$ samples, inflating the estimator variance.
The <strong>effective sample size</strong> (ESS) quantifies exactly how many “equivalent i.i.d. samples” our weighted samples represent. In our case it’s two, not a hundred.</p>

<p>But how do we quantify this effective sample size?</p>

<h3 id="the-variance-of-our-estimator">The Variance of our Estimator</h3>

<p>Let’s denote the number of effective samples as $N_{\text{eff}}$.</p>

<p>If we had $N$ i.i.d. samples from $p$, the variance of the sample mean would be:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}[\mu] = \mathbb{V}\!\left[\frac{1}{N}\sum_{i=1}^{N} f(X_i)\right] = \frac{1}{N^2} \sum_{i=0}^{N} \mathbb{V}[f(X)] = \frac{\sigma_f^2}{N}
\end{align*}
$$
</div>

<p>where $\sigma_f^2 = \text{Var}_p[f(X)]$.</p>

<p>But alas, we don’t have access to samples from $p$ because if we had, I wouldn’t have to write this blog post explaining ESS.</p>

<p>What we do have is our estimator $\hat{\mu}_{\text{SN}}$,</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\hat{\mu}_{\text{SN}} 
&amp;= \frac{\sum_{i=1}^{N} \tilde{w}(X_i) \ f(X_i)}{\sum_{j=1}^{N} \tilde{w}(X_j)} \\
\end{align*}
$$
</div>

<p>The variance is easy to set up, but actually a bit tricky to compute analytically as we’re dealing with a ratio of random variables</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}[\hat{\mu}_{\text{SN}}]
&amp;= \mathbb{V}\left[ \frac{\sum_{i=1}^{N} \tilde{w}(X_i) \ f(X_i)}{\sum_{j=1}^{N} \tilde{w}(X_j)} \right]
\end{align*}
$$
</div>

<h2 id="the-delta-method-or-everybodys-favourite-tool-linearization-with-taylor-expansions">The Delta Method (or <em>“Everybody’s Favourite Tool: Linearization with Taylor Expansions”</em>)</h2>

<p>Imagine you have two random variables, say $X$ and $Y$, and you care about the ratio $g(X, Y) = X/Y$. You know the variances and covariance of $X$ and $Y$, but what’s the variance of the ratio? That’s awkward because the ratio is nonlinear — variances don’t just “pass through” nonlinear functions.</p>

<p>The delta method’s trick is simple. All we gonna do is <strong>locally linearize</strong> the function via a first-order Taylor expansion around the means, then compute the variance of that linear approximation. Since variance behaves nicely under linear transforms, you get a clean closed-form result (albeit an only locally valid result due to the linearization).</p>

<p>Let $g(X, Y) = X / Y$.</p>

<p>Let’s start with the one-dimensional case to build intuition. The Taylor expansion of a smooth function $g(X)$ around a point $\mu_X$ is:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
g(X) = \sum_{n=0}^{\infty} \frac{g^{(n)}(\mu_X)}{n!} \cdot (X - \mu_X)^n
\end{align*}
$$
</div>

<p>Now let’s extend this to two dimensions. The exact formula is actually bit complicated for the infinite sum but the practically interesting parts are</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
g(X, Y) &amp;= g(\mu_X, \mu_Y) \\
&amp;\quad + \frac{\partial g}{\partial X}\bigg|_{\mu} (X - \mu_X) + \frac{\partial g}{\partial Y}\bigg|_{\mu} (Y - \mu_Y) + \\
&amp;\quad + \frac{1}{2}\frac{\partial^2 g}{\partial X^2}\bigg|_{\mu} (X - \mu_X)^2 + \frac{\partial^2 g}{\partial X \partial Y}\bigg|_{\mu} (X - \mu_X)(Y - \mu_Y) + \frac{1}{2}\frac{\partial^2 g}{\partial Y^2}\bigg|_{\mu} (Y - \mu_Y)^2 \\
\end{align*}
$$
</div>

<p>Only using the first order yields the truncated version in the form of</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
g(X, Y) &amp;= g(\mu_X, \mu_Y) + \frac{\partial g}{\partial X}\bigg|_{\mu} (X - \mu_X) + \frac{\partial g}{\partial Y}\bigg|_{\mu} (Y - \mu_Y)
\end{align*}
$$
</div>

<p>For the delta method, we keep only the <strong>first-order</strong> term (linearization):</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
g(X, Y) \approx g(\mu_X) + g_X(\mu_X) \cdot (X - \mu_X) + g_Y(\mu_Y) \cdot (Y - \mu_Y)
\end{align*}
$$
</div>

<p>Taking the variance of both sides (and noting that $\mu_X , \mu_Y$ is a constant so $\mathbb{V}[\mu_X] = \mathbb{V}[\mu_Y] = 0$), we have</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}[g(X,Y)] \approx [g_X(\mu_X)]^2 \cdot \mathbb{V}[X] + [g_Y(\mu_Y)]^2 \cdot \mathbb{V}[Y] + 2 \ g_X \cdot g_Y \cdot \mathbb{C}[X,Y]
\end{align*}
$$
</div>

<p>where the last term $\mathbb{C}[X, Y]$ is the covariance between $X$ and $Y$ as they might or might not be correlated.</p>

<p>The required partial derivatives are:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
g_X = \frac{\partial}{\partial X}\frac{X}{Y}\bigg|_{\mu} = \frac{1}{\mu_Y}, \qquad g_Y = \frac{\partial}{\partial Y}\frac{X}{Y}\bigg|_{\mu} = -\frac{\mu_X}{\mu_Y^2}
\end{align*}
$$
</div>

<!-- Now take the variance of the linear approximation:

$$\mathbb{V}\!\left[\frac{X}{Y}\right] \approx g_X^2 \,\mathbb{V}[X] + g_Y^2 \,\mathbb{V}[Y] + 2\, g_X\, g_Y \,\mathbb{C}[X, Y]$$ -->

<p>Substituting:</p>

<div style="overflow-x: auto;">
$$
\mathbb{V}\left[\frac{X}{Y}\right] 
\approx \frac{1}{\mu_Y^2}\left(\mathbb{V}[X] + \frac{\mu_X^2}{\mu_Y^2}\,\mathbb{V}[Y] - 2\frac{\mu_X}{\mu_Y}\,\mathbb{C}[X, Y]\right)
$$
</div>

<p>Naturally, this approximation degrades when $\mu_Y$ is close to zero since we’re effectively dividing by something close to zero or when the distributions have heavy tails.
In ML contexts (e.g., variance of a metric ratio in A/B testing), if our samples are large enough for CLT to kick in, this works extremely well.</p>

<p>For our case, we would have</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\hat{\mu}_{\text{SN}}
= \frac{\sum_{i=1}^{N} \tilde{w}(X_i) \ f(X_i)}{\sum_{j=1}^{N} \tilde{w}(X_j)}
= \frac{\frac{1}{N}\sum_{i=1}^{N} \tilde{w}(X_i) \ f(X_i)}{\frac{1}{N}\sum_{j=1}^{N} \tilde{w}(X_j)} 
\stackrel{N \rightarrow \infty}{=} \frac{\mathbb{E}_q[ \tilde{w}(X_i) \ f(X_i)]}{\mathbb{E}_q[ \tilde{w}(X_j)]} 
= \frac{"X"}{"Y"}
\end{align*}
$$
</div>

<h3 id="linearizing-the-ratio-in-the-variance-estimator">Linearizing the Ratio in the Variance Estimator</h3>

<p>For the linearization to be applicable we need the expected values and the variances of the numerator and denominator in our self-normalizing importance sampler.</p>

<p>First up we need to compute the mean and the variance of numerator and denominator</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{E}_p[f] &amp;= \mu_f \\
\mathbb{E}_q[\tilde{w} \ f] &amp;= \ldots \text{nothing to simplify here} &amp; \quad \leftarrow \mu_X  \\
\mathbb{V}_q[\tilde{w} \ f] &amp;= \mathbb{E}[\tilde{w}^2 \ f^2] - \mathbb{E}[\tilde{w} \ f]^2 &amp; \quad \leftarrow \mathbb{V}[X] \\
\mathbb{E}_q[\tilde{w}] &amp;= \int q \ \tilde{w} \ dx = \int q \ \frac{\tilde{p}}{q} \ dx = \int \tilde{p} \ dx = Z &amp; \quad \leftarrow \mu_Y \\
\mathbb{V}_q[\tilde{w}] &amp;= \mathbb{E}_q[\tilde{w}^2] - \mathbb{E}_q[\tilde{w}]^2 
= \mathbb{E}_q[\tilde{w}^2] - Z^2 &amp; \quad \leftarrow \mathbb{V}[Y]\\
\frac{\mathbb{E}_q[\tilde{w} \ f]}{\mathbb{E}_q[\tilde{w}]} 
&amp;= \frac{\mathbb{E}_q[\tilde{w} \ f]}{Z} 
= \mathbb{E}_q\left[\frac{1}{Z}\tilde{w} \ f\right] \\
&amp;= \mathbb{E}_q\left[\frac{1}{Z}\frac{\tilde{p}}{q} \ f\right] 
= \int q \ \frac{p}{q} \ f \ dx \\
&amp;= \mathbb{E}_p[f] \\
&amp;= \mu_f &amp; \quad \leftarrow \frac{\mu_X}{\mu_Y}\\
\mathbb{C}[\tilde{w} \ f, \tilde{w}] 
&amp;= \mathbb{E}_q[\tilde{w} \ f \ \tilde{w}] - \mathbb{E}_q[f] \mathbb{E}_q[\tilde{w}] \\
&amp;= \mathbb{E}_q[\tilde{w}^2 \ f ] - \mu_f \ Z &amp;\quad \leftarrow \mathbb{C}[X, Y]\\
\end{align*}
$$
</div>

<p>All of the expectations and variances above are with respect to a random variable $X$ and it’s functions $\tilde{w}(X)$ or $f(X)$.
The expectation and variance operator are both linear for a sum of random variables, so we can generically write</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{E}\left[\frac{1}{N}\sum_i X_i'\right] &amp;= \frac{1}{N}\sum_i \mathbb{E}[X_i'] = \mathbb{E}\left[ X'_i\right] \\
\mathbb{V}\left[\frac{1}{N}\sum_i X_i'\right] &amp;= \frac{1}{N^2}\sum_i \mathbb{V}[X_i'] = \frac{1}{N}\mathbb{V}\left[ X'_i\right] \\
\mathbb{C}\left[\frac{1}{N}\sum_i W'_i X_i' \ , \ \frac{1}{N}\sum_i X_i' \right]
&amp;= \mathbb{E}\left[\frac{1}{N}\sum_i W'_i X_i' \ \frac{1}{N}\sum_j W'_j \right] - \mathbb{E}\left[\frac{1}{N}\sum_i X_i'\right] \mathbb{E}\left[\frac{1}{N}\sum_i W'_i X_i'\right] \\
&amp;= \frac{1}{N^2}\left( \sum_i \mathbb{E}[ W'_i X_i' \ W'_i] + \sum_{i \neq j} \mathbb{E}[ W'_i X_i' \ W'_j] \right) - \mathbb{E}[W'_i X_i'] \ \mathbb{E}[W'_i] \\
&amp;= \frac{1}{N^2}\left( N \ \mathbb{E}[ W'_i X_i' \ W'_i] + N(N-1) \ \mathbb{E}[ W'_i X_i'] \ \mathbb{E}[ W'_j] \right) - \mathbb{E}[W'_i X_i'] \ \mathbb{E}[W'_i] \\
&amp;= \frac{1}{N}\left( \mathbb{E}[ W'_i X_i' \ W'_i] + (N-1) \mathbb{E}[ W'_i X_i'] \ \mathbb{E}[ W'_i] \right) - \mathbb{E}[W'_i X_i'] \ \mathbb{E}[W'_i] \\
&amp;= \frac{1}{N}\ \mathbb{E}[ W'_i X_i' \ W'_i] + \left(1-\frac{1}{N}\right) \mathbb{E}[ W'_i X_i'] \ \mathbb{E}[ W'_i] - \mathbb{E}[W'_i X_i'] \ \mathbb{E}[W'_i] \\
&amp;= \frac{1}{N}\ \mathbb{E}[ W'_i X_i' \ W'_i] + \frac{1}{N}\mathbb{E}[ W'_i X_i'] \ \mathbb{E}[ W'_i]
\end{align*}
$$
</div>

<p>For the covariance calculation we’ve exploited the fact that over $N$ samples, we assume that only samples with the same index are correlated, whereas samples with different indices are independent.</p>

<p>Since the terms in $\mathbb{V}[X/Y]$ are all variances, we have to simply add a $1/N$ factor to each term, if we choose to work with the sum instead of the mean. In any case, since the $1/N$ factor can be inserted in both the numerator and denominator, it won’t change the final result anyway.</p>

<p>Putting all of these terms together we get</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}\left[\frac{\sum \tilde{w} \ f}{\sum \tilde{w}}\right]
&amp;= \mathbb{V}\left[\frac{\frac{1}{N}\sum \tilde{w} \ f}{\frac{1}{N}\sum \tilde{w}}\right] \\
&amp;\approx \frac{1}{E[\frac{1}{N}\Sigma \ \tilde{w}]^2} \left( \mathbb{V}\left[\frac{1}{N} \sum \tilde{w} \ f\right] 
+ \frac{\mathbb{E}[\frac{1}{N} \sum \tilde{w} \ f]^2}{\mathbb{E}[\frac{1}{N} \sum \tilde{w}]^2} \mathbb{V}\left[\frac{1}{N} \sum \tilde{w}\right] 
- 2 \frac{\mathbb{E}[\frac{1}{N} \sum \tilde{w} \ f]}{\mathbb{E}[\frac{1}{N} \sum \tilde{w}]} \mathbb{C}\left[\frac{1}{N} \sum \tilde{w}, \frac{1}{N} \sum \tilde{w} \ f\right] \right)\\
&amp;= \frac{1}{E[ \tilde{w}]^2} \frac{1}{N} \left( \mathbb{V}\left[ \tilde{w} \ f\right] 
+ \frac{\mathbb{E}[\tilde{w} \ f]^2}{\mathbb{E}[\tilde{w}]^2} \mathbb{V}\left[\tilde{w}\right] 
- 2 \frac{\mathbb{E}[\tilde{w} \ f]}{\mathbb{E}[\tilde{w}]} \mathbb{C}\left[\tilde{w},\tilde{w} \ f\right] \right)\\
&amp;= \overbrace{\frac{1}{Z^2}}^{\frac{1}{\mu_Y^2}}\frac{1}{N} \Big( \overbrace{\mathbb{E}[\tilde{w}^2 \ f^2] - \mathbb{E}[\tilde{w} \ f]^2}^{\mathbb{V}[X]} + \underbrace{\mathbb{E}_p[f]^2}_{\frac{\mu_X^2}{\mu_Y^2}} \big(\underbrace{\mathbb{E}_q[\tilde{w}^2] - Z^2}_{\mathbb{V}[Y]} \big) - 2 \ \underbrace{\mathbb{E}_p[f]}_{\frac{\mu_X}{\mu_Y}} \big(\underbrace{\mathbb{E}_q[\tilde{w}^2 \ f] -  \mathbb{E}_q[\tilde{w}] Z}_{\mathbb{C}[X, Y]} \big)\Big) \\
&amp;= \frac{1}{Z^2}\frac{1}{N} \Big( \mathbb{E}[\tilde{w}^2 \ f^2] - \underbrace{\frac{Z^2}{Z^2} \mathbb{E}[\tilde{w} \ f]^2}_{=Z^2 \ \mathbb{E}_p[f]} 
+ \mathbb{E}_p[f]^2 \big(\mathbb{E}_q[\tilde{w}^2] - Z^2\big) 
- 2 \ \mathbb{E}_p[f] \big(\mathbb{E}_q[\tilde{w}^2 \ f] - \underbrace{\frac{Z}{Z} \mathbb{E}_q[\tilde{w}] Z}_{=Z^2 \mathbb{E}_p[f]}\big)\Big) \\
&amp;= \frac{1}{Z^2}\frac{1}{N} \Big( \mathbb{E}[\tilde{w}^2 \ f^2] \ \cancel{- Z^2 \ \mathbb{E}_p[f]} 
+ \mathbb{E}_p[f]^2 \big(\mathbb{E}_q[\tilde{w}^2] \ \cancel{- Z^2} \big) 
- 2 \ \mathbb{E}_p[f] \big(\mathbb{E}_q[\tilde{w}^2 \ f] \cancel{- Z^2 \mathbb{E}_p[f]}\big)\Big) \\
&amp;= \frac{1}{Z^2}\frac{1}{N} \Big( \mathbb{E}[\tilde{w}^2 \ f^2] 
+ \mu_f^2 \ \mathbb{E}_q[\tilde{w}^2] 
- 2 \ \mu_f \ \mathbb{E}_q[\tilde{w}^2 \ f] \Big) \\
&amp;= \frac{1}{Z^2}\frac{1}{N} \Big( \mathbb{E}\left[\tilde{w}^2 \ f^2 + \mu_f^2 \ \tilde{w}^2 - 2 \ \mu_f \ \tilde{w}^2 \ f\right] \Big) \\
&amp;= \frac{1}{Z^2}\frac{1}{N} \Big( \mathbb{E}\left[\tilde{w}^2 \ \left( f^2 + \mu_f^2 - 2 \ \mu_f \ f\right)\right] \Big) \\
&amp;= \frac{1}{Z^2}\frac{1}{N} \Big( \mathbb{E}\left[\tilde{w}^2 \ \left( f - \mu_f\right)^2\right] \Big) \\
&amp;= \frac{1}{N} \frac{\mathbb{E}\left[\tilde{w}^2 \ \left( f - \mu_f\right)^2\right]}{\mathbb{E}[\tilde{w}]^2} \\
\end{align*}
$$
</div>

<p>The equation above is an approximation of the variance of the self-normalized estimator.
If we sample the samples from a proposal distribution $q$ that is different from the unnormalized target distribution $\tilde{p}$ resulting in the weights $\tilde{w}$.</p>

<p>But what if we wouldn’t sample from $q$ but from the true target distribution $p$ resulting in the weights $w$?
This implies that we can really generate $N_p$ samples from $p$ directly, and thus our samples will be i.i.d. from $p$, not suffering from the approximation error of $q$ to $p$.
Let’s calculate the variance of the self-normalized estimator if we were to sample from $p$ directly,</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}_p\left[\frac{\sum \tilde{w} \ f}{\sum \tilde{w}}\right]
&amp;= \frac{1}{N_p} \frac{\mathbb{E}_p\left[\tilde{w}^2 \ \left( f - \mu_f\right)^2\right]}{\mathbb{E}_p[\tilde{w}]^2} \\
&amp;= \frac{1}{N_p} \frac{\mathbb{E}_p\left[\tilde{w}^2 \ \left( f - \mu_f\right)^2\right]}{\underbrace{\int p \frac{\tilde{p}}{p}\ dx^2}_{=Z^2}} \\
&amp;= \frac{1}{N_p} \mathbb{E}_p\left[\frac{1}{Z^2}\tilde{w}^2 \ \left( f - \mu_f\right)^2\right]\\
&amp;= \frac{1}{N_p} \mathbb{E}_p\left[\frac{1}{Z^2}\frac{\tilde{p}^2}{p^2} \ \left( f - \mu_f\right)^2\right]\\
&amp;= \frac{1}{N_p} \mathbb{E}_p\left[ \ \left( f - \mu_f\right)^2\right]\\
\end{align*}
$$
</div>

<p>And thus we can see that the variance is actually decreasing with the number of samples $N_p$ from $p$ directly as we would expect from a standard i.i.d. estimator.
If $N_p$ is large, we consequently decrease the inherent variance of the estimator by a factor of $N_p$.</p>

<p>But again, we can only sample from $q$, not from $p$.
But we can pose a clever question: How many samples $N_q$ from $q$ would we need to get the same variance as $N_p$ samples from $p$?
At the very beginning we said that our correlated samples from $q$ carry less information than $N_p$ truly independent samples would.
So the ESS metric asks the question how many samples $N_q$ from $q$ are worth $N_p$ perfect, i.i.d, super-duper-excellent samples from $p$?</p>

<p>In order to solidify this, we’ll set the two variance estimators equal to each other, one with $N_p$ samples and one with $N_q$ samples.
\(\begin{align*}
\mathbb{V}_p\left[\frac{\sum \tilde{w} \ f}{\sum \tilde{w}}\right]
&amp;= \mathbb{V}_q\left[\frac{\sum \tilde{w} \ f}{\sum \tilde{w}}\right] \\
\frac{1}{N_p} \mathbb{E}_p\left[ \ \left( f - \mu_f\right)^2\right] 
&amp;= \frac{1}{N_q} \frac{\mathbb{E}_q\left[ \tilde{w}^2 \ \left( f - \mu_f\right)^2\right]}{\mathbb{E}_q[\tilde{w}]^2} \\
\end{align*}\)</p>

<p>In order to arrive at the ESS formula, we’ll need to do a final approximation by assuming that the weights $\tilde{w}^2$ actually aren’t too correlated with $(f - \mu_f)^2$ which is in practice actually somewhat true,</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\frac{1}{N_p} \mathbb{E}_p\left[ \ \left( f - \mu_f\right)^2\right] 
&amp;\approx \frac{1}{N_q} \frac{\mathbb{E}_q\left[ \tilde{w}^2 \right] \ \mathbb{E}_p \left[ \left( f - \mu_f\right)^2\right]}{\mathbb{E}_q[\tilde{w}]^2} \\
\frac{1}{N_p} 
&amp;\approx \frac{1}{N_q} \frac{\mathbb{E}_q\left[ \tilde{w}^2 \right] }{\mathbb{E}_q[\tilde{w}]^2} \\
N_q 
&amp;\approx N_p \cdot \frac{\mathbb{E}_q[\tilde{w}]^2}{\mathbb{E}_q\left[ \tilde{w}^2 \right] } \\
\end{align*}
$$
</div>

<p>The samples $N_q$ are then the effective number of samples from $q$ that are worth $N_p$ samples from $p$.
The ratio $\frac{\mathbb{E}_q\left[ \tilde{w}^2 \right] }{\mathbb{E}_q[\tilde{w}]^2}$ can intuitively be interpreted as a discount factor.
You can quite literally think of the discount factor like a conversion rate between the number of samples from $q$ and the number of samples from $p$ that are worth the same.
This is really akin to conversion factors between currencies.
Take $N_q$ Zimbabwean dollars and try to convert them to $N_p$ US dollars. Since Zimbabwean dollars <a href="https://en.wikipedia.org/wiki/Hyperinflation_in_Zimbabwe">aren’t worth anything</a>, you got like 0.000001 Dollar per Zimbabwean dollar.
So you need an inordinate amount of Zimbabwean dollars to get a single $N_p=1$ US dollar.</p>

<p>We can also express the effective sampling size in terms of normalized weights $\tilde{w}$ instead of $\tilde{w}^2$.
Remember that</p>

<div style="overflow-x: auto;">
$$\begin{align*}
\frac{1}{N} \sum_{i=1}^{N} \tilde{w}_i &amp;= \mathbb{E}_q[\tilde{w}] = \int q \ \tilde{w} \ dx = \int q \ \frac{\tilde{p}}{q} \ dx = \int \tilde{p} \ dx = Z \\
&amp; \downarrow \\
\sum_{i=1}^{N} \tilde{w}_i &amp;= N \ Z
\end{align*}
$$
</div>

<div style="overflow-x: auto;">
$$
\begin{align*}
N_q 
&amp;\approx N \ \frac{\mathbb{E}_q[\tilde{w}]^2}{\mathbb{E}_q\left[ \tilde{w}^2 \right] } \\
&amp;\approx N \ \frac{(\frac{1}{N} \sum_i \tilde{w}_i)^2}{\frac{1}{N} \sum_i \tilde{w}_i^2} \\
&amp;\approx N^2 \ \frac{\frac{1}{N^2} (\sum_i \tilde{w}_i)^2}{\sum_i \tilde{w}_i^2} \\
&amp;= \frac{(\sum_i \tilde{w}_i)^2}{\sum_i \tilde{w}_i^2} \\
&amp;= \frac{1}{\sum_i \frac{1}{(\sum_j^N \tilde{w}_j)^2}  \tilde{w}_i^2}  &amp; \quad \quad \text{with} \ \bar{w}_i = \frac{\tilde{w}_i}{\sum_j^N \tilde{w}_j} \\
&amp;= \frac{1}{\sum_i \bar{w}_i^2} \\
\end{align*}
$$
</div>

<p>This gives us an easy way to calculate the ESS purely from the normalized weights $\tilde{w}_i$.</p>

<h3 id="mcmc">MCMC</h3>

<p>In the importance sampling case, the problem was that we couldn’t sample from $p$ and had to use a proposal $q$, leading to weights that could be wildly uneven.
MCMC flips the script entirely: we <em>are</em> sampling from $p$ (at least asymptotically), but our samples are <strong>serially correlated</strong> because each sample depends on the previous one.</p>

<p>Think of it this way: if importance sampling suffers from “some samples mattering way more than others”, MCMC suffers from “consecutive samples being almost the same thing”.
You’re walking through the mountainous landscape we described at the very beginning, but you can only take small steps from where you currently are.
If you’re standing on a peak, your next few hundred positions will all be near that same peak — you haven’t explored the valleys yet.
Your chain might eventually visit the entire landscape, but <em>locally</em>, your samples are redundant.</p>

<p>So the question is the same one we’ve been asking all along: <strong>how many i.i.d. samples is our correlated chain actually worth?</strong></p>

<p>Let’s set up the math. Denote the sample mean of our MCMC chain as</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\bar{f}_N = \frac{1}{N}\sum_{t=1}^{N} f(X_t)
\end{align*}
$$
</div>

<p>where $X_1, X_2, \ldots, X_N$ are the states of our Markov chain targeting $p$.</p>

<p>We want $\mathbb{V}[\bar{f}_N]$. Let’s expand this directly from the definition:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}[\bar{f}_N] = \mathbb{V}\!\left[\frac{1}{N}\sum_{t=1}^N f(X_t)\right] = \frac{1}{N^2}\,\mathbb{V}\!\left[\sum_{t=1}^N f(X_t)\right]
\end{align*}
$$
</div>

<p>Now here’s where the correlation makes life interesting.
What we’ve done is run a MCMC chain which always takes the last value $X_{t-1}$ as the starting point for the next step $X_t$.
So for all intents and purposes, the samples $X_t$ are correlated.
The variance of a sum is <em>not</em> just the sum of the variances when the summands are correlated. We have to account for all the cross-terms:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}\!\left[\sum_{t=1}^N f(X_t)\right] = \sum_{t=1}^N \sum_{s=1}^N \mathbb{C}[f(X_t), f(X_s)] = \underbrace{\sum_{t=1}^N \mathbb{V}[f(X_t)]}_{\text{diagonal } (t=s)} + \underbrace{\sum_{t=1}^N \sum_{\substack{s=1 \\ s \neq t}}^N \mathbb{C}[f(X_t), f(X_s)]}_{\text{off-diagonal } (t \neq s)}
\end{align*}
$$
</div>

<p>This is a double sum — every pair of samples $(t, s)$ contributes a covariance term. The diagonal terms ($t = s$) give us the familiar sum of variances, while the off-diagonal terms ($t \neq s$) capture the correlations between different samples. For i.i.d. samples, all the off-diagonal terms vanish because $\mathbb{C}[f(X_t), f(X_s)] = 0$ for $t \neq s$, and we’d recover</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}\!\left[\frac{1}{N}\sum_{t=1}^N f(X_t)\right] = \frac{1}{N^2}\sum_{t=1}^N \mathbb{V}[f(X_t)] = \frac{N\sigma_f^2}{N^2} = \frac{\sigma_f^2}{N}
\end{align*}
$$
</div>

<p>But for MCMC, they don’t vanish. Consecutive samples are correlated, and that correlation bleeds into the variance.</p>

<p>The saving grace is <strong>stationarity</strong>: if you run the chain long enough after which the chain is said to have “converged”, the covariance between $f(X_t)$ and $f(X_s)$ depends only on the lag $k = |t - s|$, not on the absolute positions $t$ and $s$.
With stationarity, this allows us define the <strong>autocovariance function</strong>:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
c_k = \mathbb{C}[f(X_t), f(X_{t+k})] = \mathbb{E}[(f(X_t) - \mu_f)(f(X_{t+k}) - \mu_f)]
\end{align*}
$$
</div>

<p>Note that $c_0 = \sigma_f^2$ (the marginal variance of $f$ under $p$, the same $\sigma_f^2$ we’ve been carrying around since the beginning) and $c_k = c_{-k}$ by symmetry — the covariance doesn’t care about the direction of time.</p>

<p>In the double sum $\sum_{t=1}^N \sum_{s=1}^N$, we need to count how many pairs $(t, s)$ produce each lag $k = t - s$. This is a straightforward combinatorial exercise:</p>

<ul>
  <li>Lag $k = 0$: pairs $(1,1), (2,2), \ldots, (N,N)$ → exactly $N$ terms</li>
  <li>Lag $k = 1$: pairs $(2,1), (3,2), \ldots, (N, N-1)$ → exactly $N-1$ terms</li>
  <li>Lag $k = -1$: pairs $(1,2), (2,3), \ldots, (N-1, N)$ → exactly $N-1$ terms</li>
  <li>In general, lag $k$: exactly $N - |k|$ terms, for $k = -(N-1), \ldots, N-1$</li>
</ul>

<p>This allows us to simplify the equation as for any fixed $k=t-s$, there are $N - |k|$ pairs $(t, s)$ that produce this lag with a fixed value of $c_k$ for the stationary/converged chain.
So the double sum collapses to</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\sum_{t=1}^N \sum_{s=1}^N \mathbb{C}[f(X_t), f(X_s)] = \sum_{k=-(N-1)}^{N-1} (N - |k|)\,c_k
\end{align*}
$$
</div>

<p>Using the symmetry $c_k = c_{-k}$, we fold the negative lags into the positive ones:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\sum_{t=1}^N \sum_{s=1}^N \mathbb{C}[f(X_t), f(X_s)]= Nc_0 + 2\sum_{k=1}^{N-1}(N-k)\,c_k
\end{align*}
$$
</div>

<p>Plugging this back into our variance expression:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}[\bar{f}_N] &amp;= \frac{1}{N^2}\left[Nc_0 + 2\sum_{k=1}^{N-1}(N-k)\,c_k\right] \\
&amp;= \frac{c_0}{N} + \frac{2}{N^2}\sum_{k=1}^{N-1}(N-k)\,c_k \\
&amp;= \frac{c_0}{N}\left[1 + 2\sum_{k=1}^{N-1}\underbrace{\left(1 - \frac{k}{N}\right)}_{\text{Bartlett weight}}\frac{c_k}{c_0}\right]
\end{align*}
$$
</div>

<p>Defining the <strong>autocorrelation function</strong> $\rho_k = c_k / c_0$ (the normalized version of the autocovariance), we arrive at</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}[\bar{f}_N] = \frac{\sigma_f^2}{N}\left[1 + 2\sum_{k=1}^{N-1}\left(1 - \frac{k}{N}\right)\rho_k\right]
\end{align*}
$$
</div>

<p>For i.i.d. samples, $\rho_k = 0$ for all $k \geq 1$, and the bracket collapses to $1$, recovering the familiar $\sigma_f^2/N$ we know and love.</p>

<p>The Bartlett weights $(1 - k/N)$ are worth pausing on. They arise purely from counting: at lag $k$, there are only $N - k$ pairs of samples separated by exactly $k$ steps. The further apart two samples are in the chain, the fewer such pairs exist. For large $N$, these weights are approximately $1$ for any fixed lag — the edge effects wash out.</p>

<p>Thus for large $N$, the Bartlett weights $(1 - k/N) \to 1$ for any fixed $k$. If the autocorrelations decay fast enough that $\sum_{k=1}^\infty \|\rho_k\| &lt; \infty$ — which holds for any geometrically ergodic chain, i.e. pretty much every well-behaved MCMC sampler you’ll encounter in practice — we can take the limit:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}[\bar{f}_N] \approx \frac{\sigma_f^2}{N}\left[1 + 2\sum_{k=1}^{\infty}\rho_k\right] = \frac{\sigma_f^2}{N}\cdot\tau_f
\end{align*}
$$
</div>

<p>where we’ve defined the <strong>integrated autocorrelation time (IAT)</strong>:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\tau_f = 1 + 2\sum_{k=1}^{\infty} \rho_k
\end{align*}
$$
</div>

<p>The name “integrated” comes from viewing this as the discrete analogue of integrating the continuous autocorrelation function. It collapses the entire memory structure of the chain into a single number.</p>

<p><strong>What does $\tau_f$ tell us?</strong></p>

<ul>
  <li>$\tau_f = 1$: no autocorrelation at all, we’re back in i.i.d. land because all $\rho_k = 0$ for $k \geq 1$</li>
  <li>$\tau_f = 10$: each sample carries only $1/10$-th the information of an independent draw</li>
  <li>$\tau_f = 100$: the chain is mixing like molasses — you need 100 steps to get one effective independent sample</li>
</ul>

<p>One subtlety worth flagging: notice the subscript $f$ on $\tau_f$. Unlike the importance sampling ESS we derived above — which is purely a function of the weights and therefore the same no matter what $f$ you’re estimating — the MCMC IAT is <strong>function-dependent</strong>. A chain might mix quickly for the mean of $f$ but slowly for its variance, or zip along in one direction of parameter space while crawling in another. This is a fundamental difference between the two flavors of ESS.</p>

<p>We now play exactly the same game as in the importance sampling case. We ask: how many i.i.d. samples $N_{\text{eff}}$ from $p$ would give us the same variance as our $N$ correlated MCMC samples?</p>

<p>The i.i.d. baseline is, as before:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{V}_{\text{i.i.d.}} = \frac{\sigma_f^2}{N_{\text{eff}}}
\end{align*}
$$
</div>

<p>Setting this equal to the MCMC variance:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\frac{\sigma_f^2}{N_{\text{eff}}} = \frac{\sigma_f^2}{N}\cdot \tau_f
\end{align*}
$$
</div>

<p>The $\sigma_f^2$ cancels — though this is a bit misleading, since $\tau_f$ itself depends on $f$, so the ESS is still function-dependent unlike the IS case:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
N_{\text{eff}} = \frac{N}{\tau_f} = \frac{N}{1 + 2\sum_{k=1}^{\infty}\rho_k}
\end{align*}
$$
</div>

<p>The logic is beautifully parallel to what we did for importance sampling. There, the “discount factor” was the weight variability $\mathbb{E}_q[\tilde{w}^2] / \mathbb{E}_q[\tilde{w}]^2$. Here, the discount factor is the integrated autocorrelation time $\tau_f$. Both tell you how many of your $N$ samples you’re <em>actually</em> using.</p>

<p>Going back to our currency analogy: if importance sampling was converting Zimbabwean dollars to US dollars (with the exchange rate determined by weight concentration), then MCMC is like converting a time series of correlated price observations to independent ones (with the exchange rate determined by how sticky the prices are). Either way, you’re asking: what’s my real purchasing power?</p>

<h3 id="diffusion-models-as-sequential-monte-carlo">Diffusion Models as Sequential Monte Carlo</h3>

<p>We’ve now built up the entire ESS toolkit: for importance sampling, it measures weight concentration; for MCMC, it measures autocorrelation-induced information loss. These two stories converge in one of the most consequential generative modeling frameworks of the last few years — <strong>diffusion models</strong> — where the denoising process can be interpreted as a sequential Monte Carlo sampler.</p>

<p>But before we get to diffusion models, let’s set up the general SMC framework. The whole point of SMC is to do inference in <strong>state-space models</strong> (also known as hidden Markov models in the discrete case), where we have:</p>

<ul>
  <li>A sequence of <strong>hidden states</strong> $x_0, x_1, \ldots, x_T$ that we can’t observe directly</li>
  <li>A <strong>prior</strong> over the initial state: $p(x_0)$</li>
  <li>A <strong>transition kernel</strong> $p(x_t | x_{t-1})$ describing how the hidden state evolves over time</li>
  <li>An <strong>emission model</strong> (also called the observation model) $p(y_t | x_t)$ describing how the observed data $y_t$ is generated from the hidden state</li>
</ul>

<p>The joint distribution over the full trajectory and observations factorizes as</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
p(x_{0:T}, y_{1:T}) = p(x_0) \prod_{t=1}^{T} p(x_t | x_{t-1}) \cdot \prod_{t=1}^{T} p(y_t | x_t)
\end{align*}
$$
</div>

<p>The goal is to compute the <strong>filtering distribution</strong> $p(x_t | y_{1:t})$ — what do we believe about the current hidden state, given everything we’ve observed so far? — or, more ambitiously, the <strong>smoothing distribution</strong> $p(x_{0:T} | y_{1:T})$ over the entire trajectory.</p>

<p>The emission probability $p(y_t | x_t)$ is the likelihood of the observed data $y_t$ given the hidden state $x_t$. 
In effect, it works as a critic of the trajectory $x_{0:t}$ that was sampled so far. 
At each step (but also maybe only at other steps) it evaluates the probability $p(y_t | x_{0:t})$ of the observed data $y_t$ given the trajectory $x_{0:t}$.
If you did a lot of steps that moved your samples $x_t$ into some part of space where the data $y_t$ is very unlikely, your weight will be low, and thus that particular sample path $x_{0:t}$ will be highly discouraged.</p>

<p>The trouble is that the corresponding analytical posteriors are intractable in general. The exact recursion involves integrating over all possible previous states at every step, and this integral doesn’t have a closed form except in special cases (linear-Gaussian dynamics give you the Kalman filter; finite state spaces give you the forward-backward algorithm).</p>

<p><strong>Sequential Monte Carlo</strong> approximates these intractable distributions using a weighted set of <strong>particles</strong> — sample trajectories ${x_{0:t}^{(i)}, w_t^{(i)}}_{i=1}^N$ that represent an empirical approximation to the posterior. The algorithm proceeds sequentially:</p>

<ol>
  <li><strong>Propagate</strong>: Move each particle forward in time by sampling from the transition kernel $x_t^{(i)} \sim p(x_t | x_{t-1}^{(i)})$</li>
  <li><strong>Weight</strong>: Assign each particle an importance weight based on the emission model $\tilde{w}_t^{(i)} = p(y_t | x_t^{(i)})$</li>
  <li><strong>Normalize</strong>: Compute $\bar{w}_t^{(i)} = \tilde{w}_t^{(i)} / \sum_j \tilde{w}_t^{(j)}$</li>
  <li><strong>Evaluate</strong>: Compute the ESS $\hat{N}_{\text{eff}} = 1 / \sum_i (\bar{w}_t^{(i)})^2$</li>
  <li><strong>Resample</strong>: If the ESS drops below a threshold, resample particles according to the weights — duplicate the promising ones, discard the dead weight — and reset all weights to $1/N$</li>
</ol>

<p>Essentially, you keep sampling trajectories and you keep using the “critic” to evaluate them by checking what the probability is that the trajectory would have produced the data $y_t$ that you observed.
This information is encapsulated in the weights $w_t^{(i)}$ that you assign to each trajectory.
If the weights become too low, this implies that your trajectories have become more and more unlikely for the “critic” $p(y_t | x_{0:t})$.
What do you do? You look at your trajectories and you decide to keep only the ones that are still likely, and you discard the ones that are not.
The normalized weights $\bar{w}_t^{(i)}$ can then be used as the probabilities for a categorical distribution that you use to sample new trajectories.</p>

<p>Now — what does any of this have to do with diffusion models?</p>

<p>For diffusion models, we have the generative <a href="https://ludwigwinkler.github.io/blog/SimpleReverseSDE/"><strong>reverse-time SDE</strong></a>:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
dx_t = \left[f(x_t, t) - g^2(t)\, \nabla_{x_t} \log p_t(x_t)\right] dt + g(t)\, dW_t
\end{align*}
$$
</div>

<p>where $W_t$ is a reverse-time Wiener process and $\nabla_{x_t} \log p_t(x_t)$ is the <strong>score function</strong> — the gradient of the log-density at time $t$. This is the quantity a diffusion model learns: a neural network $s_\theta(x_t, t) \approx \nabla_{x_t} \log p_t(x_t)$.</p>

<p>To actually generate samples, we discretize the reverse SDE with an Euler-Maruyama step of size $\Delta t$:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
x_{t - \Delta t} = x_t + \left[f(x_t, t) - g^2(t)\, s_\theta(x_t, t)\right] \Delta t + g(t)\sqrt{\Delta t}\, \epsilon, \quad \epsilon \sim \mathcal{N}(0, I)
\end{align*}
$$
</div>

<p>This discretization <em>is</em> our <strong>transition kernel</strong> — reading it off as a Gaussian:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
p_\theta(x_{t-\Delta t} | x_t) = \mathcal{N}\!\left(x_{t-\Delta t};\, \underbrace{x_t + [f(x_t, t) - g^2(t)\, s_\theta(x_t, t)]\,\Delta t}_{\text{mean}},\, \underbrace{g^2(t)\,\Delta t}_{\text{variance}}\, I\right)
\end{align*}
$$
</div>

<p>This is the transition kernel that enters the SMC framework. Unconditional generation is then straightforward: sample $x_T \sim \mathcal{N}(0, I)$, apply $p_\theta(x_{t-\Delta t} | x_t)$ for each step backwards, and out pops a sample $x_0 \approx p_{\text{data}}$. This is just a Markov chain — no weights, no resampling, no ESS to speak of.</p>

<h4 id="the-one-step-denoiser-tweedies-formula">The one-step denoiser: Tweedie’s formula</h4>

<p>Before we get to SMC, we need one more ingredient: the “critic” $p(y_t | x_t)$. At any noise level $t$, the diffusion model implicitly gives us a <strong>one-step denoiser</strong> — a direct estimate of what the clean data $x_0$ looks like, given only the current noisy state $x_t$.</p>

<p>This comes from <strong>Tweedie’s formula</strong>. Given the forward marginal $q(x_t | x_0) = \mathcal{N}(x_t;\, \alpha(t)\, x_0,\, \sigma^2(t)\, I)$, the posterior mean of $x_0$ given $x_t$ is:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\hat{x}_0(x_t) = \mathbb{E}[x_0 | x_t] = \frac{1}{\alpha(t)}\left(x_t + \sigma^2(t)\, \nabla_{x_t} \log p_t(x_t)\right)
\end{align*}
$$
</div>

<p>Since the diffusion model learns the score $s_\theta(x_t, t) \approx \nabla_{x_t} \log p_t(x_t)$, we get $\hat{x}_0(x_t)$ essentially for free — it’s just a different way of reading off the network output.</p>

<p>Think of $\hat{x}_0(x_t)$ as the model’s best guess of the clean image, peeking through the noise at step $t$. Early in the denoising process (large $t$, lots of noise), this guess is blurry and uncertain. Late in the process (small $t$, little noise), it’s sharp and confident.</p>

<p>We’ll call this one-step denoiser prediction the “observation” that the current noisy state $x_t$ produces — and denote the observation model as $p(y_t | x_t)$, where $y_t$ encodes what we can infer about the data from $x_t$.</p>

<h4 id="conditional-generation-enter-the-state-space-model">Conditional generation: enter the state-space model</h4>

<p>Now suppose we don’t just want any sample from $p_{\text{data}}$ — we want a sample that satisfies some condition $y$. Maybe we want an image of a golden retriever, or we want to inpaint a missing region, or we want a molecule with certain properties.</p>

<p>What we want is to sample from the <strong>conditional</strong> reverse process:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
p(x_{0:T} | y) \propto p(x_T) \prod_{t=1}^{T} p_\theta(x_{t-1} | x_t) \cdot p(y | x_0)
\end{align*}
$$
</div>

<p>The problem? The likelihood $p(y | x_0)$ only tells us about the final clean sample $x_0$. We don’t get any signal about whether our denoising trajectory is heading in the right direction until the very end. By then, it’s too late — all our computational budget is spent.</p>

<p>This is where the one-step denoiser saves the day. Instead of waiting until $x_0$ to evaluate $p(y | x_0)$, we can get an <em>approximate</em> per-step likelihood using the Tweedie estimate:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
p(y_t \| x_t) \approx p(y \| \hat{x}_0(x_t))
\end{align*}
$$
</div>

<p>We’re peeking at the model’s current best guess of the clean output and asking: “does this look like it satisfies the condition?”</p>

<p>And now we have all the pieces to write down a <strong>state-space model</strong> — exactly the structure we introduced in the general SMC framework above:</p>

<ul>
  <li><strong>Hidden states</strong> $x_T, x_{T-\Delta t}, \ldots, x_0$ — the noisy latents at each denoising step</li>
  <li><strong>Prior</strong> $p(x_T) = \mathcal{N}(0, I)$ — pure noise at the start of the reverse process</li>
  <li><strong>Transition kernel</strong> $p_\theta(x_{t-\Delta t} \| x_t)$ — the Euler-Maruyama discretization of the reverse SDE</li>
  <li><strong>Emission model</strong> $p(y_t | x_t) \approx p(y | \hat{x}_0(x_t))$ — how consistent is the current state with the desired condition, as estimated through the one-step denoiser</li>
</ul>

<p>This is <em>exactly</em> the structure we need for a particle filter.</p>

<p>Combining everything we get the following algorithm:</p>

<p><strong>Initialize:</strong> Draw $N$ particles from the prior (pure noise):</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
x_T^{(i)} \sim \mathcal{N}(0, I), \quad i = 1, \ldots, N
\end{align*}
$$
</div>

<p><strong>For each denoising step $t = T, T-\Delta t, \ldots, \Delta t$:</strong></p>

<p><strong>1. Propagate</strong> each particle through the transition kernel (one reverse SDE step):</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
x_{t-\Delta t}^{(i)} \sim p_\theta(x_{t-\Delta t} | x_t^{(i)})
\end{align*}
$$
</div>

<p><strong>2. Compute the unnormalized importance weights</strong> via the observation model:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\tilde{w}_t^{(i)} = p(y_t | x_t^{(i)}) = p(y | \hat{x}_0(x_t^{(i)}))
\end{align*}
$$
</div>

<p>The weight of each particle reflects how well its one-step denoiser prediction $\hat{x}_0(x_t^{(i)})$ matches the desired condition $y$.</p>

<p><strong>3. Normalize the weights:</strong></p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\bar{w}_t^{(i)} = \frac{\tilde{w}_t^{(i)}}{\sum_{j=1}^N \tilde{w}_t^{(j)}}
\end{align*}
$$
</div>

<p>Look familiar? These are exactly the self-normalized weights $\bar{w}_i = \tilde{w}_i / \sum_j \tilde{w}_j$ we spent the first half of this post deriving.</p>

<p><strong>4. Compute the ESS</strong> using the formula we derived:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\hat{N}_{\text{eff}} = \frac{1}{\sum_{i=1}^{N} (\bar{w}_t^{(i)})^2}
\end{align*}
$$
</div>

<p><strong>5. Resample if necessary:</strong> if $\hat{N}_{\text{eff}}$ drops below some threshold (a common choice is $N/2$), resample the particles according to the weights — duplicate the high-weight particles and discard the low-weight ones. After resampling, reset all weights to $1/N$ and continue denoising from the refreshed particle cloud.</p>

<h4 id="why-ess-matters-here-the-weight-degeneracy-problem">Why ESS matters here: the weight degeneracy problem</h4>

<p>And here’s where our entire ESS derivation pays off.</p>

<p>Remember the Dutch giants and Greek grandmas? In the diffusion context, the particles are like our text message recipients. Each particle $x_t^{(i)}$ is a different “hypothesis” for what the clean image looks like. Some particles, by chance, will be heading toward images that satisfy the condition $y$ — samples that score highly under the observation model. Others will be heading toward something completely different: samples that bear no resemblance to the desired condition, effectively dead weight in the particle cloud.</p>

<p>The weights $\tilde{w}_t^{(i)} = p(y | \hat{x}_0(x_t^{(i)}))$ quantify this mismatch: particles whose one-step denoiser prediction aligns well with the condition $y$ get large weights, while those drifting toward irrelevant regions of the sample space get negligible weights. If only 2 out of 100 particles are heading in the right direction, the ESS will be approximately 2 — just like our Dutch-and-Greek height survey.</p>

<p>Without resampling, the particle cloud degenerates: a handful of particles carry all the weight, and you’re effectively running the denoising process with just those few trajectories. The estimator variance explodes, and the quality of your conditional samples plummets.</p>

<p>Resampling — triggered when the ESS drops below a threshold — is the remedy. It prunes the dead-weight particles and replicates the promising ones, refreshing the particle cloud. But the price is that resampled particles are no longer independent (they share common ancestors), introducing the autocorrelation that we analyzed in the MCMC section. So we’re back to our old friend: correlated samples carrying less information than independent ones.</p>

<p>This is the fundamental tension of SMC: <strong>resampling fights weight degeneracy but creates sample correlation</strong>. The ESS, in both its importance-sampling and MCMC flavors, is the diagnostic that navigates this tradeoff.</p>

<h4 id="deriving-the-incremental-weights">Deriving the incremental weights</h4>

<p>So far we have the transition probabilities $p_\theta(x_{t-\Delta t} | x_t)$ — the reverse SDE that propagates our particles from noise toward data. But transitions alone just tell us <em>how</em> particles move; they say nothing about <em>which</em> particles are worth keeping. For that, we need the emission probabilities $p(y_t | x_t)$.</p>

<p>The idea is simple: we define a function that takes in a particle $x_t$ and returns a scalar value evaluating how well that particle’s trajectory aligns with the desired observation $y$. In the diffusion setting, we use the one-step denoiser to peek ahead: $p(y_t | x_t) \approx p(y | \hat{x}_0(x_t))$. This is our critic — it scores each particle at time $t$ by asking “if I were to denoise this particle all the way to $x_0$ right now, how well would the result match the condition $y$?”.
In the literature, this is often called the “potential function” and denoted by the function $G_t(x_t)$.</p>

<p>The whole point of tracking importance weights is to keep score of which samples are good and which are bad, so that we can resample accordingly — duplicate the promising trajectories and prune the hopeless ones. And for that resampling decision to be well-informed, we need the most up-to-date emission probability at every time step $t$, not just at the very end.</p>

<p>We can achieve exactly this with a telescoping product. Define the <strong>per-step potential function</strong> $G_t(x_t)$:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
G_t(x_t) = \frac{p(\hat{x}_0 \| x_t)}{p(\hat{x}_0 \| x_{t+\Delta t})}
\end{align*}
$$
</div>

<p>This is the ratio of the current emission score to the previous one — how much our assessment of the particle has changed in a single denoising step. Remember, we’re going from $T \to 0$, so the previous step is $t + \Delta t$. If the particle is making progress toward satisfying the condition, $G_t &gt; 1$ and the weight increases. If it’s drifting away, $G_t &lt; 1$ and the weight decreases. The incremental unnormalized importance weight at step $t$ is then simply</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\tilde{w}_t^{(i)} \propto G_t(x_t^{(i)})
\end{align*}
$$
</div>

<p>Now watch what happens when we accumulate these incremental updates. The denominators and numerators of consecutive steps cancel each other:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\prod_{t=T-\Delta t}^{0} G_t(x_t) &amp;= \frac{p(\hat{x}_0 | x_{T-\Delta t})}{p(\hat{x}_0 | x_T)} \cdot \frac{p(\hat{x}_0 | x_{T-2\Delta t})}{p(\hat{x}_0 | x_{T-\Delta t})} \cdots \frac{p(\hat{x}_0 | x_0)}{p(\hat{x}_0 | x_{\Delta t})} \\
&amp;= \frac{p(\hat{x}_0 | x_0)}{p(\hat{x}_0 | x_T)} \\
&amp;\approx \frac{p(x_0)}{\underbrace{p(\hat{x}_0 | x_T)}_{\text{const (pure noise)}}} \\
&amp;\propto p(x_0)
\end{align*}
$$
</div>

<p>since $\hat{x}_0(x_0) = x_0$ (fully denoised) and $p(y | \hat{x}_0(x_T))$ is essentially constant — pure noise carries no information about $y$. The telescoping gives us the best of both worlds: at any point in time $t$, the accumulated weight reflects the most up-to-date emission probability $p(y | \hat{x}_0(x_t))$, while the recursive update via $G_t$ means we only ever need to compute a single ratio per step (instead of keeping track of the entire trajectory).</p>

<h4 id="the-big-picture">The big picture</h4>

<p><img src="/blog/ESS/SMC.png" alt="ESS SMC Diagram" style="max-width: 100%; height: auto; display: block; margin: 0 auto;" /></p>

<p>Let’s step back and admire the view from the top of this particular mountain. What we’ve shown is that conditional diffusion generation is, at its core, a particle filter:</p>

<ul>
  <li>The <strong>prior</strong> $p(x_T) = \mathcal{N}(0, I)$ is where our particles start: pure isotropic noise.</li>
  <li>The <strong>transition kernel</strong> $p_\theta(x_{t-\Delta t} | x_t)$ — the discretized reverse SDE — propagates particles toward cleaner states.</li>
  <li>The <strong>emission model</strong> $p(\hat{x}_0 | x_t)$ evaluates each particle’s one-step denoiser prediction against the desired condition.</li>
  <li>The <strong>importance weights</strong> $\tilde{w}_t$ measure how well each particle’s denoised prediction matches $y$.</li>
  <li>The <strong>ESS</strong> — computed as $1/\sum_i \bar{w}_i^2$ — tells us how many particles are effectively contributing.</li>
  <li><strong>Resampling</strong> prunes bad hypotheses and replicates promising ones when the ESS signals weight degeneracy.</li>
</ul>

<p>Unconditional generation is the degenerate special case where $p(\hat{x}_0 | x_t) = \text{const}$ — there is no condition to satisfy, all weights are equal, the ESS equals $N$ at every step, and no resampling is ever triggered. You’re just running $N$ independent Markov chains in parallel. The moment you introduce any form of guidance — classifier guidance, classifier-free guidance, inpainting constraints, compositional objectives — you’ve entered SMC territory, and the ESS machinery we derived throughout this post becomes indispensable.</p>]]></content><author><name></name></author><category term="blog" /><summary type="html"><![CDATA[Weight Watchers]]></summary><media:thumbnail xmlns:media="http://search.yahoo.com/mrss/" url="/blog/ESS/ess_cover.png" /><media:content medium="image" url="/blog/ESS/ess_cover.png" xmlns:media="http://search.yahoo.com/mrss/" /></entry><entry><title type="html">Colombia</title><link href="/blog/Colombia25/" rel="alternate" type="text/html" title="Colombia" /><published>2026-01-05T00:00:00+00:00</published><updated>2026-01-05T00:00:00+00:00</updated><id>/blog/Colombia25</id><content type="html" xml:base="/blog/Colombia25/"><![CDATA[<!-- ## Berlin Over The Years -->

<style>
    .image-gallery {
        overflow: auto;
        margin-left: -1% !important;
    }

    .image-gallery li {
        float: left;
        float: top;
        display: block;
        margin: 0 0 1% 1%;
        width: 99%;
    }

    .image-gallery li a {
        text-align: top;
        text-decoration: none !important;
        color: #777;
    }

    .image-gallery li a span {
        display: block;
        text-overflow: ellipsis;
        overflow: hidden;
        white-space: nowrap;
        padding: 3px 0;
    }

    .image-gallery li a img {
        width: 100%;
        height: 100%;
        display: flex;
        vertical-align: top;
    }
</style>

<ul class="image-gallery">
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC3366.jpg"><img src="/photo_gallery/colombia25/DSC3366.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC3464.jpg"><img src="/photo_gallery/colombia25/DSC3464.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC3484.jpg"><img src="/photo_gallery/colombia25/DSC3484.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC3696-Pano.jpg"><img src="/photo_gallery/colombia25/DSC3696-Pano.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC3739.jpg"><img src="/photo_gallery/colombia25/DSC3739.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC3775.jpg"><img src="/photo_gallery/colombia25/DSC3775.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC3823.jpg"><img src="/photo_gallery/colombia25/DSC3823.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC3914.jpg"><img src="/photo_gallery/colombia25/DSC3914.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC3985-2.jpg"><img src="/photo_gallery/colombia25/DSC3985-2.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC4170-2.jpg"><img src="/photo_gallery/colombia25/DSC4170-2.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC4209.jpg"><img src="/photo_gallery/colombia25/DSC4209.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC4404.jpg"><img src="/photo_gallery/colombia25/DSC4404.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/colombia25/DSC4423.jpg"><img src="/photo_gallery/colombia25/DSC4423.jpg" /></a></li>
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
</ul>

<!--  -->

<!--  -->
<p><!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 --></p>]]></content><author><name></name></author><category term="photography" /><summary type="html"><![CDATA[The Caribbean, the Coffee Highlands, a Wedding]]></summary><media:thumbnail xmlns:media="http://search.yahoo.com/mrss/" url="/photo_gallery/colombia25/DSC3484.jpg" /><media:content medium="image" url="/photo_gallery/colombia25/DSC3484.jpg" xmlns:media="http://search.yahoo.com/mrss/" /></entry><entry><title type="html">Amalfi to Lecce</title><link href="/blog/Italy25/" rel="alternate" type="text/html" title="Amalfi to Lecce" /><published>2025-10-01T00:00:00+00:00</published><updated>2025-10-01T00:00:00+00:00</updated><id>/blog/Italy25</id><content type="html" xml:base="/blog/Italy25/"><![CDATA[<!-- ## Berlin Over The Years -->

<style>
    .image-gallery {
        overflow: auto;
        margin-left: -1% !important;
    }

    .image-gallery li {
        float: left;
        float: top;
        display: block;
        margin: 0 0 1% 1%;
        width: 99%;
    }

    .image-gallery li a {
        text-align: top;
        text-decoration: none !important;
        color: #777;
    }

    .image-gallery li a span {
        display: block;
        text-overflow: ellipsis;
        overflow: hidden;
        white-space: nowrap;
        padding: 3px 0;
    }

    .image-gallery li a img {
        width: 100%;
        height: 100%;
        display: flex;
        vertical-align: top;
    }
</style>

<ul class="image-gallery">
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    <li><a href="/photo_gallery/italy25/DSC2829.jpg"><img src="/photo_gallery/italy25/DSC2829.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/italy25/DSC2871.jpg"><img src="/photo_gallery/italy25/DSC2871.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/italy25/DSC2992-Pano.jpg"><img src="/photo_gallery/italy25/DSC2992-Pano.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/italy25/DSC3029.jpg"><img src="/photo_gallery/italy25/DSC3029.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/italy25/DSC3044.jpg"><img src="/photo_gallery/italy25/DSC3044.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/italy25/DSC3052.jpg"><img src="/photo_gallery/italy25/DSC3052.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/italy25/DSC3088-2.jpg"><img src="/photo_gallery/italy25/DSC3088-2.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/italy25/DSC3110.jpg"><img src="/photo_gallery/italy25/DSC3110.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/italy25/DSC3156.jpg"><img src="/photo_gallery/italy25/DSC3156.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/italy25/DSC3217-Enhanced-NR.jpg"><img src="/photo_gallery/italy25/DSC3217-Enhanced-NR.jpg" /></a></li>
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
</ul>

<!--  -->

<!--  -->
<p><!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 --></p>]]></content><author><name></name></author><category term="photography" /><summary type="html"><![CDATA[Driving around the boot]]></summary><media:thumbnail xmlns:media="http://search.yahoo.com/mrss/" url="/photo_gallery/italy25/DSC3110.jpg" /><media:content medium="image" url="/photo_gallery/italy25/DSC3110.jpg" xmlns:media="http://search.yahoo.com/mrss/" /></entry><entry><title type="html">Feynman-Kac</title><link href="/blog/FeynmanKac/" rel="alternate" type="text/html" title="Feynman-Kac" /><published>2025-07-12T00:00:00+00:00</published><updated>2025-07-12T00:00:00+00:00</updated><id>/blog/FeynmanKac</id><content type="html" xml:base="/blog/FeynmanKac/"><![CDATA[<script>
MathJax = {
  tex: {
    inlineMath: [['$','$'], ['\\(','\\)']],
    displayMath: [['$$','$$'], ['\\[','\\]']],
    processEscapes: true,
    tags: 'all'
  },
  options: {
    skipHtmlTags: ['script', 'noscript', 'style', 'textarea', 'pre']
  }
};
</script>

<script src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-mml-chtml.js" async=""></script>

<p>Recently there has been quite some interest in the idea of <em>Feynman-Kac Steering</em>.</p>

<p>For a given diffusion model, we can generate samples by iteratively removing noise, transforming a sample of pure noise into a sample of the target distribution.
At each step during the denoising process we can let the model estimate what the fully denoised sample would look like.
This can happen either through the score identity or is even more readily available if the model was trained on a denoising loss.
These fully denoised samples are then evaluated with a criterion.
This criterion is then “pulled back” into the denoising process to gauge whether the sample is going to be sufficiently good if fully denoised.
Feynman-Kac steering then implements a resampling step where all the samples are evaluated with the criterion in parallel and the batch is resampled from the noisy samples proportional to each samples fully denoised criterion.
This is essentially particle filtering with extra steps with stochastic PDE theory wrapped around it.</p>

<p>So let’s dissect the theory.</p>

<h3 id="ito-derivative">Ito Derivative</h3>

<p>At the core of diffusion model is a stochastic process $X_t$ that evolves according to a <a href="https://ludwigwinkler.github.io/blog/SDE/">stochastic differential equation (SDE)</a>.
The SDE in question is typically of the form:
\(dX_t = \mu(X_t, t) dt + \sigma(X_t, t) dW_t\)
where $\mu$ is the drift term, $\sigma$ is the diffusion term, and $W_t$ is a Wiener process (or Brownian motion).</p>

<p>What happens to a function $f(X_t, t)$ as it evolves over time? The <a href="https://ludwigwinkler.github.io/blog/ItosLemma/">Ito derivative</a> gives us a way to compute this by extending the classical Taylor expansion to the stochastic process realm:</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
f(X_{t+\Delta t}, t+ \Delta t) 
=&amp; f(X_t, t) + \frac{\partial f(X_t, t)}{\partial t} (t + \Delta t - t) + \frac{\partial f(X_t, t)}{\partial X_t} (X_{t+\Delta t} - X_t) \\ 
&amp;+ \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2} (X_{t+\Delta t} - X_t)^2 + o(\Delta t)
\end{align*}
$$
</div>

<p>An important factor that we’ll encounter in a second again is that we won’t consider effects of higher order as A) they get divided by ever decreasing factors  and B) they are negligible for small $\Delta t^p$ where $p&gt;2$.</p>

<p>We then get the Ito derivative:</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
f(X_{t+\Delta t}, t+ \Delta t) 
=&amp; f(X_t, t) + \frac{\partial f(X_t, t)}{\partial t} \Delta t + \frac{\partial f(X_t, t)}{\partial X_t} (X_{t+\Delta t} - X_t) \\ 
&amp;+ \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}(X_{t+\Delta t} - X_t)^2 + \cancel{o(\Delta t)} \\
% &amp;= f(X_t, t) + \frac{\partial f(X_t, t)}{\partial t} \Delta t + \frac{\partial f(X_t, t)}{\partial X_t} (X_{t+\Delta t} - X_t) \\
% &amp;+ \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2} (X_{t+\Delta t} - X_t)^2
\end{align*}
$$  
</div>

<p>For an infinitesimally small time increment $\Delta t$, we assume that the difference becomes a continuous $dt$ and we can plug in our SDE:</p>

<div style="overflow-x: auto;">
$$\begin{align*}
f(X_{t+\Delta t}, t+ \Delta t)
=&amp; f(X_t, t) + \frac{\partial f(X_t, t)}{\partial t} dt + \frac{\partial f(X_t, t)}{\partial X_t} \overbrace{(X_{t+\Delta t} - X_t)}^{=dX_t} \\
&amp;+ \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2} \overbrace{(X_{t+\Delta t} - X_t)^2}^{=dX_t^2} \\
=&amp; f(X_t, t) + \frac{\partial f(X_t, t)}{\partial t} dt + \frac{\partial f(X_t, t)}{\partial X_t} (\mu(X_t, t)dt + \sigma(X_t, t) dW_t) \\
&amp;+ \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2} (\mu(X_t, t)dt + \sigma(X_t, t) dW_t)^2 \\
=&amp; f(X_t, t) + \frac{\partial f(X_t, t)}{\partial t} dt + \frac{\partial f(X_t, t)}{\partial X_t} (\mu(X_t, t)dt + \sigma(X_t, t) dW_t) \\
&amp;+ \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2} (\mu(X_t, t)^2dt^2 + 2\mu(X_t, t)\sigma(x_t, t) dt dW_t + \sigma(X_t, t)^2 dW_t^2) \\
\end{align*}$$
</div>

<p>The next step to observe is that $dW_t = \epsilon \sqrt{dt}$ and that in the limit for a very small $dt$ (think of $dt=10^{-5}$), any effect of a term with $dt$ raised to any power larger than $1$ will diminish even faster (think of $dt^{1.5}=(10^{-5})^{1.5} = 10^{-7.5}$) and thus become negligible.
This allows us to drop any term where $dt^p$ where $p&gt;1$. Isn’t math convenient?</p>

<p>We thus get</p>
<div style="overflow-x: auto;">
$$\begin{align*}
f(X_{t+\Delta t}, t+ \Delta t)
=&amp; f(X_t, t) + \frac{\partial f(X_t, t)}{\partial t} dt + \frac{\partial f(X_t, t)}{\partial X_t} (\mu(X_t, t)dt + \sigma(X_t, t) dW_t) \\ 
&amp;+ \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2} (\underbrace{\cancel{\mu(X_t, t)^2dt^2}}_{dt^2 \rightarrow 0} + \underbrace{\cancel{2\mu(X_t, t)\sigma(x_t, t) dt dW_t}}_{dt^{1.5} \rightarrow 0} + \sigma(X_t, t)^2 \underbrace{dW_t^2}_{=dt}) \\
=&amp; f(X_t, t) + \frac{\partial f(X_t, t)}{\partial t} \Delta t + \frac{\partial f(X_t, t)}{\partial X_t} (\mu(X_t, t)dt + \sigma(X_t, t) dW_t) \\
&amp;+ \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2 dt
\end{align*}$$
</div>

<p>Rearranging a bit we have</p>
<div style="overflow-x: auto;">
$$\begin{align*}
f(X_{t+\Delta t}, t+ \Delta t)- f(X_t, t)
=&amp; \frac{\partial f(X_t, t)}{\partial t} dt + \frac{\partial f(X_t, t)}{\partial X_t} (\mu(X_t, t)dt + \sigma(X_t, t) dW_t) \\
&amp;+ \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2 dt
\end{align*}
$$
</div>

<p>So the infinitesimal change on the right hand side, properly denoted by $dt$ and $dW_t$, would equate the difference in the function $f(X_{t+\Delta t}, t+ \Delta t)$ and $f(X_t, t)$.
Given a step of $dt$ in time, the change in the function $f(X_t, t)$ is given by</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
df(X_t, t) =&amp;\left\{\frac{\partial f(X_t, t)}{\partial t} + \frac{\partial f(X_t, t)}{\partial X_t} \mu(X_t, t) + \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2\right\} dt \\
&amp; + \sigma(X_t, t) \frac{\partial f(X_t, t)}{\partial X_t} dW_t
\end{align*}$$
</div>

<h3 id="backward-kolmogorov-equation">Backward Kolmogorov Equation</h3>

<p>For a derivation of the Kolmogorov backward equation via the Kramers-Moyal expansion, see <a href="https://ludwigwinkler.github.io/blog/Kramers/">Fokker, Planck &amp; Kolmogorov Revisited</a>; it is also the central object used in <a href="https://ludwigwinkler.github.io/blog/Doob/">Doob’s h-transform</a>.</p>

<p>Ito’s lemma gives us a way to compute the infinitesimal change in a function of a stochastic process.
Since our process is stochastic, for any realized $x_t$, if we run the stochastic process again, we will get a different $X_T$ where $T= t + \Delta t$.
All the stochasticity that we accrue over time $\Delta t$ will make the function $f(X_T, T)$ a random variable.</p>

<p>As a sort of cognitive crutch, you can think of $f(X_T)$ as your Temple Run gold coin counter.
You always start from the same starting point $x_t$ and you always do the same 5 moves in the time $\Delta t$.
If the coin positions stay fixed and you do the same five moves every time you run, you’ll always get the same score $f(x_T)$ at the end.
This is what the deterministic part models.
But now the stochastic part keeps on changing the gold coin positions.
Even for a fixed score $f(x_t)$ and deterministic moves/dynamics, you final gold coin score $f(x_T)$ will vary.
Sometimes you’ll get a lot of gold coins, sometimes you’ll get none.
So essentially, your gold coin score $f(X_T)$ now is a random variable.</p>

<p>The next question we have to ask to naturally arrive at the Kolmogorov backward equation is whether there is a function $f_T(x_T)$ that describes the expected value of $f_T(X_T)$ for the current state $x_t$ <em>given that the dynamics are stochastic!</em></p>

<p>So we’d be interested in the equation</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
f(x_t, t) 
&amp;= \mathbb{E} \left[ f_T(X_T) | X_t = x_t \right] \\
&amp;= \int \underbrace{p(X_T = x_T|x_t)}_{\text{Stochastic Process}} f_T(X_T) dx_T
\end{align*}
$$
</div>

<p>where the expectation how likely a state $X_T$ is given the current state $x_t$ is given by the stochastic process $p(X_T | x_t)$.
So $f_T(x_T)$ gives you an estimate of how much payout at the end you’ll get midway through your Temple Run.</p>

<div style="max-width: 100%; overflow-x: auto;">
  <img src="/blog/blogthumbnails/FeynmanKac.png" alt="Feynman-Kac Illustration" style="width: 100%; height: auto; display: block; margin: 0 auto;" />
</div>

<p><em>Figure: Visual intuition for Feynman-Kac. The stochastic process $X_t$ evolves under random noise, while the expected value of a terminal criterion $f_T(X_T)$ is “pulled back” to earlier times via the backward PDE. Particle trajectories (red) and expectation (blue) illustrate the connection between stochastic paths and deterministic backward evolution.</em></p>

<p>But the definition $f_T(x_T)$ is inherently forward looking as it relies on the solving the SDE forward in time.
Can we also reverse time in a way to compute earlier values of $f(x_t,t)$ starting from the terminal values $f_T(X_T)$?</p>

<p>It turns out that there is a partial differential equation hidden beneath all this stochasticity that allows us to compute the expected value of $f(x_t, t)$ backwards in time.</p>

<p>To show this we’ll start out with Ito’s lemma above but this time integrate it all the way from $t$ to $T$.</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
f_T(X_T) | X_t =&amp; f(X_t, t) + \int_t^T \left\{\frac{\partial f(X_t, t)}{\partial t} + \frac{\partial f(X_t, t)}{\partial X_t} \mu(X_t, t) + \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2\right\} dt \\ 
&amp; + \int_t^T \sigma(X_t, t) \frac{\partial f(X_t, t)}{\partial X_t} dW_t
\end{align*}
$$
</div>

<p>(where I should have used the integration variable $s$ or $\tau$ instead of $t$ but I didn’t want to introduce unnecessary notational noise ;-) ).
More importantly, this equation tells us that if we start with some intermediate value $f(X_t, t)$ and integrate the infinitesimal changes in the function $f(X_t, t)$ from $t$ to $T$, we will end up with the final value of the function $f_T(X_T)$ which is our terminal payout.</p>

<p>And now comes the important step …</p>

<p>… drumroll 🥁 …</p>

<p>… we take the expectation. 👍</p>

<p>The important thing to note is that the expectation of a Wiener process, however he might be scaled over any time, is zero.</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{E}\left[ f_T(X_T) | X_t \right] =&amp; \mathbb{E} \left[ f(X_t, t) + \int_t^T\left\{\frac{\partial f(X_t, t)}{\partial t} + \frac{\partial f(X_t, t)}{\partial X_t} \mu(X_t, t) + \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2\right\} dt \right] \\ 
&amp; + \underbrace{\cancel{\mathbb{E}\left[\sigma(X_t, t) \frac{\partial f(X_t, t)}{\partial X_t} dW_t \right]}}_{=0}
\end{align*}
$$
</div>

<p>While $X_t$ is a random variable, its stochasticity arises completely from the input of the Wiener process $dW_t$.
If we eliminate the influence of the Wiener process with the expectation $\mathbb{E}$ at every step, we will essentially obtain a deterministic ODE.
With this intuition, the expectation essentially has no influence on the remaining terms and we can write</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\underbrace{\mathbb{E}\left[ f_T(x_T) | X_t \right]}_{=f(X_t, t)}
&amp;= f(X_t, t) + \int_t^T\left\{\frac{\partial f(X_t, t)}{\partial t} + \frac{\partial f(X_t, t)}{\partial X_t} \mu(X_t, t) + \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2\right\} dt \\
f(X_t, t) &amp;= f(X_t, t) + \int_t^T\Bigg\{ \underbrace{\frac{\partial f(X_t, t)}{\partial t} + \frac{\partial f(X_t, t)}{\partial X_t} \mu(X_t, t) + \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2}_{\stackrel{!}{=} 0}\Bigg\} dt
\end{align*}
$$
</div>

<p>The equation above can only hold if the term we’re integrating</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
0 &amp;\stackrel{!}{=} \int_t^T \left\{\frac{\partial f(X_t, t)}{\partial t} + \frac{\partial f(X_t, t)}{\partial X_t} \mu(X_t, t) + \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2 \right\} dt
\end{align*}
$$
</div>

<p>An important observation is that the integral above is a definite integral over the arbitrary time interval $[t, T]$.
This means that the integrand must be zero for all $t$ in the interval $[t, T]$ as we could choose $t$ arbitrarily close to $T$ and the integral would still have to be zero.
We can go backward in time from the terminal time $T$ to $T-dt$ and the integral has to be zero, i.e. $\int_{T-dt}^T \ldots dt = 0$.
If that integral is zero, then the integral $\int_{T-2dt}^{T-dt} \ldots dt = 0$ also has to be zero in order for the whole integral $\int_{T-2dt}^T \ldots dt = 0$ to hold.
This means that the integrand must be zero for all $t$ in the interval $[t, T]$.
This gives us the backward Kolmogorov equation:</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
-\frac{\partial f(X_t, t)}{\partial t} &amp;= \frac{\partial f(X_t, t)}{\partial X_t} \mu(X_t, t) + \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2
\end{align*}
$$
</div>

<p>with some terminal condition $f_T(X_T)$.
This result is quite remarkable as it allows us to compute the expected value of a function of a stochastic process at an earlier time $t$ given the terminal condition $f_T(X_T)$ with a PDE backward in time instead of solving a SDE forward in time.</p>

<p>While it looks deceptively similar to the Ito derivative, the backward Kolmogorov equation is a PDE that describes how the expected value of a function of a stochastic process evolves backward in time.</p>

<p>In practical terms, this means that if we have a function $f_T(X_T)$ at the terminal time $T$ with the stochastic dynamics of $dX_t$, we can compute the expected value of this function at an earlier time $t$ by solving the backward Kolmogorov equation.</p>

<p>For example, we could redefine the function $f_T(X_T)$ as the probability of $p(x_T)$ at the terminal time $T$.
We could then solve the Kolmogorov backward equation in the form of a PDE backwards in time to obtain the expected value of the probability at an earlier time $t$, which we would denote as $p(X_T | X_t)$.</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
- \frac{\partial p(X_T | X_t)}{\partial t} =&amp; \frac{\partial p(X_T|X_t)}{\partial X_t} \mu(X_t, t) + \frac{1}{2} \frac{\partial^2 p(X_T | X_t)}{\partial X_t^2}\sigma(X_t, t)^2
\end{align*}
$$
</div>

<p>This equation would evolve backward the probability density $p(X_T | X_t)$ from the terminal time $T$ to the earlier time $t$.
At some earlier point $t$, we could evaluate the how likely the state $X_T$ would be given the earlier state $X_t$.</p>

<h3 id="a-stochastic-sidenote">A Stochastic Sidenote</h3>

<p>For some function $f(x_t, t)$ and a terminal value $f_T(x_T)$, we derived the following identity above</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
f_T(X_T) | X_t =&amp; f(X_t, t) + \int_t^T \left\{\frac{\partial f(X_t, t)}{\partial t} + \frac{\partial f(X_t, t)}{\partial X_t} \mu(X_t, t) + \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2\right\} dt \\ 
&amp; + \int_t^T \sigma(X_t, t) \frac{\partial f(X_t, t)}{\partial X_t} dW_t
\end{align*}
$$
</div>

<p>But now with our previously gained knowledge we can ascertain that the deterministic integrand is always zero, so we obtain the following equation</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
f_T(X_T) | X_t =&amp; f(X_t, t) + \int_t^T \sigma(X_t, t) \frac{\partial f(X_t, t)}{\partial X_t} dW_t
\end{align*}
$$
</div>

<p>which is A) a martingale and B) a random variable due to the Ito integral.
We can make this more precise by observing as before that the expectation is zero because $\mathbb{E}[dW_t] = 0$.
Also, we can quite easily compute the variance of this random variable by computing</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{E}\left[ \left( \int_t^T \sigma(X_t, t) \frac{\partial f(X_t, t)}{\partial X_t} dW_t \right)^2 \right]
&amp;= 
\mathbb{E}\left[ \left( \int_t^T \sigma(X_t, t) \frac{\partial f(X_t, t)}{\partial X_t} dW_t \right) 
\left( \int_t^T \sigma(X_{t'}, {t'}) \frac{\partial f(X_{t'}, {t'})}{\partial X_{t'}} dW_{t'} \right) \right].
\end{align*}
$$
</div>

<p>By definition a Wiener process $W_t$ has the property that $\mathbb{E}[dW_t dW_{t’}] = \delta(t-t’) dt$ which says that the product of a Wiener process at two different times is zero unless the two times are the same.
This means that we can rewrite the above equation as</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
\mathbb{E}\left[ \left( \int_t^T \sigma(X_t, t) \frac{\partial f(X_t, t)}{\partial X_t} dW_t \right)^2 \right]
&amp;= \int_t^    T \int_t^T \mathbb{E}\left[ \sigma(X_t, t) \frac{\partial f(X_t, t)}{\partial X_t} \sigma(X_{t'}, {t'}) \frac{\partial f(X_{t'}, {t'})}{\partial X_{t'}} dW_t dW_{t'} \right] \\
&amp;= \int_t^T \sigma(X_t, t)^2 \frac{\partial f(X_t, t)}{\partial X_t}^2 dt
\end{align*}
$$
</div>

<p>because all the “cross terms” where $t \neq t’$ vanish due to the property of the Wiener process. This is also known as Ito Isometry.
This gives us the mean and variance of the random variable $f_T(X_T) | X_t$:</p>

<div style="overflow-x: auto;">
$$\begin{align*}
\mathbb{E}[f_T(X_T) | X_t] &amp;= f(X_t, t) \\
\text{Var}[f_T(X_T) | X_t] &amp;= \int_t^T \sigma(X_t, t)^2 \frac{\partial f(X_t, t)}{\partial X_t}^2 dt \\
&amp;\downarrow \\
f_T(X_T) | X_t &amp;\sim \mathcal{N}\left(f(X_t, t), \int_t^T \sigma(X_t, t)^2 \frac{\partial f(X_t, t)}{\partial X_t}^2 dt\right)
\end{align*}$$
</div>

<h3 id="feynman-kac-equation">Feynman-Kac Equation</h3>

<p>Previously, we’ve observed how with a terminal condition at time $T$, we can compute the expected value of a function of a stochastic process at an earlier time $t$ by solving the backward Kolmogorov equation</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
-\frac{\partial f(X_t, t)}{\partial t} &amp;= \frac{\partial f(X_t, t)}{\partial X_t} \mu(X_t, t) + \frac{1}{2} \frac{\partial^2 f(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2
\end{align*}
$$
</div>

<p>where the dynamics of a stochastic process $X_t$ are given by the SDE
\(dX_t = \mu(X_t, t) dt + \sigma(X_t, t) dW_t\)
and the terminal condition is given by $f_T(X_T)$.</p>

<p>In their famous equation, Mark Kac and Richard Feynman asked “Yo, what if we add more terms to the Kolmogorov Backward Equation?” (essentially … please don’t quote me on this).</p>

<p>In order to align with the normal notation of the Feynman-Kac equation, we will be using $u$ instead of $f$.
The FK equation then is posited as follows:</p>
<div style="overflow-x: auto;">
$$
\begin{align*}-\frac{\partial u(X_t, t)}{\partial t} &amp;= \frac{\partial u(X_t, t)}{\partial X_t} \mu(X_t, t) + \frac{1}{2} \frac{\partial^2 u(X_t, t)}{\partial X_t^2}\sigma(X_t, t)^2 \underbrace{+ c(X_t, t) u(X_t, t) + v(X_t, t)}_{\text{new FK terms}}
\end{align*}
$$
</div>

<p>where $c(X_t, t)$ and $v(X_t, t)$ are some functions that we can choose.</p>

<p>We can interpret $c(X_t, t)$ as a “potential” term that modifies the expected value of the function $u(X_t, t)$ at time $t$ by interacting with the function $u(X_t, t)$.
$v(X_t, t)$ can be interpreted as a “forcing” term that adds an additional contribution to the expected value of the function $u(X_t, t)$ at time $t$.</p>

<p>In order to make the notation a bit easier we’ll shorten the notation to</p>

<div style="overflow-x: auto;">
$$\begin{align*}
-\partial_t u &amp;= \partial_{x} u \ \mu + \frac{1}{2} \partial_{xx} u \ \sigma^2 + c \ u + v
\end{align*}
$$
</div>

<p>Solving this equation backwards in time from the terminal condition $u(X_T, T)$ gives us the expected value of the function $u(X_t, t)$ at an earlier time $t$ given the terminal condition $u(X_T, T)$.</p>

<p>So how do we solve this equation?
Solving PDEs is fiendishly hard and no easy task.</p>

<p>Usually, there are a couple of ‘ansatz’s that we can use to solve PDE’s.
For the FK equation we’ll chose the ansatz of defining a new function $y(x_t, t)$ as</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
y(X_s, s) &amp;= u(X_s, s) \ e^{\int_t^s c(x_r, r) dr} \\
&amp;\downarrow \text{Shorten the notation} \\
y_s &amp;= u_s \ e^{\int_t^s c_r dr}
\end{align*}
$$
</div>

<p>where $u_s$ is the function $u(X_s, s)$ and $c_r$ is the function $c(X_r, r)$.</p>

<p>Differentiating this function with respect to $s$ and the product rule gives us</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
dy_s &amp;= du_s \ e^{\int_t^s c_r dr} + u_s \ c_s \ e^{\int_t^s c_r dr}
\end{align*}
$$
</div>

<p>where $du$ is an Ito derivative but $u_s \ d e^{\int_t^s c_r dr}$ is not.</p>

<p>Both $u(s, X_s)$ and $\exp\left( \int_t^s c(X_r), dr \right)$ depend on the random path $X$, but only $u(s, X_s)$ is a function of the semimartingale $X_s$ at a point—so Itô’s lemma applies.
$e^{\int_t^s c_r dr}$ is a functional of the entire path, not just $X_s$, and its exponent $\int_t^s c(X_r), dr$ is absolutely continuous (no Brownian term).
You assume the values $X_r$ to be deterministic and the functional $\int c_r dr$ is evaluated on an “already” realized path of $X_r$’s.
So the ordinary chain rule suffices:
\(d e^{\int_t^s c_r dr} = e^{\int_t^s c_r dr} c_r ds\)</p>

<p>In short: use Itô’s lemma for functions of $X_s$; use the chain rule for pathwise functionals without stochastic terms.</p>

<p>Evaluating the Ito derivative we get</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
du &amp;= \left\{ \partial_t u + \partial_x u \ \mu + \frac{1}{2} \partial_{xx} u \ \sigma^2 \right\} dt + \sigma \ \partial_x u_t \ dW_t \\
\end{align*}
$$
</div>

<p>and we can compare that to our original FK equation</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
-\partial_t u &amp;= \partial_{x} u \ \mu + \frac{1}{2} \partial_{xx} u \ \sigma^2 + c \ u + v \\
- c \ u - v &amp;= \partial_t u +\partial_{x} u \ \mu + \frac{1}{2} \partial_{xx} u \ \sigma^2
\end{align*}
$$
</div>

<p>We can then proceed to plug our $-c \ u - v$ into our Ito derivative and get</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
du &amp;= \left\{ -c \ u - v \right\} dt + \sigma \ \partial_x u_t \ dW_t \\
\end{align*}
$$
</div>

<p>Now we take the $du$ and plug that back into our definition of $dy$ and get</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
dy_s &amp;= du_s \ e^{\int_t^s c_r dr} + u_s \ c_s \ e^{\int_t^s c_r dr} \\
&amp;= \left( \left\{ \cancel{-c_s \ u_s} - v_s \right\} ds + \sigma \ \partial_x u_s \ dW_s \right) \ e^{\int_t^s c_r dr} \
\cancel{ + u_s \ c_s \ e^{\int_t^s c_r dr}} \\
&amp;= \left(- v_s \ ds + \sigma \ \partial_x u_s \  dW_s \right) e^{\int_t^s c_r dr}
\end{align*}
$$
</div>
<p>Integrating both sides from $s=t$ to $s=T$ gives us</p>
<div style="overflow-x: auto;">
$$
\begin{align*}y_T - y_t &amp;= \int_t^T \left(- v_s \ ds + \sigma \ \partial_x u_s \  dW_s \right) e^{\int_t^s c_r dr} \ ds\\
&amp;= -\int_t^T v_s e^{\int_t^s c_r dr} ds + \int_t^T \sigma \ \partial_x u_s \ e^{\int_t^s c_r dr} \ dW_s
\end{align*}
$$
</div>
<p>and evaluating the left hand side gives us</p>
<div style="overflow-x: auto;">
$$
\begin{align*}
y_T - y_t &amp;= u_T \ e^{\int_t^T c_r dr} - u_t \ \underbrace{e^{\int_t^t c_r dr}}_{=e^0=1} \\
&amp;= u_T \ e^{\int_t^T c_r dr} - u_t
\end{align*}
$$
</div>

<p>where integrating any quantity a zero amount $\int_t^t \ldots dt$ is always equal to zero as nothing gets added to the integral.</p>

<p>We then ultimately have</p>

<div style="overflow-x: auto;">
$$\begin{align*}
u_T \ e^{\int_t^T c_r dr} - u_t &amp;= \int_t^T -v_s e^{\int_t^s c_r dr} ds + \int_t^T \sigma \ \partial_x u_s \ e^{\int_t^s c_r dr} \ dW_s
\end{align*}
$$
</div>

<p>Now, we’re going to pull off our favourite, slick stochastic process move and take the expectation of both sides which eliminates the martingale $\int dW_s$ term on the right hand side,</p>

<div style="overflow-x: auto;">
$$\begin{align*}
u_t &amp;= \mathbb{E}\left[ u_T \ e^{\int_t^T c_r dr} +  \int_t^T v_s e^{\int_t^s c_r dr} ds \right]
\end{align*}
$$
</div>

<p>If we now set $c_r=0$ and $v_s =0$, we indeed recover OG Kolmogorov backward equation $u_t = \mathbb{E}[u_T]$.</p>

<h3 id="feynman-kac-steering">Feynman-Kac Steering</h3>

<p>Diffusion models are able to produce an estimate of a fully denoised sample $\hat{x}_0$ from a diffusive sample $x_t$ by either leveraging the score identity</p>

<div style="overflow-x: auto;">
$$
\begin{align*}
\nabla_{x_t} \log p_t(x_t|x_0) &amp;=  -\frac{(x_t - \alpha_t x_0)}{\sigma_t^2} \quad ; \quad x_t = \alpha_t x_0 + \sigma_t \epsilon\\ 
&amp;\downarrow \\
\hat{x}_0 &amp;= x_t + \frac{\sigma_t^2 \ \nabla_{x_t} \log p_t(x_t|x_0)}{\alpha_t} \\
&amp;= \frac{x_t - \sigma_t \ \epsilon_\theta}{\alpha_t}
\end{align*}
$$
</div>

<p>or by using a denoising loss which directly aims at estimating the fully denoised sample $x_0$ from a noisy sample $x_t$.</p>

<p>In both cases, we can use the denoised sample $\hat{x}_0$ as an estimate of the fully denoised sample $x_0$.
This denoised sample is then evaluated with a criterion $c(\hat{x}_0)$ that tells us how good the sample is.</p>

<p>The Feynman-Kac/Kolmogorov Backward equation give us a mathematical tool to “pull back” $c(\hat{x}_0)$ to a more diffused sample $x_t$.
In essence it allows us to estimate a diffused version of the criterion $c(x_t)$ at an earlier time step $t$.
We use the Feynman-Kac framework to estimate the expected value of the criterion at an earlier time step $t$.
Theoretically, we could try to evaluate the full expectation $\mathbb{E}[c(x_t)] = \int c(x_t) p(x_t) dx_t$ but this is computationally expensive and not feasible in practice.
We can instead filter or resample the batch, preferentially retaining samples with higher expected criterion values and discarding those predicted to perform poorly.
FK steering then allows us to resample the batch of samples $x_t$ proportional to the expected value of the criterion $c(x_t)$.
Since we’re dealing with a stochastic process, even samples in the mini batch which were resampled from the same noisy sample $x_t$ will have different expected values of the criterion $c(x_t)$ further down the reverse diffusion process.</p>]]></content><author><name></name></author><category term="blog" /><summary type="html"><![CDATA[Is it Katz, Kak, Kaz, Katsch? Anyway, it's nice math.]]></summary><media:thumbnail xmlns:media="http://search.yahoo.com/mrss/" url="/blog/blogthumbnails/FeynmanKac.png" /><media:content medium="image" url="/blog/blogthumbnails/FeynmanKac.png" xmlns:media="http://search.yahoo.com/mrss/" /></entry><entry><title type="html">USA</title><link href="/blog/USA25/" rel="alternate" type="text/html" title="USA" /><published>2025-05-12T00:00:00+00:00</published><updated>2025-05-12T00:00:00+00:00</updated><id>/blog/USA25</id><content type="html" xml:base="/blog/USA25/"><![CDATA[<!-- ## Berlin Over The Years -->

<style>
    .image-gallery {
        overflow: auto;
        margin-left: -1% !important;
    }

    .image-gallery li {
        float: left;
        float: top;
        display: block;
        margin: 0 0 1% 1%;
        width: 99%;
    }

    .image-gallery li a {
        text-align: top;
        text-decoration: none !important;
        color: #777;
    }

    .image-gallery li a span {
        display: block;
        text-overflow: ellipsis;
        overflow: hidden;
        white-space: nowrap;
        padding: 3px 0;
    }

    .image-gallery li a img {
        width: 100%;
        height: 100%;
        display: flex;
        vertical-align: top;
    }
</style>

<ul class="image-gallery">
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC0748.jpg"><img src="/photo_gallery/usa25/DSC0748.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC0834.jpg"><img src="/photo_gallery/usa25/DSC0834.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC0973.jpg"><img src="/photo_gallery/usa25/DSC0973.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC1032.jpg"><img src="/photo_gallery/usa25/DSC1032.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC1422.jpg"><img src="/photo_gallery/usa25/DSC1422.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC1498-Pano.jpg"><img src="/photo_gallery/usa25/DSC1498-Pano.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC1556.jpg"><img src="/photo_gallery/usa25/DSC1556.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC1685.jpg"><img src="/photo_gallery/usa25/DSC1685.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC1741.jpg"><img src="/photo_gallery/usa25/DSC1741.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC1796.jpg"><img src="/photo_gallery/usa25/DSC1796.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC1844-Enhanced-NR.jpg"><img src="/photo_gallery/usa25/DSC1844-Enhanced-NR.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC2144-Pano.jpg"><img src="/photo_gallery/usa25/DSC2144-Pano.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC2217.jpg"><img src="/photo_gallery/usa25/DSC2217.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC2297.jpg"><img src="/photo_gallery/usa25/DSC2297.jpg" /></a></li>
    
    
    
    
    <li><a href="/photo_gallery/usa25/DSC2347.jpg"><img src="/photo_gallery/usa25/DSC2347.jpg" /></a></li>
    
    
</ul>

<!--  -->

<!--  -->
<p><!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 -->
  <!-- 
 --></p>]]></content><author><name></name></author><category term="photography" /><summary type="html"><![CDATA[Midwest to the West Coast]]></summary><media:thumbnail xmlns:media="http://search.yahoo.com/mrss/" url="/photo_gallery/usa25/DSC1685.jpg" /><media:content medium="image" url="/photo_gallery/usa25/DSC1685.jpg" xmlns:media="http://search.yahoo.com/mrss/" /></entry></feed>