From f9398cc336f946255d8ac882307e3711c28f1dd2 Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 11:44:51 +0800 Subject: [PATCH 01/20] =?UTF-8?q?fix(a11y):=20audit=2011=20finished=20and?= =?UTF-8?q?=2036=20=E2=80=94=20every=20control=20a=20finger=20can=20hit,?= =?UTF-8?q?=20a=20back-to-top=20that=20stays=20off=20the=20text?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 11 the tap area reaches the controls written as plain diff --git a/content/posts/head-camera/components/pick-lab.tsx b/content/posts/head-camera/components/pick-lab.tsx index 4b38abb..642274d 100644 --- a/content/posts/head-camera/components/pick-lab.tsx +++ b/content/posts/head-camera/components/pick-lab.tsx @@ -183,7 +183,7 @@ export function PickLab() {
{DRIVERS.map((d) => ( ))} diff --git a/content/posts/hydranet-fruit/components/training-lab.tsx b/content/posts/hydranet-fruit/components/training-lab.tsx index e04c677..5a25aca 100644 --- a/content/posts/hydranet-fruit/components/training-lab.tsx +++ b/content/posts/hydranet-fruit/components/training-lab.tsx @@ -203,7 +203,7 @@ export function TrainingLab() { setHeads(h); }} className={cn( - "h-7 rounded-sm border px-2.5 text-xs transition-colors", + "tap h-7 rounded-sm border px-2.5 text-xs transition-colors", heads === h ? "border-signal bg-signal/10 text-signal" : "border-border text-muted-foreground hover:text-foreground", )} > @@ -236,7 +236,7 @@ export function TrainingLab() { aria-pressed={selected === i} aria-label={`${t.testSet} ${i + 1}`} onClick={() => setSelected(i)} - className={cn("rounded-sm border p-px transition-colors", selected === i ? "border-signal" : "border-border hover:border-foreground/40")} + className={cn("tap rounded-sm border p-px transition-colors", selected === i ? "border-signal" : "border-border hover:border-foreground/40")} > diff --git a/content/posts/transformer-from-scratch/components/training-lab.tsx b/content/posts/transformer-from-scratch/components/training-lab.tsx index 856c3eb..ce7d53d 100644 --- a/content/posts/transformer-from-scratch/components/training-lab.tsx +++ b/content/posts/transformer-from-scratch/components/training-lab.tsx @@ -160,7 +160,7 @@ export function TrainingLab() { setRunning(false); }} className={cn( - "h-7 rounded-sm border px-2.5 text-xs transition-colors", + "tap h-7 rounded-sm border px-2.5 text-xs transition-colors", task === name ? "border-signal bg-signal/10 text-signal" : "border-border text-muted-foreground hover:text-foreground", )} > @@ -212,7 +212,7 @@ export function TrainingLab() { aria-pressed={speed === s} onClick={() => setSpeed(s)} className={cn( - "h-7 rounded-sm border px-2.5 text-xs transition-colors", + "tap h-7 rounded-sm border px-2.5 text-xs transition-colors", speed === s ? "border-signal bg-signal/10 text-signal" : "border-border text-muted-foreground hover:text-foreground", )} > diff --git a/mdx-components.tsx b/mdx-components.tsx index 575d613..f36e533 100644 --- a/mdx-components.tsx +++ b/mdx-components.tsx @@ -14,7 +14,7 @@ function heading(Tag: "h2" | "h3" | "h4") { return ( {id ? ( - + {children} ) : ( From 56af43646bacfe57bf8f1a447b8f3a42dcf6bc7f Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 12:01:32 +0800 Subject: [PATCH 02/20] =?UTF-8?q?docs(cnn):=20polish=20=E2=84=96=20001=20?= =?UTF-8?q?=E2=80=94=20accuracy,=20repetition,=20the=20list=20of=20numbers?= =?UTF-8?q?,=20wording?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Facts: the forward pass is 266 lines, "a little over two hundred" (was "about two hundred"); the case for ReLU is that convolutions alone fold into one (pooling is not linear, so "without ReLU the stack is linear" was loose); the receptive-field sentence says what one second-layer output covers; the occlusion caption no longer asserts what the model relies on, only what the map usually shows. - The fully connected comparison was one long sentence of numbers; it is a table, with each width's range over its three seeds (docs/research/mlp-baseline/output.jsonl) instead of a rounded mean. - "No way to say I don't know" was said twice; the real-system consequence joins the bullet it belongs to. - Filler cut ("convolution does something simple", "the diagram isn't wrong"); 儀器 becomes 圖 in the prose. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/cnn-from-scratch/en.mdx | 63 +++++++++++++---------- content/posts/cnn-from-scratch/zh.mdx | 73 +++++++++++++++------------ 2 files changed, 75 insertions(+), 61 deletions(-) diff --git a/content/posts/cnn-from-scratch/en.mdx b/content/posts/cnn-from-scratch/en.mdx index 62925bd..e2f84bc 100644 --- a/content/posts/cnn-from-scratch/en.mdx +++ b/content/posts/cnn-from-scratch/en.mdx @@ -3,7 +3,7 @@ title: "A CNN from scratch: watching a convolutional network see, in your browse description: A handwritten-digit classifier written in plain TypeScript with no machine-learning library, then opened up so you can look at the output of every layer. seoTitle: "A CNN from scratch: watch a conv net see" date: 2024-10-12 -updated: 2026-09-20 +updated: 2026-09-22 tags: [computer-vision, cnn, from-scratch] no: 1 featured: true @@ -12,9 +12,9 @@ interactive: true import { ConvStepper, DrawPredict, FeatureMaps, KernelPlayground, OcclusionMap } from "./components"; -Most introductions to convolutional neural networks open with a block diagram: a few boxes, a few arrows, an answer at the end. The diagram isn't wrong, but it hides the interesting part. What happens *inside* the boxes? +Most introductions to convolutional neural networks open with a block diagram: a few boxes, a few arrows, an answer at the end. The interesting part is exactly what it hides: what happens *inside* the boxes? -This article goes the other way round. The classifier below is already running in your browser. There is no server behind it and no TensorFlow.js or ONNX Runtime; the whole forward pass is about two hundred lines of TypeScript.The source lives in this site's repo under `lib/ml`: `tensor.ts`, `ops.ts` and `sequential.ts`, with zero dependencies. Draw a digit first. Then we'll take it apart, layer by layer. +This article goes the other way round. The classifier below is already running in your browser. There is no server behind it and no TensorFlow.js or ONNX Runtime; the whole forward pass is a little over two hundred lines of TypeScript.The source lives in this site's repo under `lib/ml`: `tensor.ts`, `ops.ts` and `sequential.ts`, with zero dependencies. Draw a digit, and then we take it apart layer by layer. @@ -22,20 +22,20 @@ This article goes the other way round. The classifier below is already running i ## An image is a grid of numbers -To a model, a greyscale image is a two-dimensional array in which every cell holds a number between 0 and 1: 0 is blank paper, 1 is ink. The small picture in the middle above is that array, $28 \times 28 = 784$ numbers in all. +To a model, a greyscale image is a two-dimensional array in which every cell holds a number between 0 and 1: 0 is blank paper, 1 is ink. The small picture in the middle above is that array: $28 \times 28 = 784$ numbers. The obvious approach is to flatten those 784 numbers and feed them into a fully connected layer. It works, but it's wasteful: a fully connected layer has no idea which pixels are neighbours. Shift the same "7" two pixels to the right and it becomes an entirely different input that has to be learned again. A CNN starts from two assumptions that hold for almost every image: - **Locality.** Meaningful features such as edges, corners and stroke ends involve only a small patch of neighbouring pixels. -- **Translation equivariance.** A vertical edge is the same thing in the top-left corner as in the bottom-right, and the same parameters should detect it in both places. What convolution gives is "shift the input one cell right and the output shifts one cell right" (equivariance); an answer that truly does not change with position (invariance) comes from the pooling that follows, and only approximately. +- **Translation equivariance.** A vertical edge is the same thing in the top-left corner as in the bottom-right, so the same parameters should detect it in both. Convolution guarantees that shifting the input one cell right shifts the output one cell right; an answer that does not change with position at all (translation invariance) comes from the pooling that follows, and only approximately. -Build those two assumptions into the structure of the model and you get convolution. +Build those two assumptions into the model's structure and you get convolution. ## Convolution: a small window sliding over the image -Convolution does something simple. Take a small matrix of weights, called a **kernel**, usually 3×3. Lay it over the top-left corner of the image, multiply the nine overlapping pairs, and add them up to get one number. Slide one pixel to the right and do it again. Once you've covered the whole image, those numbers form a new image called a **feature map**. +Take a small matrix of weights, called a **kernel**, usually 3×3. Lay it over the top-left corner of the image, multiply the nine overlapping pairs, and add them up to get one number. Slide one pixel to the right and do it again. Once you've covered the whole image, those numbers form a new image called a **feature map**. As a formula: @@ -43,15 +43,15 @@ $$ y_{i,j} = b + \sum_{u=0}^{2} \sum_{v=0}^{2} w_{u,v} \, x_{i+u,\, j+v} $$ -The instrument below slows this down. The input is a bright vertical bar (its bottom two rows are shifted one pixel to the left), and the kernel has −1 down its left column and +1 down its right. Press Step and watch each output cell being computed. +The figure below slows this down. The input is a bright vertical bar (its bottom two rows are shifted one pixel to the left), and the kernel has −1 down its left column and +1 down its right. Press Step and watch each output cell being computed. -What this kernel computes is "right minus left". Over a uniform region that difference is zero; only where dark on the left meets bright on the right does the output become a large positive number. It is a **vertical edge detector**, and it uses the same nine numbers wherever the edge happens to be.Strictly speaking, what deep-learning frameworks call convolution is cross-correlation: the kernel isn't flipped. Because the weights are learned, flipping makes no difference, and the name stuck. +This kernel computes "right minus left". Over a uniform region that is zero; only where dark on the left meets bright on the right does it become a large positive number. It is a **vertical edge detector**, and it uses the same nine numbers wherever the edge is.Strictly speaking, what deep-learning frameworks call convolution is cross-correlation: the kernel isn't flipped. Because the weights are learned, flipping makes no difference, and the name stuck. -In code it is a handful of nested loops. Below is a simplified single-channel version; the one in `lib/ml/ops.ts` wraps it in loops over the batch and the channels, and the innermost loops are the same: +In code it is four nested loops. Below is a simplified single-channel version; the one in `lib/ml/ops.ts` adds loops over the batch and the channels, and its two innermost loops are exactly these: ```ts title="simplified from lib/ml/ops.ts" {6-8} for (let oy = 0; oy < oH; oy++) { @@ -72,29 +72,29 @@ for (let oy = 0; oy < oH; oy++) { ### Change the numbers, change the feature -Nine numbers can do more than you'd expect. The input below is the digit you just drew (or the sample 7 if you drew nothing); edit the kernel and see. +Nine numbers can do more than you'd expect. The input below is the digit you just drew (the sample 7 if you drew nothing); edit the kernel and see. -In classical computer vision these kernels were designed by hand: Sobel, Laplacian, Gaussian. The key move in a CNN is to **stop designing them**. Treat the nine numbers as parameters, initialise them randomly, and let gradient descent find the sets that are most useful for the task. +Classical computer vision designed these kernels by hand: Sobel, Laplacian, Gaussian. The key move in a CNN is to **stop designing them**: treat the nine numbers as parameters, initialise them randomly, and let gradient descent find the sets most useful for the task. ## ReLU and pooling A convolution is usually followed by two very small operations. -**ReLU** sets negative values to zero: $\mathrm{ReLU}(x) = \max(0, x)$. Without it, any stack of convolutions is still one linear operation, equivalent to a single layer. ReLU is the non-linearity that makes depth mean something. +**ReLU** sets negative values to zero: $\mathrm{ReLU}(x) = \max(0, x)$. Convolutions stacked on convolutions, however many, are still one linear operation that could be folded into a single larger convolution, so the depth would buy nothing. ReLU is the non-linearity that makes depth mean something. -**Max pooling** replaces each 2×2 block with its largest value, halving the width and the height. That does two things: later layers have a quarter as many pixels to process, and a feature gives the same output wherever it falls inside its 2×2 block, so the model is less sensitive to small shifts. +**Max pooling** replaces each 2×2 block with its largest value, halving the width and the height. Two things follow: later layers have a quarter as many pixels to process, and a feature gives the same output wherever it falls inside its block, so the model is less sensitive to small shifts. -One output pixel of the first layer sees a 3×3 patch of the input. After one round of pooling, a 3×3 window in the second layer covers 8×8 of the original image. Deeper layers see wider, which is why shallow layers learn edges and deeper ones learn "a loop" or "a bend". +One output of the first layer sees a 3×3 patch of the input. After one round of pooling, one output of the second layer covers 8×8 of the original image. Deeper layers see wider, which is why shallow layers usually learn edges and only deeper ones learn combinations such as "a loop" or "a bend". ## The whole network -The model in this article has two convolutional blocks and one fully connected layer: +The model here has two convolutional blocks and one fully connected layer: | Layer | Output shape | Parameters | | --- | --- | --- | @@ -106,7 +106,16 @@ The model in this article has two convolutional blocks and one fully connected l | flatten → dense | 10 | 7,850 | | softmax | 10 | 0 | -That is 9,098 parameters in a weights file of about 66 KB, reaching 98.6% accuracy on the MNIST test set.Training was done once in PyTorch and the weights exported as JSON. The TypeScript forward pass has unit tests that compare it with PyTorch's output to within $10^{-4}$. A fully connected network spends far more parameters and still does not catch up: on the same MNIST data, one-hidden-layer networks reached 93.6% with about the same number of parameters (8,755), 97.8–98.0% with ten times as many (91,435), and only 98.3% with forty-five times as many (407,050).Three seeds per width, 15 epochs, best epoch on the test set; script and output are in `docs/research/mlp-baseline/`. +That is 9,098 parameters in a weights file of about 66 KB, reaching 98.6% accuracy on the MNIST test set.Training was run once in PyTorch and the weights exported as JSON. The TypeScript forward pass has unit tests that compare it with PyTorch's output to within $10^{-4}$. + +So few parameters is convolution's doing. For comparison I trained networks with one fully connected hidden layer on the same MNIST data:Three seeds per width, 15 epochs, best epoch on the test set; script and output are in `docs/research/mlp-baseline/`. + +| Model | Parameters | Test accuracy | +| --- | --- | --- | +| This article's CNN | 9,098 | 98.6% | +| Fully connected, about as many | 8,755 | 93.4–93.8% | +| Fully connected, about 10× as many | 91,435 | 97.8–98.0% | +| Fully connected, about 45× as many | 407,050 | 98.2–98.3% | In TypeScript the model is an array: @@ -124,7 +133,7 @@ export const MNIST_CNN: LayerSpec[] = [ ]; ``` -`Sequential.forward()` differs from an ordinary inference library in one deliberate way: it returns the output of **every** layer, not only the final answer. All the figures below depend on that. +`Sequential.forward()` differs from an ordinary inference library in one deliberate way: it returns the output of **every** layer, not only the final answer. The figures that follow all depend on it. ## What the network sees @@ -142,31 +151,29 @@ Draw a few different digits: ## Which pixels actually matter -High probability doesn't mean the model has understood anything. A direct way to find out which part of the image it relies on is to **cover that part and see how far the confidence falls**. +High probability doesn't mean the model has understood anything. The most direct way to find out which part of the image it relies on is to **cover that part and see how far the confidence falls**. -The instrument below slides a 4×4 blank patch across the image two pixels at a time, 13 × 13 = 169 positions in all. It re-runs the network at each position (skipping any where the patch covers only blank paper) and records how far the predicted class's probability drops. +The figure below slides a 4×4 blank patch across the image two pixels at a time, 13 × 13 = 169 positions in all. It re-runs the network at each position (skipping any where the patch covers only blank paper) and records how far the predicted class's probability drops. - + -Occlusion tests one small patch at a time. If the model relies on a combination of two features that are far apart, covering either one alone may barely move the confidence. It's a cheap, intuitive diagnostic, not a complete explanation. +Occlusion tests one small patch at a time. If the model relies on a combination of two features far apart, covering either alone may barely move the confidence. It is a cheap, intuitive diagnostic, not a complete explanation. ## Where it fails Play with it for a while and you'll find ways to break it: -- **Extra strokes** fool it easily, such as a bar through the middle of a 7 or a line under a 1. The training data has almost none of those. -- **Something that isn't a digit** still gets a confident answer. Softmax outputs always sum to 1; the model has no way to say "I don't know". +- **An extra stroke**, such as a bar through the middle of a 7 or a line under a 1, fools it easily: the training data has almost none of those. +- **Something that isn't a digit** still gets a confident answer. Softmax outputs always sum to 1; the model has no way to say "I don't know". On a production line or a surveillance feed a model will sooner or later see something outside its training data, and then high confidence is not the same as being right. (Drawing small, or only in a corner, is fine: preprocessing crops, scales and then centres the drawing by its centre of mass, the same way MNIST was prepared.) -Having no way to say "I don't know" is a serious problem in real systems. On a production line or a surveillance feed, a model will sooner or later see something outside its training distribution, and high confidence is not the same as being right. - ## What comes next -This model has about nine thousand parameters and recognises ten classes. Stack the same bricks (convolution, non-linearity, downsampling) deeper and wider, add residual connections, and you have the backbones that do detection, pose estimation and segmentation on edge devices today. The principle is unchanged; only the scale differs. +This model has just over nine thousand parameters and ten classes. Stack the same bricks (convolution, non-linearity, downsampling) deeper and wider, add residual connections, and you have the backbones that do object detection, pose estimation and segmentation on edge devices today. The principle is unchanged; only the scale differs. -This model was trained beforehand and then brought in. If you want to watch training itself happen in the browser, [№ 004](/en/posts/transformer-from-scratch) trains a Transformer from scratch while you watch its attention matrix take shape. +This model was trained beforehand and then brought in. To watch training itself happen in the browser, [№ 004](/en/posts/transformer-from-scratch) trains a Transformer from scratch while you watch its attention matrix take shape. diff --git a/content/posts/cnn-from-scratch/zh.mdx b/content/posts/cnn-from-scratch/zh.mdx index d56811d..20df89a 100644 --- a/content/posts/cnn-from-scratch/zh.mdx +++ b/content/posts/cnn-from-scratch/zh.mdx @@ -3,7 +3,7 @@ title: 從零開始的 CNN:在瀏覽器裡看見卷積神經網路怎麼「看 description: 不用任何機器學習函式庫,只用 TypeScript 寫出一個能辨識手寫數字的卷積神經網路,然後把它每一層的輸出攤開來看。 seoTitle: "從零開始的 CNN:看卷積網路怎麼「看」" date: 2024-10-12 -updated: 2026-09-20 +updated: 2026-09-22 tags: [computer-vision, cnn, from-scratch] no: 1 featured: true @@ -12,9 +12,9 @@ interactive: true import { ConvStepper, DrawPredict, FeatureMaps, KernelPlayground, OcclusionMap } from "./components"; -卷積神經網路(CNN)的教學通常從一張方塊圖開始:幾個方塊、幾個箭頭,最後吐出一個答案。方塊圖沒有錯,但它把最有趣的部分藏了起來。方塊「裡面」發生了什麼事? +卷積神經網路(CNN)的教學,通常從一張方塊圖開始:幾個方塊、幾個箭頭,最後吐出答案。最有趣的部分反而被藏了起來:方塊「裡面」到底發生了什麼? -這篇文章反過來做。下面這個辨識器已經在你的瀏覽器裡跑起來了,沒有伺服器,也沒有 TensorFlow.js 或 ONNX Runtime。整個前向傳播是大約兩百行 TypeScript。原始碼在這個網站 repo 的 `lib/ml`:`tensor.ts`、`ops.ts`、`sequential.ts` 三個檔案,零相依。 先畫一個數字試試看,接下來我們會把它一層一層拆開。 +這篇反過來做。下面這個辨識器已經在你的瀏覽器裡跑起來了,背後沒有伺服器,也沒有 TensorFlow.js 或 ONNX Runtime,整個前向傳播是兩百多行 TypeScript。原始碼在這個網站 repo 的 `lib/ml`:`tensor.ts`、`ops.ts`、`sequential.ts` 三個檔案,零相依。 先畫一個數字,接著我們一層一層把它拆開。 @@ -22,20 +22,20 @@ import { ConvStepper, DrawPredict, FeatureMaps, KernelPlayground, OcclusionMap } ## 影像只是一格一格的數字 -對模型來說,一張灰階影像就是一個二維陣列,每一格是 0 到 1 之間的數字:0 是空白的紙,1 是墨水。上面中間那張小圖就是這個陣列,總共 $28 \times 28 = 784$ 個數字。 +對模型來說,一張灰階影像就是一個二維陣列,每一格是 0 到 1 之間的數字:0 是白紙,1 是墨水。上面中間那張小圖就是它,共 $28 \times 28 = 784$ 個數字。 -最直覺的做法是把這 784 個數字攤平,全部接進一層全連接層。這行得通,但很浪費:全連接層不知道哪兩個像素是相鄰的。同一個「7」往右移兩格,對它來說就是一組完全不同的輸入,得重新學一次。 +最直覺的做法,是把這 784 個數字攤平,全部接進一層全連接層。這行得通,但很浪費:全連接層不知道哪兩個像素相鄰。同一個「7」往右移兩格,對它來說就是一組全新的輸入,得重新學一次。 -CNN 的出發點是兩個對影像幾乎永遠成立的假設: +CNN 從兩個對影像幾乎永遠成立的假設出發: - **局部性**:有意義的特徵(邊緣、轉角、筆畫端點)只牽涉一小塊相鄰的像素。 -- **平移等變性**:一條垂直邊緣出現在左上角或右下角,都是同一種東西,應該用同一組參數去偵測。卷積做到的是「輸入往右移一格,輸出也跟著往右移一格」(等變);要讓答案真的不隨位置改變(不變),靠的是後面的池化,而且只是近似。 +- **平移等變性**:垂直邊緣不管出現在左上角還是右下角,都是同一種東西,該用同一組參數偵測。卷積保證的是「輸入右移一格,輸出也右移一格」;答案本身不隨位置改變(平移不變性),則要靠後面的池化,而且只是近似。 -把這兩個假設直接寫進模型的結構裡,就得到了卷積。 +把這兩個假設直接寫進模型的結構,就得到卷積。 ## 卷積:一個小窗口滑過整張圖 -卷積做的事情很簡單。拿一個小小的權重矩陣,稱為**卷積核**(kernel),通常是 3×3。把它疊在影像的左上角,九個位置兩兩相乘、全部加起來,得到一個數字。往右滑一格,再算一次。滑完整張圖,這些數字排起來就是一張新的圖,叫做**特徵圖**(feature map)。 +拿一個小小的權重矩陣,稱為**卷積核**(kernel),通常是 3×3。把它疊在影像左上角,九個位置兩兩相乘再全部加起來,得到一個數字;往右滑一格,再算一次。滑完整張圖,這些數字排起來就是一張新的圖,叫做**特徵圖**(feature map)。 寫成式子: @@ -43,15 +43,15 @@ $$ y_{i,j} = b + \sum_{u=0}^{2} \sum_{v=0}^{2} w_{u,v} \, x_{i+u,\, j+v} $$ -下面這個儀器把這個過程放慢。輸入是一條亮的直線(最下面兩列往左歪了一格),卷積核左邊是 −1、右邊是 +1。按「單步」,看每一個輸出格子是怎麼算出來的。 +下面這張圖把過程放慢。輸入是一條亮的直線(最下面兩列往左歪了一格),卷積核左欄是 −1、右欄是 +1。按「單步」,看每個輸出格子怎麼算出來。 -這個卷積核算的其實是「右邊減左邊」。在顏色均勻的區域,左右相減等於零;只有在左暗右亮的交界,輸出才會是大的正數。換句話說,它是一個**垂直邊緣偵測器**,而且不管邊緣在圖的哪個位置,用的都是同樣的九個數字。嚴格來說,深度學習框架裡的「卷積」是數學上的互相關(cross-correlation):卷積核沒有翻轉。因為權重是學出來的,翻不翻沒有差別,大家就沿用了這個名字。 +這個卷積核算的是「右邊減左邊」。顏色均勻的地方相減是零,只有左暗右亮的交界才得到大的正數。它是一個**垂直邊緣偵測器**,而且不管邊緣在哪裡,用的都是同樣九個數字。嚴格來說,深度學習框架裡的「卷積」是數學上的互相關(cross-correlation):卷積核沒有翻轉。因為權重是學出來的,翻不翻沒有差別,大家就沿用了這個名字。 -寫成程式碼就是幾層迴圈。下面是單通道的簡化版;`lib/ml/ops.ts` 裡的版本外面多了批次和通道的迴圈,最裡面這幾層是一樣的: +寫成程式就是四層迴圈。下面是單通道的簡化版;`lib/ml/ops.ts` 裡的版本多了批次和通道的迴圈,最裡面兩層一模一樣: ```ts title="簡化自 lib/ml/ops.ts" {6-8} for (let oy = 0; oy < oH; oy++) { @@ -72,29 +72,29 @@ for (let oy = 0; oy < oH; oy++) { ### 換一組數字,就換一種特徵 -九個數字能做的事情比想像中多。下面的輸入是你剛才畫的數字(沒畫的話是範例的 7),請自己改卷積核試試看。 +九個數字能做的事比想像中多。下面的輸入是你剛才畫的數字(沒畫就是範例的 7),自己改卷積核試試。 -在傳統電腦視覺裡,這些卷積核是人手設計的:Sobel、Laplacian、Gaussian。CNN 的關鍵一步是**不設計了**:把九個數字當成參數,隨機初始化,讓梯度下降自己找出對任務最有用的那幾組。 +傳統電腦視覺的卷積核是人設計的:Sobel、Laplacian、Gaussian。CNN 的關鍵一步是**不設計了**:把這九個數字當成參數,隨機初始化,讓梯度下降自己找出對任務最有用的幾組。 ## ReLU 與池化 -卷積之後通常會接兩個很小的運算。 +卷積之後,通常接兩個很小的運算。 -**ReLU** 把負數歸零:$\mathrm{ReLU}(x) = \max(0, x)$。沒有它,疊再多層卷積,整體仍然是一個線性運算,等同於一層。ReLU 是讓「深度」真正有意義的那個非線性。 +**ReLU** 把負數歸零:$\mathrm{ReLU}(x) = \max(0, x)$。只有卷積疊卷積的話,疊幾層都還是線性運算,可以合併成一個大一點的卷積,深度等於白疊。ReLU 就是讓深度有意義的那個非線性。 -**最大池化**(max pooling)把每個 2×2 的區塊換成其中最大的那個值,長寬各縮一半。它有兩個作用:後面的層要算的像素變成四分之一;而且特徵只要落在那個 2×2 區塊內的任何位置,輸出都一樣,模型對小幅度的位移就沒那麼敏感。 +**最大池化**(max pooling)把每個 2×2 區塊換成其中最大的值,長寬各減半。好處有兩個:後面的層只剩四分之一的像素要算;特徵落在區塊內哪一格,輸出都一樣,模型對小幅位移就沒那麼敏感。 -第一層的一個輸出像素只看得到輸入的 3×3。經過一次池化之後,第二層的 3×3 對應到原圖 8×8 的範圍。越深的層看得越廣,這就是為什麼淺層學到邊緣、深層學到「一個圈」或「一個轉折」。 +第一層的一個輸出只看得到輸入的 3×3。經過一次池化,第二層的一個輸出就涵蓋原圖 8×8 的範圍。越深看得越廣,所以淺層通常學到邊緣,深層才學到「一個圈」或「一個轉折」這種組合。 ## 完整的網路 -這篇文章用的模型只有兩個卷積區塊和一層全連接層: +這篇的模型只有兩個卷積區塊和一層全連接層: | 層 | 輸出形狀 | 參數數量 | | --- | --- | --- | @@ -106,7 +106,16 @@ for (let oy = 0; oy < oH; oy++) { | flatten → dense | 10 | 7,850 | | softmax | 10 | 0 | -總共 9,098 個參數,權重檔大約 66 KB,在 MNIST 測試集上的準確率是 98.6%。訓練是用 PyTorch 做的,只跑一次,然後把權重匯出成 JSON。TypeScript 這邊的前向傳播有單元測試對照 PyTorch 的輸出,誤差在 $10^{-4}$ 以內。 全連接網路要花多得多的參數,而且還追不上:我在同一份 MNIST 上訓練了一層隱藏層的全連接網路,參數量差不多的(8,755 個)只有 93.6%,十倍參數(91,435 個)是 97.8% 到 98.0%,四十五倍(407,050 個)也只到 98.3%。每種寬度 3 個亂數種子、15 輪、取測試集上最好的一輪,腳本和輸出在 `docs/research/mlp-baseline/`。 +總共 9,098 個參數,權重檔約 66 KB,在 MNIST 測試集上的準確率是 98.6%。訓練用 PyTorch 跑了一次,權重匯出成 JSON。TypeScript 的前向傳播有單元測試對照 PyTorch 的輸出,誤差在 $10^{-4}$ 以內。 + +參數這麼少,是卷積的功勞。我在同一份 MNIST 上訓練了只有一層隱藏層的全連接網路做比較:每種寬度 3 個亂數種子、15 輪,取測試集上最好的一輪;腳本和輸出在 `docs/research/mlp-baseline/`。 + +| 模型 | 參數數量 | 測試集準確率 | +| --- | --- | --- | +| 這篇的 CNN | 9,098 | 98.6% | +| 全連接,參數差不多 | 8,755 | 93.4–93.8% | +| 全連接,參數約 10 倍 | 91,435 | 97.8–98.0% | +| 全連接,參數約 45 倍 | 407,050 | 98.2–98.3% | 在 TypeScript 裡,模型就是一個陣列: @@ -124,11 +133,11 @@ export const MNIST_CNN: LayerSpec[] = [ ]; ``` -`Sequential.forward()` 跟一般推論函式庫有一個刻意的差別:它回傳**每一層**的輸出,不只是最後的答案。下面的圖全都靠它。 +`Sequential.forward()` 和一般推論函式庫有一個刻意的不同:它回傳**每一層**的輸出,而不只是最後的答案。後面幾張圖全靠它。 ## 網路看到了什麼 -這是你畫的數字通過網路時,每一層的特徵圖。越亮代表那個位置的反應越強。 +這是你畫的數字通過網路時,每一層的特徵圖,越亮代表那個位置反應越強。 @@ -142,31 +151,29 @@ export const MNIST_CNN: LayerSpec[] = [ ## 哪些像素真的重要 -機率高不代表模型「懂」了。要知道模型依賴影像的哪個部分,有一個很直接的方法:**遮住它,看信心掉多少**。 +機率高,不代表模型「懂」了。想知道它靠影像的哪個部分做判斷,最直接的方法是:**遮住那裡,看信心掉多少**。 -下面的儀器用一塊 4×4 的空白滑過整張圖,每次移 2 格,共 13 × 13 = 169 個位置。每個位置重跑一次網路(遮到的全是空白就跳過),記錄預測類別的機率下降了多少。 +下面這張圖用一塊 4×4 的空白滑過整張圖,每次移 2 格,共 13 × 13 = 169 個位置。每個位置重跑一次網路(遮到的全是空白就跳過),記下預測類別的機率掉了多少。 - + -遮擋法只能一次測一小塊。如果模型依賴的是兩個相距很遠的特徵的組合,單獨遮住任何一個都可能不會讓信心下降太多。它是一個便宜、直覺的診斷工具,不是完整的解釋。 +遮擋法一次只測一小塊。如果模型靠的是兩個相距很遠的特徵的組合,單獨遮住其中一個,信心可能幾乎不變。它是便宜又直覺的診斷工具,但不是完整的解釋。 ## 它會在哪裡失敗 多玩一下,你會找到讓它出錯的方法: -- **加上多餘的筆畫**(例如在 7 中間加一橫、或在 1 底下加一條底線)很容易騙過它。訓練資料裡幾乎沒有這種寫法。 -- **畫不是數字的東西**,它還是會很有信心地回答一個數字。softmax 的輸出加起來一定是 1,模型沒有「我不知道」這個選項。 - -(畫得很小、或只畫在角落倒是沒問題:前處理會先裁切、縮放,再依質心置中,跟 MNIST 的製作方式一樣。) +- **多加一筆**,例如在 7 中間加一橫、在 1 底下加一條底線,很容易騙過它,因為訓練資料裡幾乎沒有這種寫法。 +- **畫不是數字的東西**,它照樣很有信心地回答一個數字。softmax 的輸出加起來一定是 1,模型沒有「我不知道」這個選項。在產線或監視畫面上,模型遲早會看到訓練資料以外的東西,這時高信心不等於正確。 -沒有「我不知道」這個選項,在真實系統裡是大問題。在產線或監視畫面上,模型遲早會看到訓練分佈之外的東西,而高信心不等於正確。 +(畫得很小、或只畫在角落倒是沒問題:前處理會先裁切、縮放,再依質心置中,和 MNIST 的製作方式一樣。) ## 接下來 -這個模型有約九千個參數,辨識十個類別。把同樣的積木(卷積、非線性、降採樣)疊深、加寬、再加上殘差連接,就是今天在邊緣裝置上做偵測、姿態估計和分割的骨幹網路。原理沒有變,只是規模不同。 +這個模型只有九千多個參數、分十個類別。把同樣的積木(卷積、非線性、降採樣)疊深、加寬、再加上殘差連接,就是今天在邊緣裝置上做物件偵測、姿態估計和分割的骨幹網路。原理沒變,變的只是規模。 -這個模型是先訓練好才搬進來的。如果你想看「訓練」本身在瀏覽器裡發生,[№ 004](/zh/posts/transformer-from-scratch) 會從零訓練一個 Transformer,讓你看著注意力矩陣自己長出來。 +這個模型是先訓練好才搬進來的。想親眼看「訓練」在瀏覽器裡發生,[№ 004](/zh/posts/transformer-from-scratch) 會從零訓練一個 Transformer,讓你看著注意力矩陣自己長出來。 From 8d535f495b00b634952bb9f901061649d480e6cb Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 16:18:37 +0800 Subject: [PATCH 03/20] =?UTF-8?q?docs(cnn):=20keep=20=E2=84=96=20001's=20u?= =?UTF-8?q?pdated=20date=20as=20it=20was=20(2026-09-20)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The polish had moved it to today; the date is deliberate and stays. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/cnn-from-scratch/en.mdx | 2 +- content/posts/cnn-from-scratch/zh.mdx | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/content/posts/cnn-from-scratch/en.mdx b/content/posts/cnn-from-scratch/en.mdx index e2f84bc..22e214a 100644 --- a/content/posts/cnn-from-scratch/en.mdx +++ b/content/posts/cnn-from-scratch/en.mdx @@ -3,7 +3,7 @@ title: "A CNN from scratch: watching a convolutional network see, in your browse description: A handwritten-digit classifier written in plain TypeScript with no machine-learning library, then opened up so you can look at the output of every layer. seoTitle: "A CNN from scratch: watch a conv net see" date: 2024-10-12 -updated: 2026-09-22 +updated: 2026-09-20 tags: [computer-vision, cnn, from-scratch] no: 1 featured: true diff --git a/content/posts/cnn-from-scratch/zh.mdx b/content/posts/cnn-from-scratch/zh.mdx index 20df89a..4ddf0f4 100644 --- a/content/posts/cnn-from-scratch/zh.mdx +++ b/content/posts/cnn-from-scratch/zh.mdx @@ -3,7 +3,7 @@ title: 從零開始的 CNN:在瀏覽器裡看見卷積神經網路怎麼「看 description: 不用任何機器學習函式庫,只用 TypeScript 寫出一個能辨識手寫數字的卷積神經網路,然後把它每一層的輸出攤開來看。 seoTitle: "從零開始的 CNN:看卷積網路怎麼「看」" date: 2024-10-12 -updated: 2026-09-22 +updated: 2026-09-20 tags: [computer-vision, cnn, from-scratch] no: 1 featured: true From 0c18557ff444aaeea86a94285e5690dbf9494db6 Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 16:24:48 +0800 Subject: [PATCH 04/20] =?UTF-8?q?docs(flappy):=20polish=20=E2=84=96=20002?= =?UTF-8?q?=20=E2=80=94=20the=20eight=20edits=20Paul=20chose?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - The 75% baseline in the sidenote is derived where it stands (uniform −1…1 weights, each hidden neuron opposite-signed half the time, 1 − ¼); the same-sign sentence says what such a neuron measures instead of stating a rule. - "Usually within twenty-odd generations" becomes "a dozen or so": over 40 seeds the median graduation is generation 15 (docs/research/evolution-seeds/output.txt). - The subtraction is said once less; neuroevolution is named, not re-explained after the opening. - The seed results are one sentence with the median, and say what graduating means; 儀器 becomes 圖. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/ai-flappy-bird/en.mdx | 16 ++++++++-------- content/posts/ai-flappy-bird/zh.mdx | 14 +++++++------- 2 files changed, 15 insertions(+), 15 deletions(-) diff --git a/content/posts/ai-flappy-bird/en.mdx b/content/posts/ai-flappy-bird/en.mdx index fe18ff1..ad39c16 100644 --- a/content/posts/ai-flappy-bird/en.mdx +++ b/content/posts/ai-flappy-bird/en.mdx @@ -1,8 +1,8 @@ --- title: "From rookie to boss bird: 50 birds teach themselves Flappy Bird" -description: Give each of fifty birds a brain of only six weights, add survival of the fittest, and usually within twenty-odd generations they clear a hundred pipes in a row. Then open that brain and see what it actually learned. +description: Give each of fifty birds a brain of only six weights, add survival of the fittest, and usually within a dozen or so generations they clear a hundred pipes in a row. Then open that brain and see what it actually learned. seoTitle: "50 birds teach themselves Flappy Bird" -seoDescription: "Give each of fifty birds a brain of six weights, add survival of the fittest, and within twenty-odd generations they clear a hundred pipes. Then open the brain." +seoDescription: "Give each of fifty birds a brain of six weights, add survival of the fittest, and in a dozen or so generations they clear a hundred pipes. Then open the brain." date: 2025-03-15 updated: 2026-09-20 tags: [ai-agent, neuroevolution, from-scratch] @@ -22,13 +22,13 @@ The fifty birds below are evolving in your browser right now. Watch them hit the ## A brain of six numbers -The drawing on the right of the instrument is everything a bird has: +The brain on the right of the figure above is everything a bird has: - **Two inputs**: the bird's height and the height of the next opening. - **Two hidden neurons.** - **One output**: above 0.5, the bird flaps. -The network has no biases, so its connections carry six weights in all: 2 × 2 + 2 × 1. Putting a neural network and evolution together is called neuroevolution. Instead of computing how each weight should change, you keep a crowd of individuals with different weights and let the ones that do well leave offspring. +The network has no biases, so its connections carry six weights in all: 2 × 2 + 2 × 1. Putting a neural network and evolution together is called neuroevolution. ## How one generation becomes the next @@ -66,15 +66,15 @@ Those three pieces of code are the whole algorithm.The same `Populatio ## Open the brain: what it learned is a subtraction -Once the birds fly well, look at the drawing on the right of the instrument. Brains that evolved successfully tend to look alike: at least one hidden neuron receives "bird height" and "gap height" through lines of **opposite colours**, one positive and one negative. +Once the birds fly well, look at the brain on the right of the figure. Brains that evolved successfully tend to look alike: at least one hidden neuron receives "bird height" and "gap height" through lines of **opposite colours**, one positive and one negative. -One positive plus one negative computes "my height minus the height of the opening": am I above the opening or below it? Below, so flap. This is no coincidence. Both inputs are positive numbers between 0 and 1, so a neuron whose two weights share a sign can only measure how large the two are together; it cannot tell which is higher. Comparing them takes one positive and one negative.I ran 40 random seeds for 60 generations each, and 35 learned. Of those 35 graduating birds, 33 have at least one hidden neuron like this. Take it with a pinch of salt: six purely random weights meet the condition 75% of the time by chance. The script and its output are in `docs/research/evolution-seeds/`. +One positive plus one negative computes "my height minus the height of the opening": am I above the opening or below it? Below, so flap. This is no coincidence. Both inputs are positive numbers between 0 and 1. If the two weights share a sign, then when the bird and the opening both rise, the neuron's output only grows with them: what it measures is how high the two are together, not which is higher. Comparing them takes one positive and one negative.I ran 40 random seeds for 60 generations each, and 35 learned. Of those 35 graduating birds, 33 have at least one hidden neuron like this. Take it with a pinch of salt: the starting weights are uniform between −1 and 1, so each hidden neuron has an even chance of one positive and one negative; both missing it has a chance of ¼, and a random brain meets the condition 1 − ¼ = 75% of the time. The script and its output are in `docs/research/evolution-seeds/`. -Six numbers, and what they learned is a subtraction. That subtraction is not in any line of code. It is simply the one combination of six numbers, out of all of them, that lived longest. +That subtraction is not in any line of code: it is simply the one combination of six numbers, out of all of them, that lived longest. ## Evolution doesn't guarantee success -I ran 12 different random seeds for 60 generations each: 10 graduated within 22 generations, one took until generation 43, and in one the whole population got stuck in the same bad habit and never cleared more than 3 pipes in all 60 generations. With 40 seeds, 5 never learned. +I ran 12 different random seeds for 60 generations each: 11 learned, 10 of them graduating (a hundred pipes in a row) within 22 generations; in the last one the whole population got stuck in the same bad habit and never cleared more than 3 pipes in all 60. Over 40 seeds, 35 learned, and the median graduation came in generation 15. This is the usual weakness of evolutionary methods. With no gradient to say which way is better, evolution can only stumble on better individuals by luck. If the whole flock looks alike and all of it is bad, the 20% of fresh blood takes a long time to turn things round. If the score refuses to climb, press Restart and begin with a new set of ancestors. diff --git a/content/posts/ai-flappy-bird/zh.mdx b/content/posts/ai-flappy-bird/zh.mdx index edde96b..f5cf667 100644 --- a/content/posts/ai-flappy-bird/zh.mdx +++ b/content/posts/ai-flappy-bird/zh.mdx @@ -1,6 +1,6 @@ --- title: 從菜鳥到鳥霸王:讓 50 隻小鳥自己學會 Flappy Bird -description: 給五十隻小鳥各一顆只有六個權重的大腦,加上優勝劣汰,多半二十幾代之內牠們就能連過一百根管子。然後打開那顆大腦,看牠到底學到了什麼。 +description: 給五十隻小鳥各一顆只有六個權重的大腦,加上優勝劣汰,多半十幾代牠們就能連過一百根管子。然後打開那顆大腦,看牠到底學到了什麼。 seoTitle: "讓 50 隻小鳥自己學會 Flappy Bird" date: 2025-03-15 updated: 2026-09-20 @@ -21,13 +21,13 @@ import { FlappyLab } from "./components"; ## 一顆只有六個數字的大腦 -儀器右邊那張圖就是一隻鳥的全部: +上面圖右邊那顆大腦,就是一隻鳥的全部: - **輸入**兩個:鳥現在的高度,和下一個開口的高度。 - **隱藏層**兩個神經元。 - **輸出**一個:超過 0.5 就拍翅膀。 -這個網路沒有偏置(bias),所以連線的權重總共六個:2 × 2 + 2 × 1。把神經網路和演化放在一起的做法叫神經進化(neuroevolution):不去算「每個權重該怎麼改」,而是養一群權重各不相同的個體,讓表現好的留下後代。 +這個網路沒有偏置(bias),所以連線的權重總共六個:2 × 2 + 2 × 1。把神經網路和演化放在一起,就叫神經進化(neuroevolution)。 ## 一代是怎麼變成下一代的 @@ -65,15 +65,15 @@ private breed(a: Float32Array, b: Float32Array): Float32Array { ## 打開大腦看:牠學到的是一個減法 -等牠們飛得不錯之後,看儀器右邊那張圖。演化成功的鳥腦通常長得很像:至少有一個隱藏神經元,接到它的「鳥的高度」和「開口高度」兩條線**顏色相反**,也就是一正一負。 +等牠們飛得不錯之後,看圖右邊那顆大腦。演化成功的鳥腦通常長得很像:至少有一個隱藏神經元,接到它的「鳥的高度」和「開口高度」兩條線**顏色相反**,也就是一正一負。 -一正一負相加,算的就是「我的高度減掉開口的高度」,也就是「我比開口高還是低」。低了就拍。這不是巧合:兩個輸入都是 0 到 1 之間的正數,兩個權重同號的神經元只量得出「兩者加起來有多大」,分不出誰高誰低。要比較,就得一正一負。我用 40 組隨機種子各跑 60 代,35 組學會了;這 35 隻畢業的鳥有 33 隻至少有一個這樣的隱藏神經元。要打個折扣看:六個權重全是亂數時,也有 75% 的機率碰巧符合。腳本和輸出在 `docs/research/evolution-seeds/`。 +一正一負相加,算的就是「我的高度減掉開口的高度」,也就是「我比開口高還是低」。低了就拍。這不是巧合:兩個輸入都是 0 到 1 之間的正數,如果兩個權重同號,鳥和開口一起升高時,這個神經元的輸出也只會一起變大;它量到的是兩者加起來多高,不是誰比誰高。要比較,就得一正一負。我用 40 組隨機種子各跑 60 代,35 組學會了;這 35 隻畢業的鳥有 33 隻至少有一個這樣的隱藏神經元。要打個折扣看:初始權重在 −1 到 1 之間均勻分佈,每個隱藏神經元有一半機率一正一負,兩個都不是的機率是 ¼,所以亂數的鳥腦也有 1 − ¼ = 75% 碰巧符合。腳本和輸出在 `docs/research/evolution-seeds/`。 -六個數字,學到的是一個減法。這個減法不在任何一行程式裡;它只是六個數字的所有組合當中,活得最久的那一種。 +這個減法不在任何一行程式裡:它只是六個數字的所有組合當中,活得最久的那一種。 ## 演化不保證成功 -我用 12 組不同的隨機種子各跑了 60 代:10 組在 22 代之內畢業,1 組拖到第 43 代,還有 1 組整個族群卡在同一種壞習慣裡,60 代下來最多只過了 3 根管子。把種子加到 40 組,有 5 組沒學會。 +我用 12 組不同的隨機種子各跑了 60 代:11 組學會,其中 10 組在 22 代之內畢業(連過 100 根管子);剩下 1 組整個族群卡在同一種壞習慣裡,60 代都只過得了 3 根管子。擴大到 40 組,35 組學會,中位數在第 15 代畢業。 這是演化式方法的通病。它沒有梯度告訴它「往哪邊走會更好」,只能靠運氣碰到更好的個體;如果整群都長得差不多,又剛好都不好,那 20% 的新血要很久才翻得了盤。如果你看到分數一直上不去,按「重新開始」換一批祖先就好。 From 684032d5463ec6c09c24e98e3d4a5dfa18cae202 Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 16:28:39 +0800 Subject: [PATCH 05/20] =?UTF-8?q?docs(trading):=20polish=20=E2=84=96=20003?= =?UTF-8?q?=20=E2=80=94=20the=20nine=20edits=20Paul=20chose,=20and=20one?= =?UTF-8?q?=20number=20measured?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - "About 10 ms a generation on my laptop" had no source: measured on the M4 Pro, 8.8–9.0 ms (population 50, 252 days, mean of 30 generations, three seeds, two runs); script and output in docs/research/trading-timing/. - +34% (buy-and-hold) beside "rose 35%" is explained: $10,000 buys 53 shares at $185.64 and $161 stays in cash. - "Don't trust it" is said fewer times: the sentence after the random-bot numbers goes, the six-seeds paragraph joins the warning, and "can illustrate but not place an order" gives way to the closing line. - The defaults and the 20-generation numbers move into the sidenote; what the winners do is three bullets. - 儀器 becomes 圖; the bots are 它們, not 牠們; sell-high, buy-low is buy-low, sell-high. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/trading-agent/en.mdx | 22 +++++++++------- content/posts/trading-agent/zh.mdx | 26 ++++++++++--------- docs/research/trading-timing/output.txt | 5 ++++ .../trading-timing/timing.test.ts.txt | 17 ++++++++++++ 4 files changed, 48 insertions(+), 22 deletions(-) create mode 100644 docs/research/trading-timing/output.txt create mode 100644 docs/research/trading-timing/timing.test.ts.txt diff --git a/content/posts/trading-agent/en.mdx b/content/posts/trading-agent/en.mdx index f2304be..69dc7e5 100644 --- a/content/posts/trading-agent/en.mdx +++ b/content/posts/trading-agent/en.mdx @@ -12,7 +12,7 @@ interactive: true import { TradingLab } from "./components"; -The conclusion first: the instrument below trains a trading bot in a few seconds. Given the full thirty generations, that bot makes about **+43%** on Apple's 2024 share price. Buying at the start of the same year and doing nothing makes **+34%**. +The conclusion first: the figure below trains a trading bot in a few seconds. Given the full thirty generations, that bot makes about **+43%** on Apple's 2024 share price. Buying at the start of the same year and doing nothing makes **+34%**. The number is real, but it doesn't mean the bot can trade. Half of this article is about how it was trained; the other half is about why you shouldn't believe it. Train one yourself first. @@ -20,7 +20,7 @@ The number is real, but it doesn't mean the bot can trade. Half of this article -Training runs entirely in your browser, and the result is stored there, so it is still here the next time you visit.It uses the same `lib/ml/neuroevolution.ts` as [the Flappy Bird entry](/en/posts/ai-flappy-bird). With a population of 50, one generation takes about 10 ms on my laptop, so no GPU is needed. +Training runs entirely in your browser, and the result is stored there, so it is still here the next time you visit.It uses the same `lib/ml/neuroevolution.ts` as [the Flappy Bird entry](/en/posts/ai-flappy-bird). With a population of 50, one generation takes about 9 ms on my M4 Pro (method and output in `docs/research/trading-timing/`), so no GPU is needed. ## How it is trained @@ -44,31 +44,33 @@ export function observe(closes: number[], day: number): Float32Array { } ``` -The three parameters on the instrument map onto the steps above: **generations** is how many times to repeat, **population** is how many bots each generation has, and **mutation rate** is the chance that each weight is randomly changed. Press Train and the evolution curve climbs generation by generation, ending about nine percentage points above the dashed buy-and-hold line. +The three parameters on the figure map onto the steps above: **generations** is how many times to repeat, **population** is how many bots each generation has, and **mutation rate** is the chance that each weight is randomly changed. Press Train and the evolution curve climbs generation by generation, ending about nine percentage points above the dashed buy-and-hold line. It looks as if they understand this market better and better. ## What it actually learned -I trained six different random seeds for thirty generations each (population 50 and mutation rate 15%, all of them the instrument's defaults; stopping at 20 generations gives +41.5% to +43.8%).Every number in this section comes from `docs/research/evolution-seeds/`. The seeds are fixed, so a re-run gives exactly the same results. Every return landed between +42.9% and +43.9%, and the winners all did nearly the same thing: **most of the buying happens in the first half of the year, while the price is still low (the year's lowest close is $165 on 19 April), and by year end the bot is almost fully invested.** In between it makes a few dozen small sell-high, buy-low round trips, eight or nine in ten of them profitable. +I trained six different random seeds for thirty generations each, with the figure's defaults.Every number in this section comes from `docs/research/evolution-seeds/`. The seeds are fixed, so a re-run gives exactly the same results. The defaults are a population of 50 and a mutation rate of 15%; stopping at 20 generations gives +41.5% to +43.8%. Every return landed between +42.9% and +43.9%, and the winners all did nearly the same thing: -Apple rose 35% over 2024. In a year like that, any strategy that fills up while the price is low and holds to the end beats filling up at $185 on 2 January. +- **Most of the buying happens in the first half of the year**, while the price is still low (the year's lowest close is $165 on 19 April). +- **By year end the bot is almost fully invested.** +- In between it makes a few dozen small buy-low, sell-high round trips, eight or nine in ten of them profitable. -Now look again at where the evolution curve starts. In all six seeds the best bot of the **first generation**, the best of 50 random bots that have not evolved at all, already makes +39% to +41%, above the dashed line every time. I also generated 2,000 random bots: 13% beat buy-and-hold, and the best made +41%. Thirty generations of evolution squeeze out only about three more percentage points. A baseline that guessing can beat says something about this year's prices and these rules, not about the bot. +Apple rose 35% over 2024 (buy-and-hold makes +34% because $10,000 buys only 53 shares, and the $161 left over never goes in). In a year like that, any strategy that fills up while the price is low and holds to the end beats filling up at $185 on 2 January. + +Now look again at where the evolution curve starts. In all six seeds the best bot of the **first generation**, the best of 50 random bots that have not evolved at all, already makes +39% to +41%, above the dashed line every time. I also generated 2,000 random bots: 13% beat buy-and-hold, and the best made +41%. Thirty generations of evolution squeeze out only about three more percentage points. -The bot is trained on one stretch of prices and scored on the same stretch. It didn't learn to spot a low. Evolution picked, out of a thousand or so candidate strategies, the ones that happened to buy at this year's low. On another year's data there is no reason for these strategies to keep working. +The bot is trained on one stretch of prices and scored on the same stretch. It didn't learn to spot a low. Evolution picked, out of a thousand or so candidate strategies, the ones that happened to buy at this year's low. On another year's data there is no reason for these strategies to keep working. Six seeds converging on the same behaviour looks at first like "it found a pattern"; the more plausible explanation is the opposite: this year has one obvious low, and every road to a high score passes through it. -Six seeds converging on the same behaviour looks at first like "it found a pattern". The more plausible explanation is the opposite: this year has one obvious low, and every road to a high score passes through it. - ## Three faces of the same mistake - **Overfitting.** Evolution finds any pattern in historical data that raises the score, whether or not the pattern means anything. The +43% above is a live example. - **It has seen one kind of market.** There is one stock here and one bull year, so the habit the bots pick up is "fill up and hold". In a falling year the same habit loses badly. - **What it never saw, it cannot have learned.** Sudden news, policy changes, liquidity drying up: if it didn't happen in the training data, the bot has no response to it. -Making this experiment honest would take at least three things: split the data into a training period and a test period and report results only on the test period; add trading costs; score across several stocks and several kinds of market. This page does none of them, so its numbers can illustrate the problem but not place an order. +Making this experiment honest would take at least three things: split the data into a training period and a test period and report results only on the test period; add trading costs; score across several stocks and several kinds of market. This page does none of them. What the bots on this page do best is find the best script for a stretch of history whose ending is already known. diff --git a/content/posts/trading-agent/zh.mdx b/content/posts/trading-agent/zh.mdx index 0c676b1..4c8117d 100644 --- a/content/posts/trading-agent/zh.mdx +++ b/content/posts/trading-agent/zh.mdx @@ -12,7 +12,7 @@ interactive: true import { TradingLab } from "./components"; -先講結論:下面這個儀器可以在幾秒內訓練出一支交易機器人。練滿三十代,它在 2024 年的蘋果股價上賺 **+43%** 左右,而同一年買進之後什麼都不做只有 **+34%**。 +先講結論:下面這張圖可以在幾秒內訓練出一支交易機器人。練滿三十代,它在 2024 年的蘋果股價上賺 **+43%** 左右,而同一年買進之後什麼都不做只有 **+34%**。 這個數字是真的,但它不代表機器人會交易。這篇文章有一半在講它是怎麼練出來的,另一半在講為什麼你不該相信它。先自己練一支。 @@ -20,7 +20,7 @@ import { TradingLab } from "./components"; -訓練完全在你的瀏覽器裡執行,結果會存在瀏覽器裡,下次回來還在。用的是和 [Flappy Bird 那篇](/zh/posts/ai-flappy-bird)同一套 `lib/ml/neuroevolution.ts`。族群 50 時,一代在我的筆電上大約 10 毫秒,所以不需要 GPU。 +訓練完全在你的瀏覽器裡執行,結果會存在瀏覽器裡,下次回來還在。用的是和 [Flappy Bird 那篇](/zh/posts/ai-flappy-bird)同一套 `lib/ml/neuroevolution.ts`。族群 50 時,一代在我的 M4 Pro 上約 9 毫秒(量法和輸出在 `docs/research/trading-timing/`),所以不需要 GPU。 ## 它是怎麼練出來的 @@ -28,7 +28,7 @@ import { TradingLab } from "./components"; 1. 隨機產生 50 個機器人。 2. 讓每一個在這一年的股價上交易一遍,用年底的報酬率打分數。 -3. 留下成績最好的幾個,讓牠們兩兩混合權重、加上一點隨機突變,生出下一代;另外再補幾個全新的隨機個體,免得整個族群卡在同一種做法上。 +3. 留下成績最好的幾個,讓它們兩兩混合權重、加上一點隨機突變,生出下一代;另外再補幾個全新的隨機個體,免得整個族群卡在同一種做法上。 4. 重複。 每個機器人的大腦是一個 30 → 24 → 3 的小網路。輸入是過去 30 天每天的漲跌幅(百分比),輸出三個分數,分別代表「不動」「買」「賣」,取最高的那個: @@ -44,31 +44,33 @@ export function observe(closes: number[], day: number): Float32Array { } ``` -儀器上的三個參數對應的就是上面的步驟:**世代數**是重複幾次,**族群大小**是每一代有幾個機器人,**突變率**是每個權重被隨機改動的機率。按下訓練,進化曲線一代一代往上爬,最後停在買進持有那條虛線上方九個百分點左右。 +圖上的三個參數對應的就是上面的步驟:**世代數**是重複幾次,**族群大小**是每一代有幾個機器人,**突變率**是每個權重被隨機改動的機率。按下訓練,進化曲線一代一代往上爬,最後停在買進持有那條虛線上方九個百分點左右。 -看起來牠們越來越懂這個市場。 +看起來它們越來越懂這個市場。 ## 它到底學到了什麼 -我用六組不同的隨機種子各訓練了三十代(族群 50、突變率 15%,全部是儀器的預設值;只練 20 代的話是 +41.5% 到 +43.8%)。這一節的數字都來自 `docs/research/evolution-seeds/`,種子固定,重跑會得到一模一樣的結果。 報酬率全部落在 +42.9% 到 +43.9%,而且贏家做的事幾乎一樣:**大部分的買入集中在股價還低的上半年(全年最低點是 4 月 19 日的 165 美元),到年底幾乎滿倉**,中間再做幾十次小幅的高賣低買,其中八到九成是賺錢的。 +我用六組不同的隨機種子,照圖上的預設值各訓練了三十代。這一節的數字都來自 `docs/research/evolution-seeds/`,種子固定,重跑會得到一模一樣的結果。預設值是族群 50、突變率 15%;只練 20 代的話,報酬率是 +41.5% 到 +43.8%。 報酬率全部落在 +42.9% 到 +43.9%,而且贏家做的事幾乎一樣: -2024 年蘋果整年上漲了 35%。在這種行情裡,任何「趁便宜買滿、抱到年底」的策略都會贏過「一月二日就用 185 美元買滿」。 +- **大部分的買入集中在上半年**,股價還低的時候(全年最低點是 4 月 19 日的 165 美元)。 +- **到年底幾乎滿倉。** +- 中間做幾十次小幅的低買高賣,其中八到九成是賺錢的。 -再回頭看進化曲線的起點。六組種子的**第一代**,也就是完全沒有演化過的 50 個亂數機器人裡最好的那一個,已經是 +39% 到 +41%,全部在虛線上面。我另外產生了 2,000 個亂數機器人:13% 贏過買進持有,最好的一個 +41%。三十代的演化只在這上面多擠出三個百分點左右。亂猜就能贏過的基準線,說明的是這一年的行情和這套規則,不是機器人。 +2024 年蘋果整年上漲了 35%(買進持有是 +34%,因為一萬美元只買得起 53 股,剩下的 161 美元沒有進場)。在這種行情裡,任何「趁便宜買滿、抱到年底」的策略都會贏過「一月二日就用 185 美元買滿」。 + +再回頭看進化曲線的起點。六組種子的**第一代**,也就是完全沒有演化過的 50 個亂數機器人裡最好的那一個,已經是 +39% 到 +41%,全部在虛線上面。我另外產生了 2,000 個亂數機器人:13% 贏過買進持有,最好的一個 +41%。三十代的演化只在這上面多擠出三個百分點左右。 -機器人是在同一段股價上訓練、也在同一段股價上評分的。它並沒有學會「看出低點」,而是演化過程從上千個候選策略中,挑出了那些剛好在這一年的低點出手的。換一年的數據,這些策略沒有任何理由繼續有效。 +機器人是在同一段股價上訓練、也在同一段股價上評分的。它並沒有學會「看出低點」,而是演化過程從上千個候選策略中,挑出了那些剛好在這一年的低點出手的。換一年的數據,這些策略沒有任何理由繼續有效。六組種子收斂到同一個做法,乍看像是「它找到了規律」;比較合理的解釋正好相反:這一年只有一個明顯的低點,通往高分的路都經過那裡。 -六組種子收斂到同一個做法,乍看像是「它找到了規律」。比較合理的解釋正好相反:這一年只有一個明顯的低點,通往高分的路都經過那裡。 - ## 同一個錯誤的三種樣子 - **過度擬合。** 演化會在歷史資料裡找到任何能提高分數的模式,不管那個模式有沒有意義。上面的 +43% 就是一個活生生的例子。 - **只見過一種行情。** 這裡只有一檔股票、一個多頭年,所以機器人養成的習慣是「買滿之後就抱著」。換成一個下跌的年份,同樣的習慣會賠得很慘。 - **沒見過的事不可能學到。** 突發的新聞、政策變化、流動性消失,訓練資料裡沒發生過,機器人就沒有任何應對。 -要讓這個實驗誠實一點,至少要做三件事:把資料切成訓練期和測試期,只在測試期上報成績;加入交易成本;在多檔股票、多種行情上評分。這一頁一件都沒做,所以它的數字只能拿來說明問題,不能拿來下單。 +要讓這個實驗誠實一點,至少要做三件事:把資料切成訓練期和測試期,只在測試期上報成績;加入交易成本;在多檔股票、多種行情上評分。這一頁一件都沒做。 這一頁的機器人最會的,是在一段已經知道結局的歷史裡找到最好的劇本。 diff --git a/docs/research/trading-timing/output.txt b/docs/research/trading-timing/output.txt new file mode 100644 index 0000000..d1ef34e --- /dev/null +++ b/docs/research/trading-timing/output.txt @@ -0,0 +1,5 @@ +seed 1: 8.96 ms per generation (population 50, 252 days, mean of 30) +seed 2: 9.04 ms per generation (population 50, 252 days, mean of 30) +seed 3: 8.75 ms per generation (population 50, 252 days, mean of 30) + +Measured 2026-09-22 on an Apple M4 Pro, Node v22.19.0 (V8, as in Chrome), two runs of the script above (copy it into tests/ to rerun). diff --git a/docs/research/trading-timing/timing.test.ts.txt b/docs/research/trading-timing/timing.test.ts.txt new file mode 100644 index 0000000..9990bb8 --- /dev/null +++ b/docs/research/trading-timing/timing.test.ts.txt @@ -0,0 +1,17 @@ +import { writeFileSync } from "node:fs"; +import { it } from "vitest"; +import data from "@/content/posts/trading-agent/aapl-2024.json"; +import { Trainer } from "@/content/posts/trading-agent/components/trading"; +import { mulberry32 } from "@/lib/ml"; + +it("times one generation", () => { + const lines: string[] = []; + for (const seed of [1, 2, 3]) { + const t = new Trainer({ closes: data.closes, size: 50, mutationRate: 0.15, rng: mulberry32(seed) }); + for (let i = 0; i < 3; i++) t.step(); // warm up the JIT + const n = 30, start = performance.now(); + for (let i = 0; i < n; i++) t.step(); + lines.push(`seed ${seed}: ${((performance.now() - start) / n).toFixed(2)} ms per generation (population 50, 252 days, mean of ${n})`); + } + writeFileSync("docs/research/trading-timing/output.txt", lines.join("\n") + "\n"); +}); From 307d5a7b2178ee9f8dc54a246c27424b518f8e7c Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 16:37:51 +0800 Subject: [PATCH 06/20] =?UTF-8?q?docs(transformer):=20polish=20=E2=84=96?= =?UTF-8?q?=20004=20=E2=80=94=20two=20corrections,=20one=20measured=20numb?= =?UTF-8?q?er,=20and=20the=20edits=20Paul=20chose?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - The gradient check's 10⁻⁵ is ε; the tolerance is 10⁻⁶ (tests/ml/autograd.test.ts, transformer.test.ts). - "One layer can do one lookup" contradicted "one layer learns to sort": it learns it, in twenty times the steps of reversing (650–800 against 30–40), with a solution hard to read. - "Fast finishes in a second or two" is measured: the readout first shows 100% after 0.80–0.88 s (M4 Pro, three loads); noted in docs/research/transformer-steps/output.txt, and the sidenote says why the on-screen accuracy (a rolling average) reaches 100% later than the script's step count. - Six digits, not six-digit numbers; LayerNorm's learned scale and shift; the 12-slot position table and "same structure as GPT" said once each; the twelve operations move into a sidenote; 儀器 becomes 圖. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/transformer-from-scratch/en.mdx | 20 +++++++++---------- content/posts/transformer-from-scratch/zh.mdx | 20 +++++++++---------- docs/research/transformer-steps/output.txt | 5 +++++ 3 files changed, 25 insertions(+), 20 deletions(-) diff --git a/content/posts/transformer-from-scratch/en.mdx b/content/posts/transformer-from-scratch/en.mdx index ba4921e..13efe0a 100644 --- a/content/posts/transformer-from-scratch/en.mdx +++ b/content/posts/transformer-from-scratch/en.mdx @@ -20,11 +20,11 @@ A person doesn't need to learn that, but at the start the model doesn't even kno Press Train and watch the line “what it writes now” turn from pink (wrong) to cyan (right), digit by digit. - + -If all goes well, three things happen together within a hundred steps: the loss falls to nearly zero, the answer it writes turns cyan from end to end, and the lower half of one attention map grows a **diagonal running from upper right to lower left**.Measured offline with five fixed random seeds (`docs/research/transformer-steps/`): reverse and copy first reach 100% between step 30 and step 40, and a step takes 4 to 5 ms on my laptop. When you press the button the random numbers are new, so your step count will differ a little. Which of the two heads grows the diagonal varies from run to run. +If all goes well, three things happen together within a hundred steps: the loss falls to nearly zero, the answer it writes turns cyan from end to end, and the lower half of one attention map grows a **diagonal running from upper right to lower left**.Measured offline with five fixed random seeds (`docs/research/transformer-steps/`): reverse and copy first reach 100% between step 30 and step 40, and a step takes 4 to 5 ms on my laptop. When you press the button the random numbers are new, so your step count will differ a little; the accuracy on screen is a rolling average of recent answers, so it reaches 100% somewhat later than this. Which of the two heads grows the diagonal varies from run to run. I didn't draw that line, and no line of code tells it to appear. It is a result of training. This article is about how it gets there. @@ -79,7 +79,7 @@ $$ $M$ is the **causal mask**: positions with $j > i$ get $-\infty$, so their weight after softmax is zero. While writing the second digit of the answer, the model can't peek at the third. That is why the upper-right triangle of every attention map stays dark. -The two maps in the instrument are the output of $\mathrm{softmax}(\cdot)$ itself: how bright row $i$, column $j$ is tells you how much position $i$ read from position $j$. +The two attention maps in the figure are the output of $\mathrm{softmax}(\cdot)$ itself: how bright row $i$, column $j$ is tells you how much position $i$ read from position $j$. ```ts title="lib/ml/transformer.ts" const scores = tape.scale(tape.matmul(qh, tape.transpose(kh)), 1 / Math.sqrt(dh)); @@ -106,7 +106,7 @@ Sort doesn't give such a clean picture. What sorting needs isn't "read position Attention moves information between positions. Every other part works on each position separately: 1. **Embedding.** Each token looks up a 32-dimensional vector, and a vector meaning "I am at position n" is added to it. Without position vectors the model couldn't tell the first digit from the sixth, and reversing would be impossible to learn. -2. **LayerNorm.** Rescales each position's vector to mean 0 and variance 1, which keeps training stable. +2. **LayerNorm.** Rescales each position's vector to mean 0 and variance 1, then multiplies by a learned scale and adds a learned shift, which keeps training stable. 3. **MLP.** Two fully connected layers that widen to four times the width and come back. Attention fetches information; the MLP processes it. 4. **Residual connections.** Each sub-layer's output is added to the stream instead of replacing it. 5. **Unembedding.** Finally the 32-dimensional vector is projected to a score for each of the 11 tokens. @@ -142,23 +142,23 @@ matmul(a: Mat, b: Mat): Mat { } ``` -This Transformer uses only twelve kinds of operation: matrix product, add, add bias, scale, transpose, ReLU, LayerNorm, causal softmax, splitting and re-joining attention heads, embedding lookup, and the cross-entropy loss at the end. After the forward pass, run the recorded functions **in reverse** and every parameter has its gradient. That is backpropagation. +This Transformer uses only twelve kinds of operation.Matrix product, add, add bias, scale, transpose, ReLU, LayerNorm, causal softmax, splitting and re-joining attention heads, embedding lookup, and the cross-entropy loss at the end. After the forward pass, run the recorded functions **in reverse** and every parameter has its gradient. That is backpropagation. -The frightening thing about backpropagation is that a wrong formula still runs; the model simply, mysteriously, learns badly. So every operation in this engine has a unit test that compares its analytic gradient with a numerical one, $\frac{f(x + \varepsilon) - f(x - \varepsilon)}{2\varepsilon}$, to within $10^{-5}$, and the whole Transformer is spot-checked the same way in every parameter matrix. +The frightening thing about backpropagation is that a wrong formula still runs; the model simply, mysteriously, learns badly. So every operation in this engine has a unit test that compares its analytic gradient with a numerical one, $\frac{f(x + \varepsilon) - f(x - \varepsilon)}{2\varepsilon}$ with $\varepsilon = 10^{-5}$, to within $10^{-6}$, and the whole Transformer is spot-checked the same way in every parameter matrix. With gradients in hand, all that remains is updating the parameters. This uses Adam, which keeps a running average of each parameter's gradient and of its square, and uses them to give every parameter its own step size. -Every step uses 16 brand-new random problems, and their gradients are averaged into one update. There are a million possible six-digit strings; in a hundred steps the model has seen at most 1,600 of them, yet it answers ones it has never seen. It hasn't memorised answers. It has learned the rule. +Every step uses 16 brand-new random problems, and their gradients are averaged into one update. There are a million possible strings of six digits; in a hundred steps the model has seen at most 1,600 of them, yet it answers ones it has never seen. It hasn't memorised answers. It has learned the rule. ## How this differs from a real large language model -Structurally, hardly at all. The difference is scale and everything that comes with it: +The difference is scale and everything that comes with it: - **Parameters.** Fourteen thousand here; 175 billion in GPT-3. -- **Layers.** One here. The sort task above already hints at what depth is for: one layer can do one lookup, and composing several steps takes more layers. +- **Layers.** One here. One layer can learn to sort too, but it takes twenty times the steps of reversing and its solution is hard to read; more layers let a model split the work into steps, and that is usually what depth is for. - **Data.** Here the data is unlimited, noise-free, and every problem has exactly one right answer. Real text is none of those. -- **Position encoding.** This model uses learned absolute position vectors and its position table has only 12 slots, so it handles exactly six digits. A seven-digit input doesn't even fit. +- **Position encoding.** This model uses learned absolute position vectors (the length limit is above). The core is the same, though: predict the next token, compute the loss, send the gradient back, and nudge every parameter slightly in the right direction. Those few seconds you watched in your browser were the whole of it. diff --git a/content/posts/transformer-from-scratch/zh.mdx b/content/posts/transformer-from-scratch/zh.mdx index a4268f4..9d53b56 100644 --- a/content/posts/transformer-from-scratch/zh.mdx +++ b/content/posts/transformer-from-scratch/zh.mdx @@ -20,11 +20,11 @@ import { TrainingLab } from "./components"; 按「開始訓練」,看「它現在寫出來的」那一行從粉紅色(錯)一位一位變成青色(對)。 - + -如果一切順利,你會在一百步之內看到三件事同時發生:損失掉到接近零、它寫出來的答案整串變成青色,以及其中一張注意力圖的下半部長出一條**從右上到左下的斜線**。我用五組固定的隨機種子離線量過(`docs/research/transformer-steps/`):反轉和複製在第 30 到 40 步第一次達到 100%,每一步在我的筆電上約 4 到 5 毫秒。你按下按鈕時用的是新的亂數,步數會差一點。哪一個頭負責長出斜線,每次訓練都不一定。 +如果一切順利,你會在一百步之內看到三件事同時發生:損失掉到接近零、它寫出來的答案整串變成青色,以及其中一張注意力圖的下半部長出一條**從右上到左下的斜線**。我用五組固定的隨機種子離線量過(`docs/research/transformer-steps/`):反轉和複製在第 30 到 40 步第一次達到 100%,每一步在我的筆電上約 4 到 5 毫秒。你按下按鈕時用的是新的亂數,步數會差一點;畫面上的答對率是最近幾題的滾動平均,會比這裡晚一些才到 100%。哪一個頭負責長出斜線,每次訓練都不一定。 那條斜線不是我畫上去的,也沒有任何一行程式碼叫它長成那樣。它是訓練的結果。這篇文章要講的就是:它是怎麼長出來的。 @@ -79,7 +79,7 @@ $$ $M$ 是**因果遮罩**:$j > i$ 的位置填上 $-\infty$,softmax 之後權重就是零。模型在寫答案的第 2 位時,不能偷看第 3 位。上面注意力圖的右上三角永遠是暗的,就是這個原因。 -儀器裡畫的那兩張圖,就是 $\mathrm{softmax}(\cdot)$ 的結果本身:第 $i$ 列第 $j$ 行有多亮,代表位置 $i$ 從位置 $j$ 讀了多少。 +圖裡的那兩張注意力圖,就是 $\mathrm{softmax}(\cdot)$ 的結果本身:第 $i$ 列第 $j$ 行有多亮,代表位置 $i$ 從位置 $j$ 讀了多少。 ```ts title="lib/ml/transformer.ts" const scores = tape.scale(tape.matmul(qh, tape.transpose(kh)), 1 / Math.sqrt(dh)); @@ -106,7 +106,7 @@ mixed.push(tape.matmul(weights, vh)); 注意力負責在位置之間搬運資訊,其他零件都是逐位置運作的: 1. **Embedding**:每個 token 查表得到一個 32 維向量,再加上一個代表「我在第幾個位置」的向量。沒有位置向量,模型根本分不出第 1 位和第 6 位,反轉任務就不可能學會。 -2. **LayerNorm**:把每個位置的向量調整成平均 0、變異數 1,讓訓練穩定。 +2. **LayerNorm**:把每個位置的向量調整成平均 0、變異數 1,再乘上學出來的縮放、加上偏移,讓訓練穩定。 3. **MLP**:兩層全連接,中間放大成 4 倍寬再縮回來。注意力把資訊搬過來,MLP 負責處理它。 4. **殘差連接**:每個子層的輸出是「加回去」而不是「取代」。 5. **Unembedding**:最後把 32 維向量投影回 11 個 token 的分數。 @@ -142,23 +142,23 @@ matmul(a: Mat, b: Mat): Mat { } ``` -這個 Transformer 用到的運算只有十二種:矩陣乘法、加法、加偏置、縮放、轉置、ReLU、LayerNorm、因果 softmax、切開和接回注意力頭、embedding 查表,以及最後的交叉熵損失。前向跑完之後,把記下來的函式**倒著**執行一遍,每個參數的梯度就都算好了。這就是反向傳播。 +這個 Transformer 只用到十二種運算。矩陣乘法、加法、加偏置、縮放、轉置、ReLU、LayerNorm、因果 softmax、切開和接回注意力頭、embedding 查表,以及最後的交叉熵損失。 前向跑完之後,把記下來的函式**倒著**執行一遍,每個參數的梯度就都算好了。這就是反向傳播。 -反向傳播最可怕的地方是:公式寫錯了,程式還是跑得動,模型只是莫名其妙學不好。所以這個引擎的每一種運算都有單元測試,用數值微分 $\frac{f(x + \varepsilon) - f(x - \varepsilon)}{2\varepsilon}$ 去對照解析梯度,誤差要在 $10^{-5}$ 以內;整個 Transformer 也用同樣的方法抽查了每一個參數矩陣。 +反向傳播最可怕的地方是:公式寫錯了,程式還是跑得動,模型只是莫名其妙學不好。所以這個引擎的每一種運算都有單元測試,用 $\varepsilon = 10^{-5}$ 的數值微分 $\frac{f(x + \varepsilon) - f(x - \varepsilon)}{2\varepsilon}$ 去對照解析梯度,誤差要在 $10^{-6}$ 以內;整個 Transformer 也用同樣的方法抽查了每一個參數矩陣。 有了梯度,剩下的就是更新參數。這裡用的是 Adam:它替每個參數各自記錄梯度的移動平均和平方的移動平均,藉此決定每個參數自己的步長。 -每一步是 16 題全新的隨機題目,梯度取平均之後更新一次。六位數總共有一百萬種組合,模型在一百步裡最多只看過其中 1,600 種,卻能答對沒看過的題目。它不是把答案背起來,而是學到了規則。 +每一步是 16 題全新的隨機題目,梯度取平均之後更新一次。六個數字總共有一百萬種組合,模型在一百步裡最多只看過其中 1,600 種,卻能答對沒看過的題目。它不是把答案背起來,而是學到了規則。 ## 這和真正的大型語言模型差在哪裡 -結構上幾乎沒有差別。差別在規模,以及規模帶來的一切: +差別在規模,以及規模帶來的一切: - **參數**:這裡是 1.4 萬個,GPT-3 是 1,750 億個。 -- **層數**:這裡是 1 層。上面的排序任務已經暗示了深度的用處:一層只能做「一次查找」,要組合多個步驟就需要更多層。 +- **層數**:這裡是 1 層。一層也學得會排序,但步數是反轉的 20 倍,而且解法很難讀懂;多層讓模型可以把步驟拆開來做,這通常就是深度的用處。 - **資料**:這裡的資料是無限的、沒有雜訊的,而且任務有唯一正確答案。真實的文字三者皆非。 -- **位置編碼**:這裡用的是學出來的絕對位置向量,而且位置表只有 12 格,所以它只會處理剛好六位數,七位數連放都放不進去。 +- **位置編碼**:這裡用的是學出來的絕對位置向量(長度限制見前面)。 不過核心是一樣的:預測下一個 token、算出損失、把梯度傳回去、把每個參數往對的方向推一點點。你剛才在瀏覽器裡看到的那幾秒鐘,就是這件事的全部。 diff --git a/docs/research/transformer-steps/output.txt b/docs/research/transformer-steps/output.txt index 6034276..df9b2ab 100644 --- a/docs/research/transformer-steps/output.txt +++ b/docs/research/transformer-steps/output.txt @@ -3,3 +3,8 @@ copy: first 100% at step 35, 35, 30, 40, 40 (seeds 1-5); loss at step 100: 0.010 sort: first 100% at step 800, 725, 650, 675, 725 (seeds 1-5); loss at step 100: 0.344, 0.214, 0.238, 0.216, 0.216; 4.3 ms per step on this machine (this machine: Apple-silicon Mac, Node via vitest, 2026-09-20) + +In the page (2026-09-22, Chrome via Playwright, Apple M4 Pro, reverse, speed "fast", three fresh loads): the +accuracy readout, a rolling window of recent answers, first showed 100 after 881, 813 and 803 ms (step 156, 149, +148); about 200 steps a second. The readout lags the script's accuracy(40) above, which is why its step count is +higher than 30–40. From 36eff1a2547e8035e83f7635907a285ff192fdcb Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 16:41:44 +0800 Subject: [PATCH 07/20] =?UTF-8?q?docs(hydranet):=20polish=20=E2=84=96=2000?= =?UTF-8?q?5=20=E2=80=94=20the=20ten=20edits=20Paul=20chose?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - The distribution head's edge error is 1.15 px in docs/research/hydranet-spike: shown as 1.2 px (was 1.1). - The three extra box-head attempts become rows of the table, with their edge errors (2.1, 2.0 and 1.8 px from the same results); the prose now says the best of them beats the mask's bounding box by 0.01, where it used to say lost. - The first-run anecdote keeps its direction, not its unrecorded numbers; "never the compute" is "often not". - 45 parameters and the two-object failure are each said once; the Karpathy and literature paragraphs are shorter or split; 儀器 becomes 圖, and loss is 損失 in the Chinese prose. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/hydranet-fruit/en.mdx | 25 +++++++++++++---------- content/posts/hydranet-fruit/zh.mdx | 31 +++++++++++++++++------------ 2 files changed, 33 insertions(+), 23 deletions(-) diff --git a/content/posts/hydranet-fruit/en.mdx b/content/posts/hydranet-fruit/en.mdx index 46f53dc..b5a2234 100644 --- a/content/posts/hydranet-fruit/en.mdx +++ b/content/posts/hydranet-fruit/en.mdx @@ -33,7 +33,7 @@ Your numbers will differ, for an interesting reason that comes up below. ## The data is free -The expensive part of training an object detector was never the compute. It is the labelling: someone has to drag every box and trace every outline. Here nobody does, because we draw the pictures ourselves. +The expensive part of training an object detector is often not the compute. It is the labelling: someone has to drag every box and trace every outline. Here nobody does, because we draw the pictures ourselves. We draw an emoji on a transparent canvas at a random size, position and angle, mirrored or not at random, then paste it over a noisy background. Once it is drawn, the canvas's **alpha channel** already records exactly which pixels are fruit. The mask is the alpha, and the box is the rectangle around the mask: @@ -48,7 +48,7 @@ if (x < x0) x0 = x; if (x > x1) x1 = x; ``` -There is a side effect. Every operating system ships different emoji artwork: Apple, Google and Microsoft each draw their own apple.If your system has no colour emoji font (some Linux setups), the instrument switches to coloured shapes and tells you so. So what you trained a moment ago is a detector for *your system's* fruit. Take weights trained on an iPhone to an Android phone and the score will drop. Machine learning calls that **domain shift**, and you happen to be holding a live example. +There is a side effect. Every operating system ships different emoji artwork: Apple, Google and Microsoft each draw their own apple.If your system has no colour emoji font (some Linux setups), the figure switches to coloured shapes and tells you so. So what you trained a moment ago is a detector for *your system's* fruit. Take weights trained on an iPhone to an Android phone and the score will drop. Machine learning calls that **domain shift**, and you happen to be holding a live example. ## One body, two heads @@ -62,7 +62,7 @@ Drawing a box and drawing a mask look like two jobs, but they need almost the sa | Box head | One 1×1 convolution | 36 | | Total | | 5,053 | -Look at the last two rows: **the two heads together have 45 parameters**, under 1% of the network. All of the effort goes into the shared part. This structure is called a HydraNet, after the many-headed Hydra: one body, many heads. Andrej Karpathy used the word in 2019 when describing Tesla's perception stack: many heads on one shared trunk, handling lane lines, traffic lights, pedestrians and other tasks at once. The reason is practical. The compute in a car is fixed, and sharing the trunk is how everything fits. +Look at the last two rows: **the two heads together have 45 parameters**, under 1% of the network. All of the effort goes into the shared part. This structure is called a HydraNet, after the many-headed Hydra: one body, many heads. Andrej Karpathy used the word in 2019 when describing Tesla's perception stack, for a practical reason: the compute in a car is fixed, and sharing the trunk is how that many tasks fit. The neck deserves one more sentence. To see widely, the trunk reduces resolution all the way to 8×8, but a mask needs detail. So after enlarging the deep features, the neck **joins them back** with the shallow features that still have that detail. This is U-Net's skip connection. @@ -79,7 +79,7 @@ The offline controls (same architecture, same three seeds): Both tasks do better alone. A plausible explanation is that they compete for the capacity of one trunk, and this trunk has fewer than four thousand parameters; I haven't tested that explanation separately.This conclusion has four limits: I didn't tune the weights of the two losses, there are only three seeds, the data is geometric shapes instead of emoji, and the architecture is an older one with a heavier skip connection (6,078 parameters, not the 5,053 in the table above). The careful way to put it is "no positive transfer observed", not "transfer is negative". -So why share at all? Because of how the bill adds up. In this model the two heads have 45 parameters, and nearly all the computation is in the trunk and neck. A box-only network is therefore no faster than the two-headed one, and getting a box and a mask separately means computing the whole trunk twice. **Sharing the trunk saves nearly half the compute**, at the price of a little accuracy in each task.How much you save depends on how heavy the heads are. The offline experiment used an older architecture with heavier heads and measured 1.7×; the lighter the heads, the closer it gets to 2×. On a device with fixed compute that is usually a good trade, and it gets better with more heads, because the trunk's cost is spread over more of them. +So why share at all? Because of how the bill adds up. As said above, the heads cost almost nothing; nearly all the computation is in the trunk and neck. A box-only network is therefore no faster than the two-headed one, and getting a box and a mask separately means computing the whole trunk twice. **Sharing the trunk saves nearly half the compute**, at the price of a little accuracy in each task.How much you save depends on how heavy the heads are. The offline experiment used an older architecture with heavier heads and measured 1.7×; the lighter the heads, the closer it gets to 2×. On a device with fixed compute that is usually a good trade, and it gets better with more heads, because the trunk's cost is spread over more of them. Now measure it yourself. Three networks start from the same random weights and see the same images: @@ -87,7 +87,9 @@ Now measure it yourself. Three networks start from the same random weights and s -The time column is steady: the two single-head networks together take about twice as long as the two-headed one. The accuracy columns are another matter. On my first run boxes were better when trained jointly (0.810 against 0.781) and masks were better alone (0.807 against 0.778), which agrees with only half of the offline result above. Press the button a few more times and you'll see the gaps grow, shrink and sometimes flip. That is the point: **one experiment cannot support a claim like "multi-task learning helps"; supporting it takes many runs.** The literature has results in both directions. Standley and colleagues measured systematically in 2020 which vision tasks are worth learning together, and the answer was that it depends on the combination: some pairs help each other, some hurt. +The time column is steady: the two single-head networks together take about twice as long as the two-headed one. The accuracy columns are another matter. On my first run boxes were better when trained jointly and masks were better alone, which agrees with only half of the offline result above. Press the button a few more times and you'll see the gaps grow, shrink and sometimes flip. + +That is the point: **one experiment cannot support a claim like "multi-task learning helps"; supporting it takes many runs.** The literature has results in both directions. Standley and colleagues measured systematically in 2020 which vision tasks are worth learning together, and the answer was that it depends on the combination: some pairs help each other, some hurt. ## Why boxes are hard @@ -98,12 +100,15 @@ My first version took the obvious route: cut the image into an 8×8 grid, let th | How the box is produced | Box IoU | Error per edge | | --- | --- | --- | | Choose a cell, then regress four distances | 0.56 | 2.2 px | +| The same, with a finer 16×16 grid | 0.57 | 2.1 px | +| The same, with an L1 loss | 0.60 | 2.0 px | +| The same, with 16×16, the shallow-feature skip and an L1 plus GIoU loss all at once | 0.64 | 1.8 px | | No box head: take the bounding box of the predicted mask | 0.63 | – | -| **Turn each edge into a distribution (next section)** | **0.74** | **1.1 px** | +| **Turn each edge into a distribution (next section)** | **0.74** | **1.2 px** | -The second row stings: a carefully trained box head lost to one line of code that takes the minimum and maximum of the mask. I also tried a finer 16×16 grid and a different loss, and IoU only went from 0.56 to 0.57 and 0.60; even with the 16×16 grid, the shallow-feature skip that the third row also uses, and an L1 plus GIoU loss all at once, it reached only 0.64. All of these were patches. The problem wasn't a detail; it was the output format itself, "choose a cell, then guess distances from that cell". Choose the wrong cell and the regression that follows is wasted, and gradients flow back only through the one cell that was chosen. +The second-to-last row stings: a carefully trained box head lost to one line of code that takes the minimum and maximum of the mask, and with every patch in the three rows between, it beat that line by only 0.01. All of these were patches. The problem wasn't a detail; it was the output format itself, "choose a cell, then guess distances from that cell". Choose the wrong cell and the regression that follows is wasted, and gradients flow back only through the one cell that was chosen. -The small line "box taken from the mask" in the instrument above is that baseline running live in your browser, for comparison with the box head's IoU. +The small line "box taken from the mask" in the figure above is that baseline running live in your browser, for comparison with the box head's IoU. ## Turning an edge into a distribution @@ -120,7 +125,7 @@ $x_i$ is the centre of bin $i$. This is called integral regression, or soft-argm - **It is differentiable end to end.** There is no non-differentiable "choose a cell" step, so gradients reach every bin. - **The answer can fall between two bins.** Half the probability on each of two neighbours puts the expected value midway, so there is no quantisation error. The one exception is at the image border: the expected value cannot go beyond the centres of the first and last bins, so an edge flush with the border is off by up to 1 pixel. -- **It can be drawn directly.** The four small plots at the lower right of the instrument are these four distributions. Violet bars are probabilities, the solid cyan line is the expected value, and the dashed pink line is the right answer. +- **It can be drawn directly.** The four small plots at the lower right of the figure are these four distributions. Violet bars are probabilities, the solid cyan line is the expected value, and the dashed pink line is the right answer. ```ts title="simplified from content/posts/hydranet-fruit/components/model.ts" const maps = t.conv2d(d, P.boxK, P.boxB, { h: HALF, w: HALF, k: 1 }); @@ -132,7 +137,7 @@ box = t.matmul(t.concatRows([lr, tb]), this.positions); // an expected value is Go back up and train again, this time watching the four small plots. At first all four distributions are flat, the expected values sit in the middle, and the box collapses to a small patch in the centre. Then each distribution grows a peak, the peak sharpens, and it slides to the edge of the fruit. The loss pushes it there. -Each edge has a single distribution, so the head can describe only one box. I expected that with two pieces of fruit the box would land between them. When I measured it, that turned out to be wrong: about seven times in ten it **boxes one of them and ignores the other** (details under "Where it fails"). Either way it isn't the right answer. Real detectors handle any number of objects, so they go back to predicting at every position, as CenterNet and FCOS do: the very family that lost in the table above. +Each edge has a single distribution, so the head can describe only one box. I expected that with two pieces of fruit the box would land between them. When I measured it, that turned out to be wrong: mostly it **boxes one of them and ignores the other** (details under "Where it fails"). Either way it isn't the right answer. Real detectors handle any number of objects, so they go back to predicting at every position, as CenterNet and FCOS do: the very family that lost in the table above. ## Where it fails diff --git a/content/posts/hydranet-fruit/zh.mdx b/content/posts/hydranet-fruit/zh.mdx index b446f42..79dfad8 100644 --- a/content/posts/hydranet-fruit/zh.mdx +++ b/content/posts/hydranet-fruit/zh.mdx @@ -33,7 +33,7 @@ import { HeadsRace, LabelsFigure, TrainingLab } from "./components"; ## 資料不用錢 -訓練物件偵測最貴的從來不是算力,而是標註:得有人一張一張把框拉好、把輪廓描出來。這裡完全不用,因為圖是我們自己畫的。 +訓練物件偵測最貴的往往不是算力,而是標註:得有人一張一張把框拉好、把輪廓描出來。這裡完全不用,因為圖是我們自己畫的。 做法是在一張透明的畫布上畫一顆 emoji,隨機決定大小、位置、旋轉角度、要不要左右翻轉,再把它貼到一張有雜訊的背景上。畫完之後,畫布的 **alpha 通道**(透明度)就已經精確地記錄了「哪些像素是水果」。遮罩就是 alpha,框就是遮罩的外接矩形: @@ -48,7 +48,7 @@ if (x < x0) x0 = x; if (x > x1) x1 = x; ``` -這件事有一個副作用。每個作業系統的 emoji 是不同的圖:Apple、Google、Microsoft 各畫各的蘋果。如果你的系統沒有彩色 emoji 字型(某些 Linux 環境),儀器會自動改用彩色幾何圖形,並且告訴你。 所以你剛才訓練的,是「你的系統的水果」的偵測器。把在 iPhone 上訓練好的權重拿到 Android 上用,成績會掉。這就是機器學習裡說的 **domain shift**,而你手上剛好有一個活生生的例子。 +這件事有一個副作用。每個作業系統的 emoji 是不同的圖:Apple、Google、Microsoft 各畫各的蘋果。如果你的系統沒有彩色 emoji 字型(某些 Linux 環境),圖會自動改用彩色幾何圖形,並且告訴你。 所以你剛才訓練的,是「你的系統的水果」的偵測器。把在 iPhone 上訓練好的權重拿到 Android 上用,成績會掉。這就是機器學習裡說的 **domain shift**,而你手上剛好有一個活生生的例子。 ## 一個身體,兩個頭 @@ -62,7 +62,7 @@ if (x > x1) x1 = x; | 框頭 | 一個 1×1 卷積 | 36 | | 合計 | | 5,053 | -看最後兩列:**兩個頭加起來只有 45 個參數**,不到全部的 1%。網路的力氣全花在共用的那一段。這個結構叫 HydraNet,名字來自九頭蛇:一個身體,很多顆頭。Andrej Karpathy 在 2019 年介紹 Tesla 的感知系統時用的就是這個詞:同一個共用主幹上接了許多顆頭,同時處理車道線、號誌、行人等等不同的任務。理由很實際:車上的算力是固定的,共用主幹才塞得下。 +看最後兩列:**兩個頭加起來只有 45 個參數**,不到全部的 1%。網路的力氣全花在共用的那一段。這個結構叫 HydraNet,名字來自九頭蛇:一個身體,很多顆頭。Andrej Karpathy 在 2019 年介紹 Tesla 的感知系統時用的就是這個詞,理由很實際:車上的算力是固定的,共用主幹才塞得下那麼多任務。 頸部那一步值得多說一句。主幹為了看得廣,把解析度一路降到 8×8,但畫遮罩需要細節。所以頸部把深層特徵放大之後,會把還保有細節的淺層特徵**接回來**一起用。這是 U-Net 的 skip connection。 @@ -77,9 +77,9 @@ if (x > x1) x1 = x; | 遮罩 IoU | 0.825 | 0.812 | | 框 IoU | 0.794 | 0.735 | -兩個任務都是單獨訓練比較好。一個合理的解釋是它們在搶同一個主幹的容量,而這個主幹只有三千多個參數;這個解釋我沒有另外驗證。這個結論有四個限制:兩個 loss 的權重我沒有調過、只有三組種子、資料是幾何圖形而不是 emoji、架構是 skip 接法比較重的舊版(6,078 個參數,不是上表的 5,053)。所以比較保守的說法是「沒有看到正遷移」,而不是「一定是負遷移」。 +兩個任務都是單獨訓練比較好。一個合理的解釋是它們在搶同一個主幹的容量,而這個主幹只有三千多個參數;這個解釋我沒有另外驗證。這個結論有四個限制:兩個損失的權重我沒有調過、只有三組種子、資料是幾何圖形而不是 emoji、架構是 skip 接法比較重的舊版(6,078 個參數,不是上表的 5,053)。所以比較保守的說法是「沒有看到正遷移」,而不是「一定是負遷移」。 -那為什麼還要共用?因為帳要這樣算。在這個模型裡,兩個頭只有 45 個參數,幾乎所有的計算都花在主幹和頸部。所以「只有框」的網路並不比「兩個一起」的網路快,想分別得到框和遮罩,就得把主幹整個算兩次。**共用主幹省了將近一半的算力**,代價是各掉一點準確度。省多少取決於頭有多重。離線實驗裡用的是頭比較重的舊架構,量到的是 1.7 倍;頭越輕,越接近 2 倍。 在算力固定的裝置上,這筆交易通常划算;而且頭越多越划算,因為主幹的成本被更多顆頭分攤。 +那為什麼還要共用?因為帳要這樣算。前面說過,頭幾乎不花計算,幾乎所有的計算都在主幹和頸部。所以「只有框」的網路並不比「兩個一起」的網路快,想分別得到框和遮罩,就得把主幹整個算兩次。**共用主幹省了將近一半的算力**,代價是各掉一點準確度。省多少取決於頭有多重。離線實驗裡用的是頭比較重的舊架構,量到的是 1.7 倍;頭越輕,越接近 2 倍。 在算力固定的裝置上,這筆交易通常划算;而且頭越多越划算,因為主幹的成本被更多顆頭分攤。 下面讓你自己量一次。三個網路從同一組隨機權重出發、看同一批圖: @@ -87,7 +87,9 @@ if (x > x1) x1 = x; -時間那一欄很穩定:兩個單頭網路合計大約是雙頭的兩倍。準確度那兩欄就不是了。我第一次跑的時候,框是一起訓練比較好(0.810 對 0.781),遮罩是單獨訓練比較好(0.807 對 0.778),和上面離線實驗的結論只對了一半。多按幾次,你會看到差距忽大忽小,有時候反過來。這本身就是重點:**一次實驗不能支持「多任務有幫助」這種結論,要支持它得跑很多次。** 文獻上兩個方向的結果都有。Standley 等人 2020 年的論文系統性地量過哪些視覺任務適合一起學,結論是「看任務組合而定」,有些組合互相幫忙,有些互相傷害。 +時間那一欄很穩定:兩個單頭網路合計大約是雙頭的兩倍。準確度那兩欄就不是了。我第一次跑的時候,框是一起訓練比較好、遮罩是單獨訓練比較好,和上面離線實驗的結論只對了一半。多按幾次,你會看到差距忽大忽小,有時候反過來。 + +這本身就是重點:**一次實驗不能支持「多任務有幫助」這種結論,要支持它得跑很多次。** 文獻上兩個方向的結果都有。Standley 等人 2020 年的論文系統性地量過哪些視覺任務適合一起學,結論是「看任務組合而定」,有些組合互相幫忙,有些互相傷害。 ## 框為什麼這麼難 @@ -98,12 +100,15 @@ if (x > x1) x1 = x; | 框的做法 | 框 IoU | 每邊誤差 | | --- | --- | --- | | 選一格,再回歸四個距離 | 0.56 | 2.2 px | +| 同上,格子加密到 16×16 | 0.57 | 2.1 px | +| 同上,改用 L1 損失 | 0.60 | 2.0 px | +| 同上,16×16、接回淺層特徵、L1 加 GIoU 損失全部一起 | 0.64 | 1.8 px | | 不訓練框頭,直接對預測的遮罩取外框 | 0.63 | – | -| **把每條邊變成一個分佈(下一節)** | **0.74** | **1.1 px** | +| **把每條邊變成一個分佈(下一節)** | **0.74** | **1.2 px** | -第二列很傷人:辛苦訓練出來的框頭,輸給了「對遮罩取最小值和最大值」這種一行程式。我也試過把格子加密到 16×16、換一種 loss,IoU 只從 0.56 變成 0.57 和 0.60;把 16×16、第三列也用到的淺層特徵接回、L1 加 GIoU loss 全部一起給它,也只到 0.64。都是小修小補。問題不在細節,而在「選一格、再從那一格猜距離」這個輸出方式本身:只要選錯格,後面的回歸全部白費,而且梯度只從被選中的那一格流回去。 +倒數第二列很傷人:辛苦訓練出來的框頭,輸給了「對遮罩取最小值和最大值」這種一行程式;中間三列的修補全部用上,也只贏它 0.01。都是小修小補。問題不在細節,而在「選一格、再從那一格猜距離」這個輸出方式本身:只要選錯格,後面的回歸全部白費,而且梯度只從被選中的那一格流回去。 -上面的儀器裡有一行小字「由遮罩取外框」,就是這個基準線在你的瀏覽器裡的即時成績,可以拿來和框頭的 IoU 比。 +上面的圖裡有一行小字「由遮罩取外框」,就是這個基準線在你的瀏覽器裡的即時成績,可以拿來和框頭的 IoU 比。 ## 把邊界變成分佈 @@ -120,7 +125,7 @@ $x_i$ 是第 $i$ 格的中心位置。這個做法叫 integral regression,也 - **整條路都可微**。不需要「選一格」這種不可微的動作,梯度會流到每一格。 - **答案可以落在兩格之間**。兩格各佔一半機率,期望值就在中間,所以沒有量化誤差。唯一的例外在圖的邊緣:期望值出不了第一格和最後一格的中心,貼著圖邊的邊界會差最多 1 個像素。 -- **可以直接畫出來**。儀器右下方的四張小圖就是這四個分佈。紫色長條是機率,青色實線是期望值,粉紅色虛線是正確答案。 +- **可以直接畫出來**。圖右下方的四張小圖就是這四個分佈。紫色長條是機率,青色實線是期望值,粉紅色虛線是正確答案。 ```ts title="簡化自 content/posts/hydranet-fruit/components/model.ts" const maps = t.conv2d(d, P.boxK, P.boxB, { h: HALF, w: HALF, k: 1 }); @@ -129,10 +134,10 @@ const tb = t.softmax(t.marginal(t.sliceRows(maps, 2, 2), { h: HALF, w: HALF, axi box = t.matmul(t.concatRows([lr, tb]), this.positions); // 期望值就是和位置向量做內積 ``` -回到上面重新訓練一次,這次盯著那四張小圖。一開始四條分佈都是平的,期望值落在正中間,所以框縮成中央一小塊。接著你會看到每一條分佈長出一個峰,峰變尖,然後滑向水果的邊緣。這是 loss 把它推過去的。 +回到上面重新訓練一次,這次盯著那四張小圖。一開始四條分佈都是平的,期望值落在正中間,所以框縮成中央一小塊。接著你會看到每一條分佈長出一個峰,峰變尖,然後滑向水果的邊緣。這是損失把它推過去的。 -四條邊各只有一個分佈,所以它只能描述一個框。我原本以為畫面上有兩顆水果時,框會落在兩顆中間;實際量了才發現不是。大約七成的時候它是**挑其中一顆框住,無視另一顆**(細節在下面「它會在哪裡失敗」)。不管是哪一種,都不是正確答案。真實的偵測器要處理任意多個物體,所以還是得回到「每個位置各自預測」的做法,例如 CenterNet 和 FCOS,也就是上表裡輸掉的那一類。 +四條邊各只有一個分佈,所以它只能描述一個框。我原本以為畫面上有兩顆水果時,框會落在兩顆中間;實際量了才發現,它多半是**挑其中一顆框住,無視另一顆**(細節在下面「它會在哪裡失敗」)。不管是哪一種,都不是正確答案。真實的偵測器要處理任意多個物體,所以還是得回到「每個位置各自預測」的做法,例如 CenterNet 和 FCOS,也就是上表裡輸掉的那一類。 ## 它會在哪裡失敗 @@ -147,7 +152,7 @@ box = t.matmul(t.concatRows([lr, tb]), this.positions); // 期望值就是和位 ## 這和真的系統差在哪 - **物體數量**:真實場景有任意多個物體,框頭要換成每個位置各自預測的設計。 -- **loss 權重**:這裡框的 loss 固定乘上 10,再和遮罩的 loss 相加,沒有調過。頭一多,怎麼配重就是大問題。Kendall 等人 2018 年提出讓網路自己學每個任務的不確定性,用它來決定權重。 +- **損失的權重**:這裡框的損失固定乘上 10,再和遮罩的損失相加,沒有調過。頭一多,怎麼配重就是大問題。Kendall 等人 2018 年提出讓網路自己學每個任務的不確定性,用它來決定權重。 - **誰來標註**:這裡的標註是免費的。真實系統的做法常常是用又大又慢的基礎模型離線產生標註,再拿去訓練一個又小又快的多頭網路,讓它在邊緣裝置上每一幀只跑一次前向傳播。這正是我在工作上做的事。 - **規模**:五千個參數對上幾百萬個。原理一樣。 From 7d4bc6059238fd9042a39dda47a0cb1dbc930d3b Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 16:46:11 +0800 Subject: [PATCH 08/20] =?UTF-8?q?fix(a11y):=20the=20last=20two=20short=20t?= =?UTF-8?q?ouch=20targets=20=E2=80=94=20the=20home=20page's=20view-all=20l?= =?UTF-8?q?ink=20and=20the=20copy=20button?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found by paul-67's recheck of f9398cc. Measured in a touch context at 390: 查看全部文章 is 78×44 (was 84×24; it now has the tap area), 複製程式碼 44×44 (was 42: the ::before sits inside the 1 px border, so -inset-2 gave 26 + 16). On desktop the copy button still sits 8 px from the top and right corner, dark and light. Co-Authored-By: Claude Opus 5 (1M context) --- app/[locale]/page.tsx | 2 +- components/mdx/code-block.tsx | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/app/[locale]/page.tsx b/app/[locale]/page.tsx index 1856238..9f60233 100644 --- a/app/[locale]/page.tsx +++ b/app/[locale]/page.tsx @@ -149,7 +149,7 @@ export default async function HomePage({ params }: PageProps<"/[locale]">) {

{t.home.indexLead}

- + {t.home.viewAll} diff --git a/components/mdx/code-block.tsx b/components/mdx/code-block.tsx index e394990..7f67b12 100644 --- a/components/mdx/code-block.tsx +++ b/components/mdx/code-block.tsx @@ -22,7 +22,7 @@ export function CodeBlock(props: ComponentProps<"pre">) { type="button" onClick={copy} aria-label={copied ? t.copied : t.copy} - className="absolute right-2 top-2 grid size-7 place-items-center rounded-md border border-border bg-background/80 text-muted-foreground transition-opacity before:absolute before:-inset-2 hover:text-foreground focus-visible:opacity-100 [@media(hover:hover)]:opacity-0 [@media(hover:hover)]:group-hover/code:opacity-100" + className="absolute right-2 top-2 grid size-7 place-items-center rounded-md border border-border bg-background/80 text-muted-foreground transition-opacity before:absolute before:-inset-[9px] hover:text-foreground focus-visible:opacity-100 [@media(hover:hover)]:opacity-0 [@media(hover:hover)]:group-hover/code:opacity-100" > {copied ? : } From 7f09b33493117d62745d2e235b0876302e3f7a49 Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 16:49:55 +0800 Subject: [PATCH 09/20] =?UTF-8?q?docs(lite3):=20polish=20=E2=84=96=20006?= =?UTF-8?q?=20=E2=80=94=20ten=20edits,=20one=20number=20measured?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - "Under 0.2 ms a call" had no record: measured on the M4 Pro, 151.8–154.6 µs (mean of 5,000 calls, three runs); script and output in docs/research/lite3-spike/policy-timing*. It says about 0.15 ms, and the "83 times a second" it repeated goes. - The foot slip at μ 0.05 is 0.27 m/s (the note's table), not 0.26; the weights are 757 KB (189,324 float32), the size of the file the page loads, where the article said "a 758 KB ONNX file"; PPO is "most likely", read from a folder name; the push table says left and right (+y, −y), not "in" and "out". - The noise results are a table; the long parenthesis about where the gain speeds come from is a sidenote; "I have not checked the training setup" is said once; 儀器 becomes 圖. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/lite3-walking/en.mdx | 26 ++++++++++++------- content/posts/lite3-walking/zh.mdx | 26 ++++++++++++------- .../lite3-spike/policy-timing-output.txt | 5 ++++ .../lite3-spike/policy-timing.test.ts.txt | 18 +++++++++++++ 4 files changed, 57 insertions(+), 18 deletions(-) create mode 100644 docs/research/lite3-spike/policy-timing-output.txt create mode 100644 docs/research/lite3-spike/policy-timing.test.ts.txt diff --git a/content/posts/lite3-walking/en.mdx b/content/posts/lite3-walking/en.mdx index b1909ff..1a54a3a 100644 --- a/content/posts/lite3-walking/en.mdx +++ b/content/posts/lite3-walking/en.mdx @@ -28,7 +28,7 @@ Nothing in this dog's program says what to do when pushed, or how to lift a foot ## The brain is four matrix multiplications -The published policy is a 758 KB ONNX file. Open it and there is nothing special inside: +The published policy is an ONNX file whose weights come to 757 KB (189,324 32-bit floats). Open it and there is nothing special inside: | | This dog's brain | | --- | --- | @@ -55,7 +55,7 @@ ELU differs from ReLU only on the negative side: ReLU cuts to zero, ELU is a smo I took the actions ONNX Runtime computes from the original ONNX file as the reference and compared them with this code's output: they differ in the sixth decimal place.The physics was checked too: from the same starting pose and the same command, MuJoCo as WebAssembly (run in Node) and MuJoCo in Python both end a ten-second walk at 4.69 m. With the collision meshes replaced by convex hulls, the version on this page ends at 4.70. -The network is called 83 times a second and each call takes under 0.2 ms (measured in Node). The physics is what costs time. +Each call of the network takes about 0.15 ms (measured in Node); the physics is what costs time.M4 Pro, Node 22, the mean of 5,000 calls in a row, three runs at 151.8 to 154.6 µs; script and output in `docs/research/lite3-spike/`. ## What it can feel @@ -112,7 +112,7 @@ The price is that the network grew up in a body with $K_p = 30$ and $K_d = 1$, a -Pull $K_p$ down to 10 and the solid line cannot keep up with the dashed one; the dog slowly sinks to the floor. Push it to 100 and the solid line hugs the dashed one, and the dog walks faster than you asked. The forward speeds I measured (averages over 10 s, `docs/research/lite3-spike/shipped-build-output.txt`; the instrument's readout also counts sideways drift and leaves out the start, so it reads higher than this table): +Pull $K_p$ down to 10 and the solid line cannot keep up with the dashed one; the dog slowly sinks to the floor. Push it to 100 and the solid line hugs the dashed one, and the dog walks faster than you asked. The forward speeds I measured:Averages over 10 s, `docs/research/lite3-spike/shipped-build-output.txt`. The figure's readout also counts sideways drift and leaves out the start, so it reads higher than this table. | $K_p$ | $K_d$ | Actual speed when asked for 0.5 | | --- | --- | --- | @@ -136,7 +136,7 @@ Push it and see. Every shove comes from the side and lasts a tenth of a second; I swept the force from 100 N to 400 N, and at each force shoved it once from each side at 8 different moments of its stride: -| Force | Pushed away: falls out of 8 | Pushed towards you | +| Force | Pushed left: falls out of 8 | Pushed right | | --- | --- | --- | | 175 N and below | 0 | 0 | | 200 N | 1 | 0 | @@ -144,7 +144,7 @@ I swept the force from 100 N to 400 N, and at each force shoved it once from eac | 250 N | 7 | 4 | | 275 N and above | 8 | 8 | -There is no clean threshold in the middle. At the same 225 N in the same direction, moving the shove by 60 ms turns a recovery into a fall. The two sides are not symmetric either. Try one force several times in the instrument and you will get different outcomes. In the two 175 N runs (one from each side) the torso tipped by at most 5.5 and 5.8 degrees before coming back. +There is no clean threshold in the middle. At the same 225 N in the same direction, moving the shove by 60 ms turns a recovery into a fall. The two sides are not symmetric either. Try one force several times in the figure and you will get different outcomes. In the two 175 N runs (one from each side) the torso tipped by at most 5.5 and 5.8 degrees before coming back. ## The real world is not this clean @@ -156,18 +156,26 @@ In the simulator the sensors have no noise, signals have no delay, and the floor **Latency.** The network always sees the world as it was a few beats ago. Every extra 24 ms slows it a little: asked for 0.5, its actual speed falls from 0.47 all the way to 0.26 at 72 ms. Up to 72 ms it never fell; it only walks slower and slower. -**Noise.** Random error added to the readings the network sees (to the scaled numbers, so it has no physical unit): at ±0.2 the speed drops from 0.47 to 0.42, which is hard to see by eye; at ±0.3 it stays up but slows to between 0.17 and 0.36 m/s; ±0.4 is the edge, where over 10 random seeds of 10 seconds each it fell 4 times and crawled at 0.1 to 0.25 m/s the other 6; at ±0.5 it fell all 10 times; at ±0.8 it is down within 0.6 seconds. +**Noise.** Random error added to the readings the network sees (to the scaled numbers, so it has no physical unit). ±0.4 is the edge: -**Friction.** This one surprised me the most. Change the floor from rubber (μ = 1) to something slipperier than ice (μ = 0.05) and the speed only drops from 0.47 to 0.44. The feet really do slip: the average sliding speed of the feet on the ground rises from 0.12 to 0.26 m/s. It walks anyway. Only at μ = 0.01 does it visibly struggle. +| Noise | Result (asked for 0.5) | +| --- | --- | +| ±0.2 | 0.42, hard to see by eye | +| ±0.3 | stays up, but only 0.17 to 0.36 | +| ±0.4 | out of 10 random seeds, 4 falls; the other 6 crawl at 0.10 to 0.25 | +| ±0.5 | falls all 10 times | +| ±0.8 | down within 0.6 seconds | + +**Friction.** This one surprised me the most. Change the floor from rubber (μ = 1) to something slipperier than ice (μ = 0.05) and the speed only drops from 0.47 to 0.44. The feet really do slip: the average sliding speed of the feet on the ground rises from 0.12 to 0.27 m/s. It walks anyway. Only at μ = 0.01 does it visibly struggle. Taken together, these three results say the same thing. When a policy like this is trained, the simulator deliberately scrambles the friction, shoves the robot and adds noise to the sensors. It is called domain randomization, and the point is to stop the policy from trusting the simulator too much.I have not looked up the training configuration of this particular policy, so "friction was randomized during training" is a guess inferred from its behaviour, not a fact. The ways you fail to knock it over here are mostly ways somebody already tried on your behalf during training; the ways you succeed (blindfolding the joint angles, softening the spring) are ones nobody thought to defend against. ## What is not here -This article only runs the policy; it never learns anything. Where that 758 KB file came from has not been touched at all. The first two points below are how policies of this kind are generally trained; I have not checked this policy's actual training setup: +This article only runs the policy; it never learns anything. Where those 757 KB of weights came from has not been touched at all. The first two points below are how policies of this kind are generally trained: - **The reward function.** Training usually has only a score: points for matching the commanded speed, penalties for falling, shaking, wasting energy and dragging feet. The gait is what that score squeezed out. -- **Thousands of dogs.** The policy file sits under `policy/ppo/` in the official repo, so the algorithm is PPO. Training of this kind usually simulates thousands of dogs at once on a GPU, each one falling over on different terrain and different friction. +- **Thousands of dogs.** The policy file sits under `policy/ppo/` in the official repo, so the algorithm is most likely PPO. Training of this kind usually simulates thousands of dogs at once on a GPU, each one falling over on different terrain and different friction. - **Flat ground only.** The official deployment code includes stair terrain. It is not included here. Reproducing the training in a browser would take several orders of magnitude more compute. This page cannot do it. diff --git a/content/posts/lite3-walking/zh.mdx b/content/posts/lite3-walking/zh.mdx index 77f35e6..8380dd9 100644 --- a/content/posts/lite3-walking/zh.mdx +++ b/content/posts/lite3-walking/zh.mdx @@ -28,7 +28,7 @@ import { GainsLab, PushLab, RemoteLab, SensesLab, WorldLab } from "./components" ## 大腦只有四次矩陣乘法 -公開的策略檔是一個 758 KB 的 ONNX 檔。打開來看,裡面沒有任何特別的東西: +公開的策略檔是一個 ONNX 檔,權重一共 757 KB(189,324 個 32 位元浮點數)。打開來看,裡面沒有任何特別的東西: | | 這隻狗的大腦 | | --- | --- | @@ -55,7 +55,7 @@ ELU 和 ReLU 只差在負的那一半:ReLU 直接歸零,ELU 是一條平滑 我拿 ONNX Runtime 對原本的 ONNX 檔算出來的動作當標準答案,和這段程式的輸出比對,差距在小數第六位。物理那一邊也對過:同一個起始姿勢、同一個指令,WebAssembly 版 MuJoCo(在 Node 裡跑)和 Python 版 MuJoCo 走十秒,終點都是 4.69 公尺。碰撞網格換成凸包之後,這一頁的版本是 4.70。 -這個網路每秒被呼叫 83 次,每次不到 0.2 毫秒(在 Node 裡量的)。真正花時間的是物理。 +這個網路每一次呼叫只要 0.15 毫秒左右(在 Node 裡量的),真正花時間的是物理。M4 Pro、Node 22,連續呼叫 5,000 次取平均,三次量到 151.8 到 154.6 微秒;腳本和輸出在 `docs/research/lite3-spike/`。 ## 它感覺得到什麼 @@ -112,7 +112,7 @@ $K_p$ 是彈簧有多硬,$K_d$ 是阻尼有多黏。力矩最多 30 牛頓公 -把 $K_p$ 拉到 10,實線就追不上虛線了,它會慢慢趴下去。拉到 100,實線幾乎貼著虛線,它也走得比你要求的快。我量到的前進速度(10 秒的平均,`docs/research/lite3-spike/shipped-build-output.txt`;儀器上的讀數把橫向的漂移也算進去,而且不含起步,所以會比這張表高): +把 $K_p$ 拉到 10,實線就追不上虛線了,它會慢慢趴下去。拉到 100,實線幾乎貼著虛線,它也走得比你要求的快。我量到的前進速度是這樣:10 秒的平均,`docs/research/lite3-spike/shipped-build-output.txt`。圖上的讀數把橫向的漂移也算進去,而且不含起步,所以會比這張表高。 | $K_p$ | $K_d$ | 指令 0.5 時的實際速度 | | --- | --- | --- | @@ -136,7 +136,7 @@ $K_p$ 是彈簧有多硬,$K_d$ 是阻尼有多黏。力矩最多 30 牛頓公 我把力道從 100 牛頓掃到 400,每個力道在步伐的 8 個不同時間點、從兩邊各推一次: -| 力道 | 往裡推,8 次裡倒幾次 | 往外推 | +| 力道 | 往左推,8 次裡倒幾次 | 往右推 | | --- | --- | --- | | 175 N 以下 | 0 | 0 | | 200 N | 1 | 0 | @@ -144,7 +144,7 @@ $K_p$ 是彈簧有多硬,$K_d$ 是阻尼有多黏。力矩最多 30 牛頓公 | 250 N | 7 | 4 | | 275 N 以上 | 8 | 8 | -中間那一段沒有一個乾淨的門檻。同樣 225 牛頓、同一個方向,我只是把推的時間點挪 60 毫秒,結果就從撐住變成倒地。兩邊也不對稱。在儀器裡用同樣的力道多推幾次,你會推出不一樣的結果。175 牛頓的兩次(左右各一),機身最多只歪了 5.5 和 5.8 度就回正了。 +中間那一段沒有一個乾淨的門檻。同樣 225 牛頓、同一個方向,我只是把推的時間點挪 60 毫秒,結果就從撐住變成倒地。兩邊也不對稱。在圖裡用同樣的力道多推幾次,你會推出不一樣的結果。175 牛頓的兩次(左右各一),機身最多只歪了 5.5 和 5.8 度就回正了。 ## 真實世界沒有這麼乾淨 @@ -156,18 +156,26 @@ $K_p$ 是彈簧有多硬,$K_d$ 是阻尼有多黏。力矩最多 30 牛頓公 **延遲。** 讓網路看到的永遠是幾拍之前的世界。每多 24 毫秒,它就慢一點:指令 0.5 時的實際速度從 0.47 一路掉到 72 毫秒時的 0.26。到 72 毫秒都沒有倒,只是越走越慢。 -**雜訊。** 在網路看到的讀數上加隨機誤差(加在縮放過的數字上,所以沒有物理單位):±0.2 時速度從 0.47 掉到 0.42,肉眼看不太出來;±0.3 還站得住,但速度掉到 0.17 到 0.36;±0.4 是邊界,我用 10 組不同的亂數各走 10 秒,4 次倒了,剩下 6 次只剩每秒 0.1 到 0.25 公尺;±0.5 十次全倒;±0.8 不到 0.6 秒就倒。 +**雜訊。** 在網路看到的讀數上加隨機誤差(加在縮放過的數字上,所以沒有物理單位)。±0.4 是邊界: -**摩擦。** 這是我最意外的一個。把地板從橡膠(μ = 1)換成比冰還滑(μ = 0.05),速度只從 0.47 掉到 0.44。腳確實在滑,著地的腳平均滑動速度從每秒 0.12 公尺升到 0.26,但它照走。要到 μ = 0.01 才明顯走不動。 +| 雜訊 | 結果(指令 0.5) | +| --- | --- | +| ±0.2 | 0.42,肉眼看不太出來 | +| ±0.3 | 站得住,但只剩 0.17 到 0.36 | +| ±0.4 | 10 組亂數裡 4 次倒,另外 6 次只剩 0.10 到 0.25 | +| ±0.5 | 10 次全倒 | +| ±0.8 | 不到 0.6 秒就倒 | + +**摩擦。** 這是我最意外的一個。把地板從橡膠(μ = 1)換成比冰還滑(μ = 0.05),速度只從 0.47 掉到 0.44。腳確實在滑,著地的腳平均滑動速度從每秒 0.12 公尺升到 0.27,但它照走。要到 μ = 0.01 才明顯走不動。 這三個結果放在一起,其實在說同一件事。這種策略訓練的時候,模擬器會故意亂調摩擦、亂推機器人、在感測器上加雜訊,這叫 domain randomization,目的就是讓它不要太相信模擬器。我沒有查這個策略實際的訓練設定,所以「它訓練時隨機化過摩擦」是從行為反推的猜測,不是事實。你在這裡弄不倒它的那些方式,多半是有人已經在訓練時替你試過了;你弄得倒它的方式(蒙住關節角度、把彈簧換軟),則是訓練時沒人想過要防的。 ## 這裡沒有的東西 -這篇文章只有「執行」,沒有「學習」。那個 758 KB 的檔案是怎麼來的,這裡一個字都還沒說。下面前兩點是這類策略一般的訓練方式;這個策略實際的訓練設定我沒有查過: +這篇文章只有「執行」,沒有「學習」。那個 757 KB 的權重是怎麼來的,這裡一個字都還沒說。下面前兩點是這類策略一般的訓練方式: - **獎勵函數。** 訓練時通常只有一個分數:速度跟上指令加分,摔倒、抖動、耗電、腳拖地扣分。走路的樣子是這個分數逼出來的。 -- **幾千隻狗。** 策略檔放在官方 repo 的 `policy/ppo/` 底下,所以演算法是 PPO。這類訓練通常在 GPU 上同時模擬幾千隻狗,每一隻都在不同的地形和不同的摩擦係數上摔。 +- **幾千隻狗。** 策略檔放在官方 repo 的 `policy/ppo/` 底下,所以演算法多半是 PPO。這類訓練通常在 GPU 上同時模擬幾千隻狗,每一隻都在不同的地形和不同的摩擦係數上摔。 - **只有平地。** 官方的部署程式裡有樓梯地形,這裡沒有放。 要在瀏覽器裡重現訓練,需要的算力差了好幾個數量級,這一頁做不到。 diff --git a/docs/research/lite3-spike/policy-timing-output.txt b/docs/research/lite3-spike/policy-timing-output.txt new file mode 100644 index 0000000..0d4488d --- /dev/null +++ b/docs/research/lite3-spike/policy-timing-output.txt @@ -0,0 +1,5 @@ +run 1: 151.9 µs per call (mean of 5000) +run 2: 151.8 µs per call (mean of 5000) +run 3: 154.6 µs per call (mean of 5000) + +Measured 2026-09-22 on an Apple M4 Pro, Node v22.19.0 via vitest; the script is policy-timing.test.ts.txt beside this file. diff --git a/docs/research/lite3-spike/policy-timing.test.ts.txt b/docs/research/lite3-spike/policy-timing.test.ts.txt new file mode 100644 index 0000000..04a644f --- /dev/null +++ b/docs/research/lite3-spike/policy-timing.test.ts.txt @@ -0,0 +1,18 @@ +import { readFileSync, writeFileSync } from "node:fs"; +import { it } from "vitest"; +import { Policy } from "@/content/posts/lite3-walking/components/policy"; +import golden from "./fixtures/lite3-policy.json"; + +it("times one policy call", () => { + const file = readFileSync("public/lite3/policy.f32"); + const policy = new Policy(file.buffer.slice(file.byteOffset, file.byteOffset + file.byteLength)); + const obs = golden.pairs.map((p: { obs: number[] }) => Float32Array.from(p.obs)); + for (let i = 0; i < 2000; i++) policy.act(obs[i % obs.length]); // warm up + const lines: string[] = []; + for (let run = 1; run <= 3; run++) { + const n = 5000, t0 = performance.now(); + for (let i = 0; i < n; i++) policy.act(obs[i % obs.length]); + lines.push(`run ${run}: ${(((performance.now() - t0) / n) * 1000).toFixed(1)} µs per call (mean of ${n})`); + } + writeFileSync("docs/research/lite3-spike/policy-timing-output.txt", lines.join("\n") + `\n\nMeasured 2026-09-22 on an Apple M4 Pro, Node ${process.version} via vitest; the script is policy-timing.test.ts.txt beside this file.\n`); +}); From 708ac2634a66ff74af2fd812fdcb7d62d81d2c73 Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 16:56:34 +0800 Subject: [PATCH 10/20] =?UTF-8?q?docs(diffusion):=20polish=20=E2=84=96=200?= =?UTF-8?q?07=20=E2=80=94=20the=20six=20edits=20Paul=20chose?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - ᾱ is nearly zero in the first steps, not zero: the formula divides by a number close to zero; that aside becomes a sidenote, so "nothing is broken" follows the blurry balls directly. - "Recitation, not invention" is said once; "in my run" once in the guidance paragraph; the last difference is two bullets (network size, and working in a latent space instead of on pixels); 儀器 / instrument becomes 圖 / figure. Every number was checked against the code and docs/research/2026-09-19-diffusion-spike.md; none changed. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/diffusion-points/en.mdx | 17 +++++++++-------- content/posts/diffusion-points/zh.mdx | 17 +++++++++-------- 2 files changed, 18 insertions(+), 16 deletions(-) diff --git a/content/posts/diffusion-points/en.mdx b/content/posts/diffusion-points/en.mdx index ebd480f..d75ee27 100644 --- a/content/posts/diffusion-points/en.mdx +++ b/content/posts/diffusion-points/en.mdx @@ -22,7 +22,7 @@ Choose your fruit and press Train. -At first nothing happens: an untrained model has no idea what fruit is, so after 40 steps the noise is still noise. Once training runs, colour comes first: after about five seconds the left cloud leans towards one colour and the right cloud towards another. At around ten seconds you can make out shapes, and after half a minute the apple has its leaf.That is the pace on my machine (M-series Mac, Chrome), about 90 training steps per second. Yours will differ. Training and sampling both run in a Web Worker, so scrolling stays smooth; the price is that the two take turns on one core, and the steps/s under the instrument counts only training's share of the time, so it reads higher. +At first nothing happens: an untrained model has no idea what fruit is, so after 40 steps the noise is still noise. Once training runs, colour comes first: after about five seconds the left cloud leans towards one colour and the right cloud towards another. At around ten seconds you can make out shapes, and after half a minute the apple has its leaf.That is the pace on my machine (M-series Mac, Chrome), about 90 training steps per second. Yours will differ. Training and sampling both run in a Web Worker, so scrolling stays smooth; the price is that the two take turns on one core, and the steps/s under the figure counts only training's share of the time, so it reads higher. Let me be clear about what this is and is not. **This model cannot "draw fruit". It has memorised the two fruit you picked.** Those are the only two things it has ever seen, so they are all it can produce, along with things in between. A real image model has seen billions of pictures, and that is what lets it draw combinations it never saw. @@ -94,11 +94,11 @@ And ask again from the new position.This is the DDIM way of stepping: Here is one complete run of the model you trained, which you can scrub back and forth. Do try "the model's guess of the result" on the right: - + -Pull the slider to about step 4 or 5 and switch to the model's guess. You will see two fuzzy balls, roughly the right colours and with no shape at all. (In the first two or three steps you see not balls but points stuck to the sides of a box: $\bar\alpha_t$ is almost 0 there, so the formula above amounts to dividing by zero, and the smallest error in the model's answer is blown up to the ±1.5 limit. That is a numerical problem, not the model's answer.) **Nothing is broken; this is its best answer at that moment**, the average from the previous section: while the noise is large, the model's guess at the result is the average of the whole fruit, a blob of average colour sitting in the middle. +Pull the slider to about step 4 or 5 and switch to the model's guess. You will see two fuzzy balls, roughly the right colours and with no shape at all.In the first two or three steps you see not balls but points stuck to the sides of a box: $\bar\alpha_t$ is almost 0 there, so the formula above divides by a number very close to zero, and the smallest error in the model's answer is blown up to the ±1.5 limit. That is a numerical problem, not the model's answer. **Nothing is broken; this is its best answer at that moment**, the average from the previous section: while the noise is large, the model's guess at the result is the average of the whole fruit, a blob of average colour sitting in the middle. This is also why it cannot be done in one step. Trust the model's guess completely on the first step and that average blob is what you get. Taking many small steps means trusting it only a little each time and moving to where the noise is smaller, where the range of possible answers is narrower, their average less blurred, and so the next guess more specific. That is how the shape goes from fuzzy to sharp, round after round. @@ -108,7 +108,7 @@ Set the total number of steps to 3 or 5 to see what too few does: every step is Each point is not three numbers but six: **x, y, z, red, green, blue**. To the model, colour and position are the same kind of thing, numbers to be recovered from noise, and every $x$ in the formulas above means all six. -So what it learns is not a shape plus a paint job but one object in six dimensions: "a point at this position should have this colour". The points at the top of the apple are green because, in those six dimensions, that position combined with green is where the fruit is. It is also why the colours scramble along with the positions when you raise the noise in instrument 02. +So what it learns is not a shape plus a paint job but one object in six dimensions: "a point at this position should have this colour". The points at the top of the apple are green because, in those six dimensions, that position combined with green is where the fruit is. It is also why the colours scramble along with the positions when you raise the noise in figure 02. ## How it knows which one to grow @@ -116,7 +116,7 @@ One model sculpts two fruit because it is shown two more numbers, $c$: `[1, 0]` Text-to-image models do the same thing, except that $c$ is a long list of numbers computed from a piece of text. -And the third cloud in instrument 01? It is given `[0.5, 0.5]`, **numbers the model never saw in training**. Nobody taught it what "half apple, half banana" looks like, so it has to find a compromise between its two answers by itself. The time I trained apple and banana, what grew was something apple-shaped, banana-coloured, with the leaf still on top. A different pair compromises differently. +And the third cloud in figure 01? It is given `[0.5, 0.5]`, **numbers the model never saw in training**. Nobody taught it what "half apple, half banana" looks like, so it has to find a compromise between its two answers by itself. The time I trained apple and banana, what grew was something apple-shaped, banana-coloured, with the leaf still on top. A different pair compromises differently. ### Telling it to listen harder @@ -137,12 +137,13 @@ The difference between knowing and not knowing is the part of the direction that Pull $w$ down to 0 and the two clouds grow into the same thing, an average fruit somewhere between the two, because the model is not listening to which one you want at all. $w = 1$ is the normal look. Going up, the fruit first becomes more typical and the points tighter, and then it starts to deform: in my run, at $w = 4$ the apple was down to a red rim and the banana had shrunk to a sliver. -Real models face the same trade-off: stronger guidance gives results closer to the textbook answer with less variety, and too much breaks them. In that training run of mine it looked best between 1 and 2; image models commonly default to around 7, because their condition (a piece of text) is far vaguer than "one of two fruit" and has to be listened to much harder. +Real models face the same trade-off: stronger guidance gives results closer to the textbook answer with less variety, and too much breaks them. Mine looked best between 1 and 2; image models commonly default to around 7, because their condition (a piece of text) is far vaguer than "one of two fruit" and has to be listened to much harder. ## How this differs from a real image model -- **It works on points, not pictures.** One piece of data here is a point, six numbers, so a fully connected network of some twenty thousand parameters is enough. For an image model one piece of data is a whole picture, hundreds of thousands of numbers whose relationships matter, so it takes a far larger convolutional or Transformer network, usually not on the pixels themselves but on a compressed latent version of the picture. -- **It has seen two things.** As said at the start: this is recitation, not invention. The difference comes from the amount of data, not from the method. +- **It works on points, not pictures.** One piece of data here is a point, six numbers, so a fully connected network of some twenty thousand parameters is enough. For an image model one piece of data is a whole picture, hundreds of thousands of numbers whose relationships matter, so it takes a far larger convolutional or Transformer network. +- **It doesn't work on pixels.** Image models usually compress the picture into a small latent space first and run the diffusion there. +- **It has seen two things.** The difference comes from the amount of data, not from the method. - **Everything else is the same.** The noising formula, the guess-the-noise loss, stepping back a little at a time, steering with a condition, strengthening it with guidance: those five things are the ones inside the image models you use. ## Sources diff --git a/content/posts/diffusion-points/zh.mdx b/content/posts/diffusion-points/zh.mdx index f270392..21fade3 100644 --- a/content/posts/diffusion-points/zh.mdx +++ b/content/posts/diffusion-points/zh.mdx @@ -24,7 +24,7 @@ import { FieldLab, GuidanceLab, NoiseLab, StepsLab, TrainLab } from "./component -一開始什麼都不會發生:沒訓練過的模型不知道水果是什麼,雜訊走完 40 步還是雜訊。按下訓練之後,先出現的是顏色:大約五秒,左邊那團就偏向一種顏色、右邊偏向另一種。十秒左右看得出形狀,半分鐘之後連蘋果的葉子都有了。這是我的電腦(M 系列 Mac、Chrome)上的速度,大約每秒訓練 90 步。你的會不一樣。訓練和取樣都在一個 Web Worker 裡跑,所以頁面捲動不會卡;代價是兩者要輪流用同一顆核心,儀器下面的「步/秒」只算輪到訓練的那部分時間,所以數字會比較高。 +一開始什麼都不會發生:沒訓練過的模型不知道水果是什麼,雜訊走完 40 步還是雜訊。按下訓練之後,先出現的是顏色:大約五秒,左邊那團就偏向一種顏色、右邊偏向另一種。十秒左右看得出形狀,半分鐘之後連蘋果的葉子都有了。這是我的電腦(M 系列 Mac、Chrome)上的速度,大約每秒訓練 90 步。你的會不一樣。訓練和取樣都在一個 Web Worker 裡跑,所以頁面捲動不會卡;代價是兩者要輪流用同一顆核心,圖下面的「步/秒」只算輪到訓練的那部分時間,所以數字會比較高。 先講清楚這是什麼、不是什麼。**這個模型不是「會畫水果」,它是把你選的這兩顆水果背起來了。** 它這輩子只看過這兩樣東西,所以它生得出來的也只有這兩樣,以及介於兩者之間的東西。真的畫圖模型看過幾十億張圖,才有辦法畫出沒看過的組合。 @@ -96,11 +96,11 @@ $$ 下面是你自己訓練的那個模型走的一次完整過程,可以前後拖。請特別試試右邊的「模型此刻猜的成品」: - + -把滑桿拉到第 4、5 步左右,切到「模型此刻猜的成品」。你會看到兩團糊糊的球,顏色大致對,形狀完全沒有。(最前面兩三步看到的不是球,而是一堆貼在方盒子邊上的點:那時 $\bar\alpha_t$ 幾乎是 0,上面的式子等於除以零,模型只要猜錯一點點就被放大到 ±1.5 的上限。那是數值問題,不是模型的答案。)**這不是壞掉,這是它此刻最好的答案**,也就是上一節說的「平均」:雜訊還很大的時候,模型猜的成品就是整顆水果的平均,一團在正中央的、平均顏色的東西。 +把滑桿拉到第 4、5 步左右,切到「模型此刻猜的成品」。你會看到兩團糊糊的球,顏色大致對,形狀完全沒有。最前面兩三步看到的不是球,而是一堆貼在方盒子邊上的點:那時 $\bar\alpha_t$ 幾乎是 0,上面的式子是除以一個很接近零的數,模型只要猜錯一點點就被放大到 ±1.5 的上限。那是數值問題,不是模型的答案。 **這不是壞掉,這是它此刻最好的答案**,也就是上一節說的「平均」:雜訊還很大的時候,模型猜的成品就是整顆水果的平均,一團在正中央的、平均顏色的東西。 這也是為什麼不能一步到位。如果第一步就完全相信模型的猜測,得到的就是那團平均。走很多小步的意思是:每次只相信一點點,走到雜訊小一些的地方,那裡的「所有可能答案」範圍變窄了,平均起來就不那麼糊,於是下一次的猜測會更具體。形狀是這樣一輪一輪從糊變清楚的。 @@ -110,7 +110,7 @@ $$ 每個點其實不是三個數字,是六個:**x、y、z、紅、綠、藍**。對模型來說,顏色和位置沒有差別,都是要從雜訊裡還原的數字,上面所有式子裡的 $x$ 都是這六個數字。 -所以它學到的不是「形狀」再加上「上色」,而是一個六維空間裡的東西:「在這個位置的點,應該是這個顏色」。蘋果頂端的點是綠色的,是因為六維空間裡,那個位置配上綠色的組合才在水果上。儀器 02 把雜訊拉高的時候顏色也會跟著亂掉,就是這個原因。 +所以它學到的不是「形狀」再加上「上色」,而是一個六維空間裡的東西:「在這個位置的點,應該是這個顏色」。蘋果頂端的點是綠色的,是因為六維空間裡,那個位置配上綠色的組合才在水果上。圖 02 把雜訊拉高的時候顏色也會跟著亂掉,就是這個原因。 ## 它怎麼知道要長哪一種 @@ -118,7 +118,7 @@ $$ 文字生成圖片的模型做的是同一件事,只是 $c$ 換成了一段文字算出來的一長串數字。 -那儀器 01 的第三團呢?它拿到的是 `[0.5, 0.5]`,**一組模型在訓練時從來沒看過的數字**。沒有人教過它「一半蘋果一半香蕉」長什麼樣子,它只能自己在兩個答案之間找一個折衷。我選蘋果和香蕉訓練的那一次,它長出來的是一顆蘋果形狀、香蕉顏色、頭上還留著葉子的東西。換一對水果,折衷的方式也會不一樣。 +那圖 01 的第三團呢?它拿到的是 `[0.5, 0.5]`,**一組模型在訓練時從來沒看過的數字**。沒有人教過它「一半蘋果一半香蕉」長什麼樣子,它只能自己在兩個答案之間找一個折衷。我選蘋果和香蕉訓練的那一次,它長出來的是一顆蘋果形狀、香蕉顏色、頭上還留著葉子的東西。換一對水果,折衷的方式也會不一樣。 ### 叫它「更聽話一點」 @@ -139,12 +139,13 @@ $$ 把 $w$ 拉到 0,兩團會長成同一個東西:一顆介於兩種水果之間的「平均水果」,因為模型完全沒聽你要哪一種。$w = 1$ 是正常的樣子。繼續往上拉,水果會先變得更「典型」、點更集中,再往上就開始變形:在我的那次訓練裡,$w = 4$ 的蘋果只剩一圈紅色的邊,香蕉縮成一小條。 -這是真的模型也有的取捨:引導越強,結果越像「標準答案」,多樣性越低,太強就壞掉。在我的那次訓練裡,1 到 2 之間最好看;畫圖模型常見的預設值在 7 左右,因為它們的條件(一段文字)比「兩種水果選一種」模糊得多,需要更用力地聽。 +這是真的模型也有的取捨:引導越強,結果越像「標準答案」,多樣性越低,太強就壞掉。我那次是 1 到 2 之間最好看;畫圖模型常見的預設值在 7 左右,因為它們的條件(一段文字)比「兩種水果選一種」模糊得多,需要更用力地聽。 ## 這和真的畫圖模型差在哪 -- **它處理的是點,不是圖。** 這裡的一筆資料是一個點、六個數字,所以一個兩萬多個參數的全連接網路就夠了。圖片模型的一筆資料是一整張圖、幾十萬個數字,而且像素之間的關係很重要,所以要用大得多的卷積或 Transformer 網路,而且通常不直接在像素上做,而是先把圖壓縮到一個小的 latent 空間裡再做擴散。 -- **它只看過兩樣東西。** 開頭說過了:這是背誦,不是創作。差別來自資料量,不是方法。 +- **它處理的是點,不是圖。** 這裡的一筆資料是一個點、六個數字,所以一個兩萬多個參數的全連接網路就夠了。圖片模型的一筆資料是一整張圖、幾十萬個數字,像素之間的關係很重要,要用大得多的卷積或 Transformer 網路。 +- **它不在像素上做。** 圖片模型通常先把圖壓縮到一個小的 latent 空間,再在那裡做擴散。 +- **它只看過兩樣東西。** 差別來自資料量,不是方法。 - **其他全部一樣。** 加雜訊的式子、「猜雜訊」的損失函數、一步一步往回走、用條件控制、用 guidance 加強,這五件事和你在用的畫圖模型裡的是同一套。 ## 來源 From e3079574649135dc73e195bc47c32bae25c99e83 Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 17:06:17 +0800 Subject: [PATCH 11/20] =?UTF-8?q?docs(slam):=20polish=20=E2=84=96=20008=20?= =?UTF-8?q?=E2=80=94=20seven=20edits,=20the=20ICP=20figure's=20numbers=20m?= =?UTF-8?q?easured?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - The align figure's default runs, reproduced by a script with its own inputs (docs/research/slam-2d-icp/): the corner start is 2.02 m and 14.3° off, not "nearly thirty degrees", and ICP ends 1 mm away after 10 steps; the corridor start is 2.40 m off ("a little over two metres"), and it stops 0.92 m away with 75% matched at 3.5 cm, as the text said. - A 3-D sweep takes thirty to forty milliseconds (28 ms per match at the page's sampling, 39 ms a sweep in the lidar-odometry run), not ten to thirty. - The robot dog is 它 as in № 006, and "our robot dog" is "a real robot dog (the real Lite3, for one)", Paul's words. - Loop closure is named where its button first appears; "nominates" becomes "finds the likely"; the card's button note stands on its own line. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/slam-2d/en.mdx | 16 +++++++------ content/posts/slam-2d/zh.mdx | 18 ++++++++------- docs/research/slam-2d-icp/align.test.ts.txt | 25 +++++++++++++++++++++ docs/research/slam-2d-icp/output.txt | 4 ++++ 4 files changed, 48 insertions(+), 15 deletions(-) create mode 100644 docs/research/slam-2d-icp/align.test.ts.txt create mode 100644 docs/research/slam-2d-icp/output.txt diff --git a/content/posts/slam-2d/en.mdx b/content/posts/slam-2d/en.mdx index 986b886..f6416cd 100644 --- a/content/posts/slam-2d/en.mdx +++ b/content/posts/slam-2d/en.mdx @@ -50,9 +50,9 @@ Try dragging the cyan scan onto the white one by hand, then press "Let ICP do it -At **a corner**, starting two metres and nearly thirty degrees off, it is back within ten steps, to better than a centimetre. Press Scramble a few times: most starts are recovered, and when one is not, ICP knows it, because only one or two points in ten found a partner. +At **a corner**, starting two metres and some fifteen degrees off, it is back in ten steps, to within a millimetre.The figure's default starts, rerun by a script: at the corner the start is 2.02 m and 14.3° off and ICP stops at step 10; in the corridor the start is 2.40 m off and it stops 0.92 m away, with 75% of points matched, 3.5 cm from the wall. Script and output in `docs/research/slam-2d-icp/`. Press Scramble a few times: most starts are recovered, and when one is not, ICP knows it, because only one or two points in ten found a partner. -Now switch to **mid-corridor** and press again. This time the cyan scan has only been slid two metres back along the corridor, yet ICP stops almost a metre from the right answer, looking quite sure of itself: three quarters of the points matched, 3.5 cm from the wall on average. Two parallel walls tell you how far you are from them and nothing about how far along them you are; slide a metre forward and the scan looks almost the same. +Now switch to **mid-corridor** and press again. This time the cyan scan has only been slid a little over two metres back along the corridor, yet ICP stops almost a metre from the right answer, looking quite sure of itself: three quarters of the points matched, 3.5 cm from the wall on average. Two parallel walls tell you how far you are from them and nothing about how far along them you are; slide a metre forward and the scan looks almost the same. The good news is that this kind of doubt can be computed, so the car can simply refuse such a match.The small equation ICP solves already carries "which directions am I sure about": it is the pink ellipse in the figure, a small circle at the corner and stretched along the corridor in the corridor. Any match whose ellipse is too flat is thrown away. @@ -67,7 +67,7 @@ The car now holds two kinds of information: Draw every position the car remembered as a dot, and every "I believe" as a **spring** between two dots. The spring's natural length is what the sentence says, and the surer the sentence, the stiffer the spring. This picture is called a pose graph. -With only the wheel springs the chain is slack; any shape will do, so it lies bent the way the wheels said. Press "Close the loop" and one short, stiff spring appears, pulling the two ends of the chain together: +With only the wheel springs the chain is slack; any shape will do, so it lies bent the way the wheels said. Press "Close the loop" (recognising an old place is called closing the loop, or loop closure) and one short, stiff spring appears, pulling the two ends of the chain together: @@ -89,13 +89,15 @@ The same route and the same car, twice round under four different settings. This **Very wrong wheels.** It does look for places it has been, and never finds one. It looks through old records near *where it believes it is*, and that belief has drifted more than ten metres away. As far as it knows, it never came back. -**The same wheels, plus a camera.** Every wall is painted with its own stripes, and the car gets a panoramic camera one pixel tall. Now "I have seen this view before" nominates old places, and ICP confirms them. In two laps it recognises a place twenty-one times and is never wrong. Search by position, and once you have drifted far enough you are lost for good; search by appearance, and it does not matter how far you drifted. +**The same wheels, plus a camera.** Every wall is painted with its own stripes, and the car gets a panoramic camera one pixel tall. Now "I have seen this view before" finds the likely old places, and ICP confirms them. In two laps it recognises a place twenty-one times and is never wrong. Search by position, and once you have drifted far enough you are lost for good; search by appearance, and it does not matter how far you drifted. -**One wrong recognition.** The map was fine; one wrong spring was forced in, claiming that two places far apart are the same spot. I measured this separately: a hundred and four correct springs cannot save the map from one wrong one. The mean error goes from 0.10 m to 3.9 m, because least squares has no resistance to an outlier. That is why a real system would rather miss ten places than misrecognise one. Press the button on this card and the map that gets wrecked is the one you drew yourself at the top. +**One wrong recognition.** The map was fine; one wrong spring was forced in, claiming that two places far apart are the same spot. I measured this separately: a hundred and four correct springs cannot save the map from one wrong one. The mean error goes from 0.10 m to 3.9 m, because least squares has no resistance to an outlier. That is why a real system would rather miss ten places than misrecognise one. + +(Press the button on this card and the map that gets wrecked is the one you drew yourself at the top.) ## What is different on the robot dog -Our robot dog uses **a 3-D lidar and cameras**. Compared with this little car, the first half is very different and the second half is almost the same. +A real robot dog (the real Lite3 from the previous article, for one) uses **a 3-D lidar and cameras**. Compared with this little car, the first half is very different and the second half is almost the same. **The lidar's ring becomes a cloud.** Sixteen lasers stacked vertically, more than five thousand points per turn. And a position is no longer three numbers in a plane (x, y, heading) but six, adding height, pitch and roll, because a walking dog sways. Lining two sweeps up works exactly as before; the small equation just has six unknowns instead of three. @@ -103,7 +105,7 @@ Our robot dog uses **a 3-D lidar and cameras**. Compared with this little car, t -A single match is good to better than a centimetre and a quarter of a degree, in ten to thirty milliseconds a sweep. After thirty metres round the room it has still drifted more than sixty centimetres and six degrees. **Small errors are not errors that do not add up.** The cure is [the chain from before](#every-i-believe-in-one-picture), with six numbers per dot instead of three. +A single match is good to better than a centimetre and a quarter of a degree, in thirty to forty milliseconds a sweep. After thirty metres round the room it has still drifted more than sixty centimetres and six degrees. **Small errors are not errors that do not add up.** The cure is [the chain from before](#every-i-believe-in-one-picture), with six numbers per dot instead of three. **The cameras recognise places.** They are the real version of the camera in the previous section: features across a whole image rather than one row of pixels, for the same purpose. diff --git a/content/posts/slam-2d/zh.mdx b/content/posts/slam-2d/zh.mdx index 85fc7ec..43c859b 100644 --- a/content/posts/slam-2d/zh.mdx +++ b/content/posts/slam-2d/zh.mdx @@ -10,7 +10,7 @@ interactive: true import { AlignLab, CloudLab, DriveLab, OutcomesLab, SpringsLab, WheelsLab } from "./components"; -[上一篇](/zh/posts/lite3-walking)的機器狗會走路了,但牠不知道自己走到哪裡。牠沒有 GPS(室內收不到),也沒有人給牠地圖。 +[上一篇](/zh/posts/lite3-walking)的機器狗會走路了,但它不知道自己走到哪裡。它沒有 GPS(室內收不到),也沒有人給它地圖。 這件事麻煩在它是一個雞生蛋的問題:**要知道自己在哪,得先有地圖;要畫地圖,得先知道自己在哪。** 兩件事只能同時做。這個問題叫 SLAM,同時定位與建圖。 @@ -48,9 +48,9 @@ import { AlignLab, CloudLab, DriveLab, OutcomesLab, SpringsLab, WheelsLab } from -在**轉角**,一開始差了兩公尺、快三十度,它十步之內就對回來,最後的誤差不到一公分。按「打亂」多試幾次:大部分的起點都救得回來;救不回來的時候,它自己也知道,因為對得上的點只剩一兩成。 +在**轉角**,一開始差了兩公尺、十幾度,它十步就對回來,最後只差 1 毫米。照這張圖的預設起點用腳本重跑:轉角起點差 2.02 公尺、14.3 度,第 10 步停下;走廊起點差 2.40 公尺,停在 0.92 公尺外、75% 的點對上、離牆 3.5 公分。腳本和輸出在 `docs/research/slam-2d-icp/`。按「打亂」多試幾次:大部分的起點都救得回來;救不回來的時候,它自己也知道,因為對得上的點只剩一兩成。 -然後切到**走廊中段**再按一次。這次青色的掃描只是沿著走廊往後退了兩公尺,它卻停在離正確答案快一公尺的地方,而且看起來很有把握:七成多的點對上了,離牆平均只有 3.5 公分。兩面平行的牆只告訴你「離牆多遠」,沒有告訴你「沿著牆走了多遠」;往前滑一公尺,掃描看起來幾乎一樣。 +然後切到**走廊中段**再按一次。這次青色的掃描只是沿著走廊往後退了兩公尺多,它卻停在離正確答案快一公尺的地方,而且看起來很有把握:七成多的點對上了,離牆平均只有 3.5 公分。兩面平行的牆只告訴你「離牆多遠」,沒有告訴你「沿著牆走了多遠」;往前滑一公尺,掃描看起來幾乎一樣。 好消息是,這種不確定算得出來,所以車子可以直接不採用這種配對。ICP 解的那個小方程式本來就帶著「哪個方向我有把握」的資訊,就是圖上粉紅色的橢圓:在轉角是一個小圓,在走廊裡沿著走廊拉得很長。橢圓太扁的配對,車子一律不要。 @@ -65,7 +65,7 @@ import { AlignLab, CloudLab, DriveLab, OutcomesLab, SpringsLab, WheelsLab } from 把車子沿路記下的每個位置畫成一個點,每一句「我覺得」畫成兩點之間的一根**彈簧**,彈簧的自然長度就是那句話的內容,越有把握的話,彈簧越硬。這張圖叫位姿圖。 -只有輪子的彈簧時,整條鏈子是鬆的,怎麼擺都行,所以它就照輪子說的歪著。按下「加上閉環」,多出一根又硬又短的彈簧,把鏈子的頭和尾拉在一起: +只有輪子的彈簧時,整條鏈子是鬆的,怎麼擺都行,所以它就照輪子說的歪著。按下「加上閉環」(認出舊地方,專有名詞叫閉環,loop closure),多出一根又硬又短的彈簧,把鏈子的頭和尾拉在一起: @@ -87,13 +87,15 @@ import { AlignLab, CloudLab, DriveLab, OutcomesLab, SpringsLab, WheelsLab } from **輪子很不準。** 它有在找舊地方,但一次都找不到。它是用「自己估計的位置」去翻附近的舊記錄,而估計已經漂到十幾公尺外,它根本不覺得自己回來過。 -**同樣的輪子,加一台相機。** 每面牆漆上自己的條紋,車子多了一個只有一列像素的環景相機,改用「這個畫面我看過」來提名舊地方,再交給 ICP 確認。兩圈裡它認出舊地方二十一次,沒有一次認錯。靠位置找,漂遠了就找不回來;靠長相找,漂多遠都沒差。 +**同樣的輪子,加一台相機。** 每面牆漆上自己的條紋,車子多了一個只有一列像素的環景相機,改用「這個畫面我看過」來找出可能的舊地方,再交給 ICP 確認。兩圈裡它認出舊地方二十一次,沒有一次認錯。靠位置找,漂遠了就找不回來;靠長相找,漂多遠都沒差。 -**認錯一次。** 地圖本來是好的,只是硬加了一根錯的彈簧,宣稱兩個相距很遠的地方是同一點。我另外量過:一百零四根對的彈簧也救不回一根錯的,平均誤差從 0.10 公尺變成 3.9 公尺,因為最小平方對離群的資料沒有抵抗力。所以真的系統寧可漏掉十個舊地方,也不要認錯一個。按這張卡的按鈕,被弄壞的會是你自己在最上面繞出來的那張地圖。 +**認錯一次。** 地圖本來是好的,只是硬加了一根錯的彈簧,宣稱兩個相距很遠的地方是同一點。我另外量過:一百零四根對的彈簧也救不回一根錯的,平均誤差從 0.10 公尺變成 3.9 公尺,因為最小平方對離群的資料沒有抵抗力。所以真的系統寧可漏掉十個舊地方,也不要認錯一個。 + +(按這張卡的按鈕,被弄壞的會是你自己在最上面繞出來的那張地圖。) ## 機器狗身上的,差在哪裡 -我們的機器狗用的是 **3D 光達加相機**。和這篇的小車比,前半段差很多,後半段幾乎一樣。 +真的機器狗(例如上一篇的 Lite3 實機)用的是 **3D 光達加相機**。和這篇的小車比,前半段差很多,後半段幾乎一樣。 **光達從一圈變成一團。** 16 條雷射上下排開,轉一圈得到五千多個點的點雲。位置也不再是平面上的三個數字(x、y、朝向),而是六個:多了高度、俯仰和側傾,因為狗走路時身體會晃。但「把兩次掃描對起來」的做法完全一樣,只是小方程式從三個未知數變成六個。 @@ -101,7 +103,7 @@ import { AlignLab, CloudLab, DriveLab, OutcomesLab, SpringsLab, WheelsLab } from -單次比對的誤差不到一公分、四分之一度,每一圈只要十到三十毫秒。但繞房間三十公尺之後,它還是漂了六十幾公分、六度多。**誤差小,不等於誤差不累積。** 補救的方法就是[前面那條鏈子](#所有的我覺得放進同一張圖),只是每個點從三個數字換成六個。 +單次比對的誤差不到一公分、四分之一度,每一圈約三四十毫秒。但繞房間三十公尺之後,它還是漂了六十幾公分、六度多。**誤差小,不等於誤差不累積。** 補救的方法就是[前面那條鏈子](#所有的我覺得放進同一張圖),只是每個點從三個數字換成六個。 **相機負責認地方。** 就是上一節那台相機的真實版本:不是一列像素,而是整張影像裡的特徵點,用途相同。 diff --git a/docs/research/slam-2d-icp/align.test.ts.txt b/docs/research/slam-2d-icp/align.test.ts.txt new file mode 100644 index 0000000..78c9569 --- /dev/null +++ b/docs/research/slam-2d-icp/align.test.ts.txt @@ -0,0 +1,25 @@ +import { writeFileSync } from "node:fs"; +import { it } from "vitest"; +import { icp } from "@/content/posts/slam-2d/components/icp"; +import { compose, inverse, type Pose } from "@/content/posts/slam-2d/components/se2"; +import { RING, scan } from "@/content/posts/slam-2d/components/world"; +import { mulberry32 } from "@/lib/ml"; + +// The figure's two places and starting guesses (align-lab.tsx), run exactly as its "hand it to ICP" button does. +const PLACES = { + corner: { a: { x: 18, y: 2, theta: 0 }, b: { x: 18.3, y: 2.35, theta: 0.25 }, start: { x: 1.6, y: -1.2, theta: 0.5 } }, + corridor: { a: { x: 10.5, y: 12, theta: Math.PI }, b: { x: 10.1, y: 12.15, theta: Math.PI + 0.1 }, start: { x: -2, y: 0, theta: 0 } }, +}; +it("measures the align figure", () => { + const lines: string[] = []; + for (const [name, p] of Object.entries(PLACES)) { + const rng = mulberry32(3), target = scan(RING, p.a, rng).points, source = scan(RING, p.b, rng).points, truth: Pose = compose(inverse(p.a), p.b); + const off = (q: Pose) => ({ d: Math.hypot(q.x - truth.x, q.y - truth.y), a: Math.abs(Math.atan2(Math.sin(q.theta - truth.theta), Math.cos(q.theta - truth.theta))) * 180 / Math.PI }); + const s = off(p.start); + let k = 0, m; + do { k++; m = icp(target, source, p.start, { maxIterations: k, gate: 3 }); } while (!(m.iterations < k || k >= 30)); + const e = off(m.pose); + lines.push(`${name}: start ${s.d.toFixed(2)} m and ${s.a.toFixed(1)}° off; stopped after ${m.iterations} steps, ${e.d.toFixed(3)} m and ${e.a.toFixed(2)}° off; inliers ${(m.inliers * 100).toFixed(0)}%, rms ${(m.rms * 100).toFixed(1)} cm`); + } + writeFileSync("docs/research/slam-2d-icp/output.txt", lines.join("\n") + "\n\nMeasured 2026-09-22 with the figure's own inputs (seed 3 scans, gate 3, up to 30 steps); script: align.test.ts.txt beside this file.\n"); +}); diff --git a/docs/research/slam-2d-icp/output.txt b/docs/research/slam-2d-icp/output.txt new file mode 100644 index 0000000..26efdc2 --- /dev/null +++ b/docs/research/slam-2d-icp/output.txt @@ -0,0 +1,4 @@ +corner: start 2.02 m and 14.3° off; stopped after 10 steps, 0.001 m and 0.02° off; inliers 98%, rms 2.1 cm +corridor: start 2.40 m and 5.7° off; stopped after 18 steps, 0.918 m and 0.01° off; inliers 75%, rms 3.5 cm + +Measured 2026-09-22 with the figure's own inputs (seed 3 scans, gate 3, up to 30 steps); script: align.test.ts.txt beside this file. From 3dd29a17d4647dfb81610b4e7ce5ad381c6370ef Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 17:12:31 +0800 Subject: [PATCH 12/20] =?UTF-8?q?docs(city):=20polish=20=E2=84=96=20009=20?= =?UTF-8?q?=E2=80=94=20the=20three=20edits=20Paul=20chose?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - The busiest ten minutes are 11% (utility) against 9% (random), as docs/research/city-of-agents/RESULTS.md has them, not 12% and 10%. - Two numbers with no record go (Paul chose not to measure them): the median trip of 15 minutes, now "most trips are short", and the median replay gap of 6 units, now "a few tens of seconds of walking". - How the city grows is three bullets (roads, zones, height) with the seed-1 counts after them. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/city-of-agents/en.mdx | 14 ++++++++++---- content/posts/city-of-agents/zh.mdx | 14 ++++++++++---- 2 files changed, 20 insertions(+), 8 deletions(-) diff --git a/content/posts/city-of-agents/en.mdx b/content/posts/city-of-agents/en.mdx index 7723e0c..f80c2bf 100644 --- a/content/posts/city-of-agents/en.mdx +++ b/content/posts/city-of-agents/en.mdx @@ -30,7 +30,13 @@ First, what this is not: **there is no machine learning here.** No training, no The whole city comes from one number, the seed. The same seed always grows the same city. -Lay out 8 × 8 blocks, every fourth road a wider arterial. The east column is the riverside; every other block is zoned by its distance from the centre: commercial nearest, then restaurant streets and more offices, residential outside, a few parks at random. A building's height is a random number times how close it is to the centre. Seed 1 grows 373 buildings: 248 houses, 29 office blocks and 96 shopfronts, the tallest 80 metres. +Lay out 8 × 8 blocks, then decide, in order: + +- **Roads.** Every fourth road is a wider arterial, and the east column is the riverside. +- **Zones.** By distance from the centre: commercial nearest, then restaurant streets and more offices, residential outside, a few parks at random (two for seed 1). +- **Height.** A random number times how close the block is to the centre, so the middle is tall and the edges low. + +Seed 1 grows 373 buildings: 248 houses, 29 office blocks and 96 shopfronts, the tallest 80 metres. People walk on pavements and zebra crossings only. Each block is ringed by a pavement, joined to its neighbours at the corners, and every building has a door on it. Those points and lines are a graph of 639 nodes and 863 edges; going somewhere is a shortest path on it, found with A\* and remembered for the next person. @@ -73,7 +79,7 @@ The square makes urgent things matter out of proportion: a hunger of 0.9 is not -To begin with, A is more urgent but forty minutes away, B milder but next door, and A wins (0.573 to 0.482). Drag B's need from 0.70 to 0.77 and B overtakes: seven points of urgency are worth thirty-five minutes on foot. The median trip in this city is 15 minutes, so distance mostly decides between things about equally urgent — which restaurant, for instance. +To begin with, A is more urgent but forty minutes away, B milder but next door, and A wins (0.573 to 0.482). Drag B's need from 0.70 to 0.77 and B overtakes: seven points of urgency are worth thirty-five minutes on foot. Most trips in this city are short, so distance mostly decides between things about equally urgent — which restaurant, for instance. The formula alone is not enough. Two things go wrong: @@ -97,7 +103,7 @@ The timetable's day is cut with a knife. At eight sharp, 100% of people leave wi The random day is flat: three in the morning looks like three in the afternoon. -The one in the middle keeps hours: most people asleep at night, the cyan of work rising through the morning and ebbing through the evening, meals and company scattered across the day. Nobody drew up that rota; it is what three hundred slightly different sets of numbers work out separately. In its busiest ten minutes only 12% of people set off together, much like random's 10% — but random has no shape. +The one in the middle keeps hours: most people asleep at night, the cyan of work rising through the morning and ebbing through the evening, meals and company scattered across the day. Nobody drew up that rota; it is what three hundred slightly different sets of numbers work out separately. In its busiest ten minutes only 11% of people set off together, much like random's 9% — but random has no shape. Switch the backlog off and measure again (offline: 300 people, seed 1, the last of five days): sleeping, eating and company barely change, and free time goes from 11% to 40%. People do not find themselves something to do, and that is the most honest thing about this model: **it only answers the needs you give it.** If you want them to shop or exercise, that is another number. @@ -115,7 +121,7 @@ Drag the timeline back. The picture returns to that moment, but the simulation i **Need bars.** Between two events nothing about a person changes, so every need is a straight line. Each event notes the person's four numbers, and any later value is that number plus rate times elapsed time. -**Positions.** A "departed" event says where the walk started and where it is going, so the route can be found again (same start, same goal: A\* always gives the same path); the time of arrival then says how far along it the person is at this moment. A replay does not reproduce people stepping around each other, so positions are slightly off: measured over people who were walking, the median gap between the replayed and the real position is 6 units, about forty seconds of walking. +**Positions.** A "departed" event says where the walk started and where it is going, so the route can be found again (same start, same goal: A\* always gives the same path); the time of arrival then says how far along it the person is at this moment. A replay does not reproduce people stepping around each other, so positions are slightly off: the gap is a few tens of seconds of walking. The record keeps at most 5000 events, about ten hours of three hundred people. Older events are not simply dropped: they are folded into a snapshot of "the state when the record begins", so the start of the timeline is always a known state. diff --git a/content/posts/city-of-agents/zh.mdx b/content/posts/city-of-agents/zh.mdx index a0f8a32..7adb073 100644 --- a/content/posts/city-of-agents/zh.mdx +++ b/content/posts/city-of-agents/zh.mdx @@ -28,7 +28,13 @@ import { PARAMS } from "./components/sim/params"; 整座城市由一個數字(種子)決定,同一個種子永遠長出同一座城市。 -先畫 8 × 8 的街區,每四條路有一條是比較寬的主幹道。最東邊一排是河岸。其他街區看它離市中心多遠:近的是商業區,再外一圈是餐飲街和一些辦公樓,最外面是住宅區,中間隨機留幾塊公園(種子 1 是兩塊)。建築的高度是「一個隨機數 × 離市中心有多近」,所以中間高、邊緣矮。種子 1 長出來的是 373 棟建築:248 棟住家、29 棟辦公樓、96 間店面,最高的 80 公尺。 +先畫 8 × 8 的街區,再依序決定: + +- **路**:每四條路有一條是比較寬的主幹道,最東邊一排是河岸。 +- **分區**:看街區離市中心多遠。近的是商業區,再外一圈是餐飲街和一些辦公樓,最外面是住宅區,中間隨機留幾塊公園(種子 1 是兩塊)。 +- **高度**:「一個隨機數 × 離市中心有多近」,所以中間高、邊緣矮。 + +種子 1 長出來的是 373 棟建築:248 棟住家、29 棟辦公樓、96 間店面,最高的 80 公尺。 小人只走人行道和斑馬線。每個街區外圍一圈人行道,轉角和隔壁街區用斑馬線接起來,每棟建築在人行道上有一個門口。這些點和線就是一張圖:639 個節點、863 條邊。要去哪裡,就在這張圖上用 A\* 找最短路;同一對門口的路線會被記下來重複使用。 @@ -71,7 +77,7 @@ import { PARAMS } from "./components/sim/params"; -一開始 A 比較急但要走四十分鐘,B 沒那麼急但就在旁邊,A 贏(0.573 對 0.482)。把 B 的需求從 0.70 慢慢拉到 0.77,B 就反超了:七個百分點的急迫,抵得過三十五分鐘的路。在這座城市裡,出門一趟的中位數是 15 分鐘,所以路程通常只在兩件事差不多急的時候才有決定權,例如挑哪一間餐廳。 +一開始 A 比較急但要走四十分鐘,B 沒那麼急但就在旁邊,A 贏(0.573 對 0.482)。把 B 的需求從 0.70 慢慢拉到 0.77,B 就反超了:七個百分點的急迫,抵得過三十五分鐘的路。在這座城市裡,多數的路程都不長,所以路程通常只在兩件事差不多急的時候才有決定權,例如挑哪一間餐廳。 光有這條式子還不夠,有兩個地方會出事: @@ -95,7 +101,7 @@ import { PARAMS } from "./components/sim/params"; 隨機的一天是平的,半夜三點和下午三點長得一樣。 -中間那張有作息:夜裡大部分的人在睡,早上工作的青色慢慢漲上來,傍晚慢慢退掉,吃飯和社交散在整天。沒有人排這張班表,它是三百組稍微不一樣的數字各自算出來的結果。最擠的十分鐘只有 12% 的人同時出門,和隨機的 10% 差不多;但隨機沒有那個形狀。 +中間那張有作息:夜裡大部分的人在睡,早上工作的青色慢慢漲上來,傍晚慢慢退掉,吃飯和社交散在整天。沒有人排這張班表,它是三百組稍微不一樣的數字各自算出來的結果。最擠的十分鐘只有 11% 的人同時出門,和隨機的 9% 差不多;但隨機沒有那個形狀。 把「待辦事項」整條關掉再量一次(離線量的:300 人、種子 1、跑五天取最後一天):睡覺、吃飯、社交的時間幾乎不變,空出來的時間從 11% 變成 40%。小人不會自己找事做,這是這個模型最誠實的地方:**它只會回應你給它的需求。** 想要他們買東西、運動,就得再加一條數字。 @@ -113,7 +119,7 @@ import { PARAMS } from "./components/sim/params"; **需求條。** 兩筆事件之間,一個人的狀態不會變,所以每條需求都是一條直線。每筆事件順便記下他當時的四個數字,之後任何時刻的值就是「那個數字 + 速度 × 經過的時間」。 -**位置。** 「出發」那一筆記了從哪裡出發、要去哪裡,路線可以重新找出來;再看他是幾點抵達的,就知道這一刻走到了路線的幾分之幾。重播不會重現小人之間互相閃避,所以位置會差一點:我量過,走在路上的人,重播的位置和當時真正的位置差的中位數是 6 個單位,大約是他走四十秒的距離。 +**位置。** 「出發」那一筆記了從哪裡出發、要去哪裡,路線可以重新找出來;再看他是幾點抵達的,就知道這一刻走到了路線的幾分之幾。重播不會重現小人之間互相閃避,所以位置會差一點:差距大約是幾十秒的路。 記錄最多留 5000 筆,大約是三百個人的十個小時。更早的事件不是直接丟掉,而是先併進一張「記錄開始時的快照」,所以時間軸的起點永遠是一個已知的狀態。 From aa281ce249f84e1dc10344195036b562054ce9d9 Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 17:17:01 +0800 Subject: [PATCH 13/20] =?UTF-8?q?docs(scheduler):=20polish=20=E2=84=96=200?= =?UTF-8?q?10=20=E2=80=94=20the=20three=20edits=20Paul=20chose?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - "A job that takes over two hours" is the one-worker figure (142 min); the article runs four workers (38 min), so the sentence now says "a job that would take one worker over two hours". - Job length by workers is a table row; tests/scheduler/prose.test.ts now checks that row in both languages (it still recomputes the five numbers first). - The last-section bullet on what real schedulers handle is two: the plain rules, and gang scheduling on its own. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/task-scheduler/en.mdx | 14 +++++++++++--- content/posts/task-scheduler/zh.mdx | 14 +++++++++++--- tests/scheduler/prose.test.ts | 3 ++- 3 files changed, 24 insertions(+), 7 deletions(-) diff --git a/content/posts/task-scheduler/en.mdx b/content/posts/task-scheduler/en.mdx index 8694f6c..7a24c16 100644 --- a/content/posts/task-scheduler/en.mdx +++ b/content/posts/task-scheduler/en.mdx @@ -33,7 +33,7 @@ A job is a graph: every task remembers which tasks it is waiting for. The schedu 3. **Jump to the moment the next attempt ends**, and give that worker back. 4. **If that attempt failed, the task is not done**: it goes back among the tasks that can start and waits for the next free worker; whatever depends on it keeps waiting. Then back to rule 1. -No clock ticks. Time only moves when an attempt ends, which is why a job that takes over two hours is scheduled in about 8 microseconds (measured on my machine). The core is two functions and a loop (`Scheduler` in `sim/job.ts`; only the book-keeping lines are left out here): +No clock ticks. Time only moves when an attempt ends, which is why a job that would take one worker over two hours is scheduled in about 8 microseconds (measured on my machine). The core is two functions and a loop (`Scheduler` in `sim/job.ts`; only the book-keeping lines are left out here): ```ts assign() { // rules 1 and 2 @@ -117,7 +117,13 @@ Work done is not time passed, because the number of busy workers keeps changing. Drag the slider from 1 to 32. Two things happen together. -First, the job gets faster and then meets a horizontal line and stops (the upper chart). On average: 122 minutes with one worker, 63 with two, 38 with four, 31.4 with eight, and 31.0 from there on however many you add. Those 31 minutes are the longest chain of tasks waiting on each other; more hands can only wait with it. +First, the job gets faster and then meets a horizontal line and stops (the upper chart). On average: + +| Workers | 1 | 2 | 4 | 8 | 16 and more | +| --- | --- | --- | --- | --- | --- | +| The whole job (minutes) | 122 | 63 | 38 | 31.4 | 31.0 | + +Those 31 minutes are the longest chain of tasks waiting on each other; more hands can only wait with it. Second, the counting bar gets less honest: 10 points off with 4 workers, 17 with 16, a third of the run spent above 90%. More hands make the head of the job go faster; the tail is as long as ever. @@ -145,7 +151,9 @@ It has a price too: **it goes backwards.** When a task outruns its estimate, the **The task sizes are a distribution I chose** (log-normal). I have only checked its spread against this project's own two test suites (1.0 and 2.8, above), not against anyone else's workload; what it is for your build or your CI, the slider in figure 04 lets you set. -**The scheduling rule is the plainest there is**: whichever task is ready goes first, and a failed one is retried at once with no waiting; no priorities, no pre-emption, no worker itself ever fails, and no job needs four workers to start together. Those are exactly what real schedulers of the Kubernetes or Airflow kind have to deal with. Clusters that train large models add one more: every member of a job has to get a machine at the same moment, or the whole job waits (gang scheduling). What to do downstream when a task is cancelled, and who should go first, are other articles too. +**The scheduling rule is the plainest there is**: whichever task is ready goes first, and a failed one is retried at once with no waiting; no priorities, no pre-emption, no worker itself ever fails. Those are exactly what real schedulers of the Kubernetes or Airflow kind have to deal with; what to do downstream when a task is cancelled, and who should go first, are other articles too. + +**Nothing has to start together.** Clusters that train large models often need every member of a job to get a machine at the same moment, or the whole job waits (gang scheduling). Here every task is one worker's. **The unit is a whole job of tasks that wait for each other, not a single request.** How one request queues, gets routed, and how long to wait before retrying it, Sam Rose has covered in three fine interactive essays (see the sources); this article does not repeat them. diff --git a/content/posts/task-scheduler/zh.mdx b/content/posts/task-scheduler/zh.mdx index 319aea1..06d0a22 100644 --- a/content/posts/task-scheduler/zh.mdx +++ b/content/posts/task-scheduler/zh.mdx @@ -33,7 +33,7 @@ import { PARAMS } from "./components/sim/params"; 3. **把時間快轉到下一次嘗試結束的那一刻**,把那個 worker 放回來。 4. **那一次如果失敗了,任務不算做完**:它回到「可以開始」的那一堆,等下一個有空的 worker 再試一次;等它的任務繼續等。然後回到第 1 步。 -沒有時鐘在滴答走。時間只在「有一次嘗試結束」的時候往前跳,所以一份要跑兩個多小時的工作,排一次只要大約 8 微秒(我的機器上量的)。核心是兩個函式和一個迴圈(`sim/job.ts` 裡的 `Scheduler`,這裡只省略了記錄用的幾行): +沒有時鐘在滴答走。時間只在「有一次嘗試結束」的時候往前跳,所以一份交給一個 worker 要跑兩個多小時的工作,排一次只要大約 8 微秒(我的機器上量的)。核心是兩個函式和一個迴圈(`sim/job.ts` 裡的 `Scheduler`,這裡只省略了記錄用的幾行): ```ts assign() { // 規則 1 和 2 @@ -117,7 +117,13 @@ while (s.left > 0) { s.assign(); s.advance(); } // 整個排程 把滑桿從 1 拉到 32。有兩件事同時發生。 -第一,整份工作先是變快,然後撞到一條水平線就不再變快(上面那張圖)。我量的平均是:1 個 worker 122 分鐘,2 個 63,4 個 38,8 個 31.4,之後不管加到幾個都是 31.0。那個 31 分鐘是工作裡最長的一條「你等我、我等他」的鏈子,再多人手也只能等。 +第一,整份工作先是變快,然後撞到一條水平線就不再變快(上面那張圖)。我量的平均: + +| worker | 1 | 2 | 4 | 8 | 16 以上 | +| --- | --- | --- | --- | --- | --- | +| 整份工作(分鐘) | 122 | 63 | 38 | 31.4 | 31.0 | + +那個 31 分鐘是工作裡最長的一條「你等我、我等他」的鏈子,再多人手也只能等。 第二,數件數的進度條越來越不誠實:4 個 worker 時平均差 10 點,16 個時差 17 點,有三分之一的時間待在 90% 以上。人手越多,開頭衝得越快,尾巴還是一樣長。 @@ -145,7 +151,9 @@ while (s.left > 0) { s.assign(); s.advance(); } // 整個排程 **任務的大小是我選的分佈**(對數常態)。懸殊程度我只拿這個專案自己的兩組測試對過(1.0 和 2.8,見上),沒有對過別人的工作負載;你的編譯、你的 CI 是多少,圖 04 的滑桿可以自己拉。 -**排程的規則是最單純的那一種**:哪個任務先好就先做,失敗了立刻重試、不等待;沒有優先順序,沒有搶先,worker 自己不會壞,也沒有「這個工作要四個 worker 同時開始」這種限制。這些正是 Kubernetes、Airflow 這一類真的排程器要處理的事。訓練大模型的叢集還多一種:一個工作的所有成員要同時拿到機器,湊不齊就整組等(gang scheduling)。取消一個任務之後下游怎麼辦、誰該先做,也是另外的題目。 +**排程的規則是最單純的那一種**:哪個任務先好就先做,失敗了立刻重試、不等待;沒有優先順序,沒有搶先,worker 自己不會壞。這些正是 Kubernetes、Airflow 這一類真的排程器要處理的事;取消一個任務之後下游怎麼辦、誰該先做,也是另外的題目。 + +**沒有「要一起開始」的限制。** 訓練大模型的叢集常常要一個工作的所有成員同時拿到機器,湊不齊就整組等(gang scheduling)。這裡每個任務都是一個人做。 **單位是一整份互相等待的工作,不是一個個請求。** 一個請求怎麼排隊、怎麼分流、重試要等多久,Sam Rose 寫過三篇很好的互動文章(見來源),這篇沒有重講。 diff --git a/tests/scheduler/prose.test.ts b/tests/scheduler/prose.test.ts index 55c0d73..4be8d77 100644 --- a/tests/scheduler/prose.test.ts +++ b/tests/scheduler/prose.test.ts @@ -41,7 +41,8 @@ describe("the numbers in the article", () => { const jobs = Array.from({ length: 100 }, (_, k) => makeJob(k + 1, GOOD)), minutes = (w: number) => jobs.reduce((s, j) => s + schedule(j.tasks, w, j.tasks.map((x) => x.attempts)).total, 0) / 100; expect([1, 2, 4].map((w) => Math.round(minutes(w)))).toEqual([122, 63, 38]); expect([8, 16, 32].map((w) => minutes(w).toFixed(1))).toEqual(["31.4", "31.0", "31.0"]); - says("1 個 worker 122 分鐘,2 個 63,4 個 38,8 個 31.4", "122 minutes with one worker, 63 with two, 38 with four, 31.4 with eight"); + // The prose gives them as a table row, in both languages. + says("| 整份工作(分鐘) | 122 | 63 | 38 | 31.4 | 31.0 |", "| The whole job (minutes) | 122 | 63 | 38 | 31.4 | 31.0 |"); }, 60_000); it("how honest the bars are, with good estimates and with biased ones", () => { From 78d61942c42d603d9db752e8e2c1e9f35804fc9e Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 17:20:37 +0800 Subject: [PATCH 14/20] =?UTF-8?q?docs(light):=20polish=20=E2=84=96=20011?= =?UTF-8?q?=20=E2=80=94=20the=20line=20count=20as=20it=20is=20now,=20and?= =?UTF-8?q?=20the=20sources=20as=20a=20list?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - "About 1,100 lines" was the engine before part two grew it: lib/rt is 1,584 lines today, and Paul chose to count all of it, so the text says about 1,600, with what is counted in a sidenote instead of a long parenthesis. - "Where the numbers come from" is five bullets instead of one paragraph. Every other number was checked against docs/research/light/RESULTS.md and the kernel; none changed. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/light-from-noise/en.mdx | 10 ++++++++-- content/posts/light-from-noise/zh.mdx | 10 ++++++++-- 2 files changed, 16 insertions(+), 4 deletions(-) diff --git a/content/posts/light-from-noise/en.mdx b/content/posts/light-from-noise/en.mdx index d00e8c7..d4d8de1 100644 --- a/content/posts/light-from-noise/en.mdx +++ b/content/posts/light-from-noise/en.mdx @@ -19,7 +19,7 @@ The room below has had one sample so far, and it is almost all snow. Press Start Not one pixel of that picture is drawn. There is no code for shadows and no code that says the red wall should tint the floor. Nobody wrote why the corners the lamp cannot reach are still a little bright. The whole picture does one thing, many times: **shoot a ray out through a pixel, let it bounce around the room at random, and see whether it ends up at the lamp.** -This is path tracing. This article writes it from scratch: the scene, the acceleration structure, a reference version on the CPU and the WGSL kernel on the GPU, about 1,100 lines in all (TypeScript, plus the WGSL written for the GPU; what part two needs outdoors is counted in), with no graphics library. Then it answers a few questions. How does one ray become a colour? How long does the noise take to clear, and can it clear faster? What more do metal and glass need? And why does it still run when a thousand triangles become a million? +This is path tracing. This article writes it from scratch: the scene, the acceleration structure, a reference version on the CPU and the WGSL kernel on the GPU, about 1,600 lines in all, with no graphics library.TypeScript, plus the WGSL written for the GPU; everything part two's playground needs is counted in. Then it answers a few questions. How does one ray become a colour? How long does the noise take to clear, and can it clear faster? What more do metal and glass need? And why does it still run when a thousand triangles become a million? This is part one. [Part two](/en/posts/light-playground) takes the same engine outdoors: a playground to walk, drive and fly a helicopter and an aeroplane through, every frame computed this way. @@ -126,7 +126,13 @@ The tree is built by binned SAH: whenever a set of triangles has to be split in **Nothing moves.** When the camera or the scene moves, every accumulated sample is void. That is the subject of [part two](/en/posts/light-playground). -Where the numbers come from: the readouts in the figures are measured by your browser as it runs, so they will differ from mine. Where the text says "I measured", that was Chrome on an M4 Pro, in this page, recorded in this project's `docs/research/light/RESULTS.md`. That asking the lamp does not change the answer is held by a test: on three pixels of the CPU version, 6,000 paths each, the means with and without asking differ by no more than the noise allows (`tests/rt/strategies.test.ts`). The CPU and GPU versions are the same algorithm written twice: on eight pixels of figure 02 (both coloured walls, floor, ceiling, back wall, a torus, the lamp) the mean of 2,001 CPU paths and the GPU's 1,024 samples differ by at most 12 levels out of 255 in any channel, some above and some below. That the BVH finds the same hit as testing every triangle is held by the tests in `tests/rt/`. A browser without WebGPU gets a picture rendered earlier, and figure 02 still works on it, because its rays are the CPU's. +Where the numbers come from: + +- **The readouts in the figures** are measured by your browser as it runs, so they will differ from mine. +- **Where the text says "I measured"**, that was Chrome on an M4 Pro, in this page, recorded in this project's `docs/research/light/RESULTS.md`. +- **Asking the lamp does not change the answer**: on three pixels of the CPU version, 6,000 paths each, the means with and without asking differ by no more than the noise allows (`tests/rt/strategies.test.ts`). +- **The CPU and GPU versions** are the same algorithm written twice: on eight pixels of figure 02 (both coloured walls, floor, ceiling, back wall, a torus, the lamp) the mean of 2,001 CPU paths and the GPU's 1,024 samples differ by at most 12 levels out of 255 in any channel, some above and some below. That the BVH finds the same hit as testing every triangle is held by the tests in `tests/rt/`. +- **A browser without WebGPU** gets a picture rendered earlier, and figure 02 still works on it, because its rays are the CPU's. ## Sources diff --git a/content/posts/light-from-noise/zh.mdx b/content/posts/light-from-noise/zh.mdx index 0b4899e..1741967 100644 --- a/content/posts/light-from-noise/zh.mdx +++ b/content/posts/light-from-noise/zh.mdx @@ -19,7 +19,7 @@ import { BouncesLab, BvhLab, ConvergeLab, MaterialLab, PathLab } from "./compone 畫面裡沒有一個像素是「畫」上去的。沒有陰影的程式,沒有「紅牆要把地板染紅」的程式,天花板那盞燈照不到的角落為什麼還有一點亮,也沒有人寫。整個畫面只做一件事,做了很多次:**從一個像素射一條光線出去,讓它在房間裡亂彈,看它最後有沒有碰到燈。** -這種做法叫路徑追蹤(path tracing)。這一篇把它從零寫出來:場景、加速結構、CPU 上的對照版、GPU 上的 WGSL 核心,一共約一千一百行(TypeScript,加上寫給 GPU 的 WGSL;下篇戶外要用的部分也算在裡面),沒有用任何繪圖函式庫。然後回答幾個問題:一條光線怎麼變成一個顏色?雜訊要多久才會散,能不能讓它散快一點?金屬和玻璃要多寫什麼?三角形從一千個變成一百萬個,為什麼還跑得動? +這種做法叫路徑追蹤(path tracing)。這一篇把它從零寫出來:場景、加速結構、CPU 上的對照版、GPU 上的 WGSL 核心,一共約一千六百行,沒有用任何繪圖函式庫。TypeScript 加上寫給 GPU 的 WGSL;下篇遊樂園要用的部分也全部算在裡面。然後回答幾個問題:一條光線怎麼變成一個顏色?雜訊要多久才會散,能不能讓它散快一點?金屬和玻璃要多寫什麼?三角形從一千個變成一百萬個,為什麼還跑得動? 這是上篇。[下篇](/zh/posts/light-playground)把同一個引擎搬到戶外:一座可以走、可以開車、可以開直升機和飛機的遊樂園,每一幀都是這樣算出來的。 @@ -126,7 +126,13 @@ import { BouncesLab, BvhLab, ConvergeLab, MaterialLab, PathLab } from "./compone **東西不會動。** 相機和場景一動,累積的樣本就全部作廢。這是[下篇](/zh/posts/light-playground)的題目。 -數字的來源:圖裡的讀數都是你的瀏覽器現場量的,所以會和我的不一樣。文中寫「我量過」的,是在 M4 Pro 的 Chrome 上、在這一頁裡量的,紀錄在這個專案的 `docs/research/light/RESULTS.md`。「問燈」不會改變答案這件事有測試守著:CPU 版的三個像素、各 6,000 條路徑,問和不問的平均值差距都在雜訊容許的範圍內(`tests/rt/strategies.test.ts`)。CPU 版和 GPU 版是同一套演算法各寫一次:我在圖 02 挑了八個像素(兩面色牆、地板、天花板、後牆、環面、燈),CPU 的 2,001 條路徑平均和 GPU 的 1,024 個樣本,每個色版最多差 12 級(滿分 255),有高有低,沒有一邊倒;BVH 找到的交點和逐一測試每個三角形的結果一致,由 `tests/rt/` 的測試保證。沒有 WebGPU 的瀏覽器會看到一張預先算好的圖,圖 02 在那張圖上照樣能用,因為它的光線是 CPU 算的。 +數字的來源: + +- **圖裡的讀數**是你的瀏覽器現場量的,所以會和我的不一樣。 +- **文中寫「我量過」的**,是在 M4 Pro 的 Chrome 上、在這一頁裡量的,紀錄在這個專案的 `docs/research/light/RESULTS.md`。 +- **「問燈」不會改變答案**:CPU 版的三個像素、各 6,000 條路徑,問和不問的平均值差距都在雜訊容許的範圍內(`tests/rt/strategies.test.ts`)。 +- **CPU 版和 GPU 版**是同一套演算法各寫一次:我在圖 02 挑了八個像素(兩面色牆、地板、天花板、後牆、環面、燈),CPU 的 2,001 條路徑平均和 GPU 的 1,024 個樣本,每個色版最多差 12 級(滿分 255),有高有低,沒有一邊倒;BVH 找到的交點和逐一測試每個三角形的結果一致,由 `tests/rt/` 的測試保證。 +- **沒有 WebGPU 的瀏覽器**會看到一張預先算好的圖,圖 02 在那張圖上照樣能用,因為它的光線是 CPU 算的。 ## 來源 From 6173f5ef62a32b7191d77b75e66386ae3834d6ab Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 17:27:40 +0800 Subject: [PATCH 15/20] =?UTF-8?q?docs(light-playground):=20polish=20?= =?UTF-8?q?=E2=84=96=20012=20=E2=80=94=20what=20changed=20since=20it=20was?= =?UTF-8?q?=20written,=20measured=20again?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Two limits were no longer true: the aeroplane's weight change is ported (as landing gear that softens with speed), and doors do swing open while driving; the text said neither, and contradicted its own lamp paragraph. - Measured again with the original method (docs/research/light/RESULTS.md, 2026-09-22): 6.7 ms a sample (was 6.8), 116–117 fps while moving (was 121, before the lamps), 11,317 moving triangles (was 11,275; the lenses add 42). - The long paragraph on what was ported is three bullets (person, car, aircraft); the door sentence moves there from the lamp paragraph; the no-WebGPU line says why a still picture would be pointless. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/light-playground/en.mdx | 16 ++++++++++------ content/posts/light-playground/zh.mdx | 16 ++++++++++------ docs/research/light/RESULTS.md | 8 ++++++++ 3 files changed, 28 insertions(+), 12 deletions(-) diff --git a/content/posts/light-playground/en.mdx b/content/posts/light-playground/en.mdx index 0c2c461..e0ddd0c 100644 --- a/content/posts/light-playground/en.mdx +++ b/content/posts/light-playground/en.mdx @@ -19,11 +19,11 @@ Play first. Press "Enter the playground", then click the picture with a mouse an [Part one](/en/posts/light-from-noise) ended on a sentence: when the camera or the scene moves, every accumulated sample is void. That room needed hundreds of samples to be clean. Things here move sixty times a second or more, and a frame has time for one sample. -This part is about four things that together make the figure above playable. The result first (my machine, 960×540, full path tracing): 6.8 ms a sample, keeping up with a 120 Hz display while moving (121 frames a second measured), and in motion a fifth of the noise of "start from nothing every frame". +This part is about four things that together make the figure above playable. The result first (my machine, 960×540, full path tracing): 6.7 ms a sample, nearly keeping up with a 120 Hz display while moving (116 to 117 frames a second measured), and in motion a fifth of the noise of "start from nothing every frame". ## One: the world is built once, and what moves only gets new boxes -Part one's BVH took a second to build and then never changed. Here there is a person, five cars, a helicopter and an aeroplane, 11,275 moving triangles in all, and rebuilding every frame is out of the question. +Part one's BVH took a second to build and then never changed. Here there is a person, five cars, a helicopter and an aeroplane, 11,317 moving triangles in all, and rebuilding every frame is out of the question. So there are two trees. The playground itself (ground, ramps, tracks) is built once; what moves has a tree of its own, a ray walks both, and the nearer hit wins. @@ -85,9 +85,13 @@ I measured this wrong once myself. The first numbers made no sense, and the caus The scene's geometry, the character, the vehicles and all 34 animation clips are from [Sketchbook](https://github.com/swift502/Sketchbook) by Jan Blaha (swift502), under the MIT licence. Its textures are not used: the playground's are photographs that may not be redistributed, and the vehicles' are baked light, which is the very thing this article computes as you watch. So the packed files keep vertices, the skeleton, the animations and the materials' names, and the colours are mine. -The character behaves by Sketchbook's state machine, ported state by state: a start-walk clip chosen by the angle to turn, a stop, turning on the spot, sprinting, two jumps, three landings chosen by how hard the ground was hit; after F it walks to the nearer door by itself, opens it, sits down, closes it from inside (and from the passenger's side, slides over to the wheel), and closes it again after getting out. G gets in as a passenger instead, who stays put; X slides over to the connected seat, and V is first person. The physics is Rapier: the ground is one triangle mesh, the person a capsule, a car four rays on springs (five gears that shift by themselves, a steering wheel that turns, and keys that spin it while it is in the air), the helicopter and the aeroplane rigid bodies, flown by Sketchbook's own arithmetic and keys: W S pitch, A D roll, Q E yaw, Shift to climb or for throttle. The helicopter's engine takes five seconds to come up and it levels itself when you let go; the aeroplane leaves the ground after about eight seconds of full throttle, and its ailerons, elevators and rudder move with the keys. +Behaviour and physics are both ported from Sketchbook piece by piece; the physics engine is Rapier, and the ground is one triangle mesh: -Drag the time of day past the evening and it gets dark; the cars' head lamps and the helicopter's searchlight come on by themselves (L switches them by hand). The models have no lamps of their own, so the packer adds them: a ray from straight ahead, and a glowing lens where it first meets the body. What lights the road is not that lens but a spot lamp, asked the way the sun is asked: at every bounce one of up to twelve lamps is picked in proportion to what it could give that point, and one shadow ray goes to it. It costs 2 ms a sample (I measured 8.4 → 10.4 ms), and for that a car's lamps fall on the next car and a person standing in the beam casts a shadow, with nothing written for either. A door left open while driving is swung by the car's acceleration, and shuts itself if it swings hard enough. +- **The person** is a capsule, run by Sketchbook's state machine, ported state by state: a start-walk clip chosen by the angle to turn, a stop, turning on the spot, sprinting, two jumps, three landings chosen by how hard the ground was hit. After F it walks to the nearer door by itself, opens it, sits down, closes it from inside (and from the passenger's side, slides over to the wheel), and closes it again after getting out. G gets in as a passenger instead, who stays put; X slides over to the connected seat, and V is first person. +- **A car** is four rays on springs: five gears that shift by themselves, a steering wheel that turns, and keys that spin it while it is in the air. A door left open while driving is swung by the car's acceleration, and shuts itself if it swings hard enough. +- **The helicopter and the aeroplane** are rigid bodies, flown by Sketchbook's own arithmetic and keys: W S pitch, A D roll, Q E yaw, Shift to climb or for throttle. The helicopter's engine takes five seconds to come up and it levels itself when you let go; the aeroplane leaves the ground after about eight seconds of full throttle, and its ailerons, elevators and rudder move with the keys. + +Drag the time of day past the evening and it gets dark; the cars' head lamps and the helicopter's searchlight come on by themselves (L switches them by hand). The models have no lamps of their own, so the packer adds them: a ray from straight ahead, and a glowing lens where it first meets the body. What lights the road is not that lens but a spot lamp, asked the way the sun is asked: at every bounce one of up to twelve lamps is picked in proportion to what it could give that point, and one shadow ray goes to it. It costs 2 ms a sample (I measured 8.4 → 10.4 ms), and for that a car's lamps fall on the next car and a person standing in the beam casts a shadow, with nothing written for either. The character's skeleton (14 bones, 186 triangles) is evaluated on the CPU every frame and goes straight into the moving tree. @@ -101,9 +105,9 @@ The character's skeleton (14 bones, 186 triangles) is evaluated on the CPU every **Reflections and the sea trail.** What a mirror shows moves with the camera, but the past is looked up by where the surface itself was, so reflections in car paint and on the sea are half a beat late. -**The flying is Sketchbook's arcade flying.** The helicopter balances itself; the aeroplane has no real aerodynamics, its velocity is only bent a little at a time towards where the nose points, so holding S takes it straight over in a loop, and pulling up when it is slow drops it. Sketchbook also makes the aeroplane lighter with speed; that part is not here. A door does not fly open while driving. The suspension numbers are mine, because the two physics engines' springs are not alike; the engine's force is Sketchbook's figure scaled to this car's weight. +**The flying is Sketchbook's arcade flying.** The helicopter balances itself; the aeroplane has no real aerodynamics, its velocity is only bent a little at a time towards where the nose points, so holding S takes it straight over in a loop, and pulling up when it is slow drops it. Sketchbook also makes the aeroplane lighter with speed; Rapier cannot change a body's mass in flight, so here the landing gear softens with speed instead, to the same effect. The suspension numbers are mine, because the two physics engines' springs are not alike; the engine's force is Sketchbook's figure scaled to this car's weight. -**Without WebGPU there is nothing to play.** Part one could show a picture rendered earlier; here that would be pointless. +**Without WebGPU there is nothing to play.** Part one could show a picture rendered earlier; this one is about things that move, and a still picture would be pointless. Where the numbers come from: the readouts in the figure are measured by your browser as it runs. Where the text says "measured", that was Chrome on an M4 Pro at 960×540, with the camera at the spawn point looking over the car park; the method and the full tables are in this project's `docs/research/light/RESULTS.md`. The cars', the helicopter's and the aeroplane's physics and the character's state machine each have automatic tests that need no GPU (`tests/rt/`). diff --git a/content/posts/light-playground/zh.mdx b/content/posts/light-playground/zh.mdx index 49ccf77..4d004da 100644 --- a/content/posts/light-playground/zh.mdx +++ b/content/posts/light-playground/zh.mdx @@ -19,11 +19,11 @@ import { PlaygroundLab } from "./components"; [上篇](/zh/posts/light-from-noise)的結尾留了一句話:相機和場景一動,累積的樣本就全部作廢。那個房間要幾百個樣本才乾淨,而這裡的東西每秒要動六十次以上,每一幀只來得及算一個樣本。 -這一篇講四件事,它們合起來才讓上面那張圖玩得動。先說結果(我的機器,960×540,完整光追):一個樣本 6.8 ms,移動時跟得上 120 Hz 的螢幕(量到每秒 121 幀),移動中的雜訊只剩「每幀從零開始」的五分之一。 +這一篇講四件事,它們合起來才讓上面那張圖玩得動。先說結果(我的機器,960×540,完整光追):一個樣本 6.7 ms,移動時幾乎跟得上 120 Hz 的螢幕(量到每秒 116 到 117 幀),移動中的雜訊只剩「每幀從零開始」的五分之一。 ## 第一件事:世界蓋一次,會動的每幀只重算盒子 -上篇的 BVH 要蓋一秒,蓋好就不動了。這裡有一個人、五台車、一架直升機、一架飛機,一共 11,275 個會動的三角形,不可能每一幀重蓋。 +上篇的 BVH 要蓋一秒,蓋好就不動了。這裡有一個人、五台車、一架直升機、一架飛機,一共 11,317 個會動的三角形,不可能每一幀重蓋。 所以有兩棵樹。遊樂園本身(地面、坡道、軌道)蓋一次;會動的東西另外一棵,光線兩棵都走,取比較近的那個交點。 @@ -85,9 +85,13 @@ import { PlaygroundLab } from "./components"; 場景的幾何、人物、載具和全部 34 段動畫,來自 Jan Blaha(swift502)的 [Sketchbook](https://github.com/swift502/Sketchbook),MIT 授權。原始的貼圖沒有用:遊樂園的貼圖是不能轉散佈的照片,載具的貼圖是預先烘好的光影,而那正是這一篇要現場算的東西。所以打包的時候只留下頂點、骨架、動畫和材質的名字,顏色是我重新配的。 -人物的行為照 Sketchbook 的狀態機移植,一個狀態一個狀態對:依要轉的角度選起步動畫、急停、原地轉身、衝刺、兩種跳、依落地的力道選三種落地;按 F 之後自己走到比較近的那一扇門、開門、坐下、從裡面關門(從副駕上車的話,再滑到駕駛座),下車再關一次。按 G 是當乘客,坐下就不動;X 換到相連的座位,V 是第一人稱。物理用 Rapier:地面是一整個三角網格,人是一顆膠囊,車是四條帶彈簧的射線(五個檔自己換,方向盤會轉,飛起來的時候方向鍵能讓它在空中翻),直升機和飛機是剛體,飛行的算法和按鍵也照 Sketchbook:W S 俯仰、A D 側傾、Q E 轉向,Shift 是上升或油門。直升機引擎要五秒才轉得起來,放手會自己回正;飛機加滿油門大約八秒離地,副翼、升降舵和方向舵會跟著按鍵擺動。 +行為和物理都照 Sketchbook 一個一個移植,物理引擎用 Rapier,地面是一整個三角網格: -把「太陽的時間」拉過傍晚,天就黑了,車燈和直升機的探照燈會自己亮(L 可以手動開關)。模型本身沒有燈,燈是打包時加上去的:從正前方射一條線,碰到車身的地方貼一片會發光的鏡片。照亮路面的不是那片鏡片,是一盞聚光燈,問法和問太陽一樣:每一次反彈,從最多十二盞燈裡依「這盞燈能給這個點多少光」的比例挑一盞,射一條陰影線。這樣一個樣本多 2 ms(我量到 8.4 → 10.4 ms),換來的是車燈打在別台車上、人站在光柱裡有影子,全部不用另外寫。行駛中沒關好的車門會被加速度甩動,甩得夠用力就自己關上。 +- **人**:一顆膠囊,照 Sketchbook 的狀態機一個狀態一個狀態對:依要轉的角度選起步動畫、急停、原地轉身、衝刺、兩種跳、依落地的力道選三種落地。按 F 之後自己走到比較近的那一扇門、開門、坐下、從裡面關門(從副駕上車的話,再滑到駕駛座),下車再關一次。按 G 是當乘客,坐下就不動;X 換到相連的座位,V 是第一人稱。 +- **車**:四條帶彈簧的射線。五個檔自己換,方向盤會轉,飛起來的時候方向鍵能讓它在空中翻;行駛中沒關好的車門會被加速度甩動,甩得夠用力就自己關上。 +- **直升機和飛機**:剛體,飛行的算法和按鍵也照 Sketchbook:W S 俯仰、A D 側傾、Q E 轉向,Shift 是上升或油門。直升機引擎要五秒才轉得起來,放手會自己回正;飛機加滿油門大約八秒離地,副翼、升降舵和方向舵會跟著按鍵擺動。 + +把「太陽的時間」拉過傍晚,天就黑了,車燈和直升機的探照燈會自己亮(L 可以手動開關)。模型本身沒有燈,燈是打包時加上去的:從正前方射一條線,碰到車身的地方貼一片會發光的鏡片。照亮路面的不是那片鏡片,是一盞聚光燈,問法和問太陽一樣:每一次反彈,從最多十二盞燈裡依「這盞燈能給這個點多少光」的比例挑一盞,射一條陰影線。這樣一個樣本多 2 ms(我量到 8.4 → 10.4 ms),換來的是車燈打在別台車上、人站在光柱裡有影子,全部不用另外寫。 人物每一幀在 CPU 上算骨架(14 根骨頭,186 個三角形),算完直接進那棵會動的樹。 @@ -101,9 +105,9 @@ import { PlaygroundLab } from "./components"; **反射和海面會拖影。** 鏡面裡看到的東西會隨相機移動,但我是用表面本身的位置去找上一幀,所以車漆和海面上的倒影是慢半拍的。 -**飛行是 Sketchbook 的街機式飛行。** 直升機自己會平衡;飛機沒有真的氣動力,只是把速度一點一點拗向機頭的方向,所以按住 S 會直接翻一個筋斗,太慢的時候拉機頭就掉下來。Sketchbook 會讓飛機隨速度變輕,這裡沒有做。車門在行駛中不會被甩開。懸吊的數字是我調的,因為兩個物理引擎的彈簧算法不一樣;引擎的力則是把 Sketchbook 的數字照車重放大。 +**飛行是 Sketchbook 的街機式飛行。** 直升機自己會平衡;飛機沒有真的氣動力,只是把速度一點一點拗向機頭的方向,所以按住 S 會直接翻一個筋斗,太慢的時候拉機頭就掉下來。Sketchbook 會讓飛機隨速度變輕;Rapier 沒辦法在飛行中改質量,所以這裡改成讓起落架隨速度變軟,效果相同。懸吊的數字是我調的,因為兩個物理引擎的彈簧算法不一樣;引擎的力則是把 Sketchbook 的數字照車重放大。 -**沒有 WebGPU 就玩不了。** 上篇還能放一張預先算好的圖,這一篇沒有意義。 +**沒有 WebGPU 就玩不了。** 上篇還能放一張預先算好的圖;這一篇的重點是會動,放一張靜止的圖沒有意義。 數字的來源:圖上的讀數是你的瀏覽器現場量的。文中寫「我量到」的,是 M4 Pro 的 Chrome、960×540、鏡頭在出生點看著停車場,方法和完整的表在這個專案的 `docs/research/light/RESULTS.md`。車、直升機、飛機的物理,和人物的狀態機,各有不需要 GPU 的自動測試(`tests/rt/`)。 diff --git a/docs/research/light/RESULTS.md b/docs/research/light/RESULTS.md index 0f40dd8..81d6189 100644 --- a/docs/research/light/RESULTS.md +++ b/docs/research/light/RESULTS.md @@ -224,3 +224,11 @@ the same night with the lamps switched off (L): 8.4 ms; by day (16:00): 8.0 ms. the one the article's 6.8 ms came from; compare the three with each other, not with that.) Strength 220 in the sun's units: the sun gives 18000 × 2.14e-3 ≈ 38.5 on a facing surface, the night exposure is 6×, so 220/d² × 6 is a third of that at 10 m. + +## 2026-09-22 — the opening numbers of part two, measured again + +The same view and method as the 2026-09-21 table above (player's spawn over the car park, 960×540, 8 bounces, four +16-sample bursts through `window.__lights`), now with the lamps' lenses in the models and the spot-lamp loop in the +kernel (no lamp lit by day): 6.72 and 6.59 ms a sample on two fresh loads, 35.4 node visits per ray; 11,317 moving +triangles (the figure's readout; +42 from the lenses). Frame rate while strafing, one sample a frame, three 3 s +windows: 117, 116, 117 fps (121 with nothing moving). The adaptive quality stayed at 960×540. From 01404391e71cb5b21e799929f638a8fc4cc8b22d Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 17:32:03 +0800 Subject: [PATCH 16/20] =?UTF-8?q?docs(head-camera):=20polish=20=E2=84=96?= =?UTF-8?q?=20013=20=E2=80=94=20two=20seed=20scores=20corrected,=20"twice"?= =?UTF-8?q?=20is=20three=20times,=20a=20paragraph=20split?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - The five seeds at 20 000 steps are 100, 97, 59.5, 36 and 2.5 on the page's layouts (docs/research/head-camera/run-22/closed-shake-48-*.json); the text had 59 and 37. - 20 000 → 60 000 steps is three times, as the English already said; the Chinese said 多練兩倍. - The hand-hides-the-block paragraph is two: the grasp tolerance, and the blind strip. Three numbers without a record (190 s in Chromium, 88–98.5 % over five trainings, the look-once failure row) are unchanged, waiting on the feat/vla session's answer. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/head-camera/en.mdx | 6 ++++-- content/posts/head-camera/zh.mdx | 6 ++++-- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/content/posts/head-camera/en.mdx b/content/posts/head-camera/en.mdx index 09b8b4b..94cad01 100644 --- a/content/posts/head-camera/en.mdx +++ b/content/posts/head-camera/en.mdx @@ -68,7 +68,7 @@ Four ways of doing it, each the mean of five random seeds: Read across: the two rows that never saw a shaken camera go to almost nothing the moment it tilts; the two that did barely notice. Read down: when something is moved halfway through, “look once” cannot follow and “keep looking” does not care, because it was looking again at every step anyway. -The last two rows both trained for 60,000 steps and are equally good with the camera untouched (96% and 95%); they differ only once it tilts. In the last row all five seeds are above 92%, and that number was hard won: at 20,000 steps the five seeds were 100, 97, 59, 37 and 2.5. The method was the same; some seeds simply learn slowly, and three times the steps brought every one of them in.The two “look once” rows trained for only 10,000 steps: it sees one picture an episode, and more training does not give it a second look. +The last two rows both trained for 60,000 steps and are equally good with the camera untouched (96% and 95%); they differ only once it tilts. In the last row all five seeds are above 92%, and that number was hard won: at 20,000 steps the five seeds were 100, 97, 59.5, 36 and 2.5. The method was the same; some seeds simply learn slowly, and three times the steps brought every one of them in.The two “look once” rows trained for only 10,000 steps: it sees one picture an episode, and more training does not give it a second look. ## Why picking up is harder than touching @@ -78,7 +78,9 @@ I failed at this task three times before it trained. **A rare action needs extra practice.** "Grip" only comes up when the hand is low and right over the block: under 1% of training pictures drawn at random. The network learned "almost never grip". With a fifth of the training pictures drawn from that region, it dared to. -**The hand hides the block.** Once the hand is low over the block, seen from the head, it covers the block, and there is nothing left to line up on for the last two centimetres. That is why this gripper forgives 3 cm. The same thing has a bigger version, which is why fig. 01's dashed outline does not cover the whole bench: outside it, on the side away from the camera, the reaching hand and forearm come between the camera and the block, and with the hand still 8 cm away not one pixel of the block is left in the picture. The network has no memory; what it cannot see, it cannot head for. I measured the bench every 2 cm: in every cell inside the outline it places at least 19 blocks in 20; outside there is a large patch where it never places one, so the page does not let the block go there. The pad is large and flat and is fine anywhere. +**The hand hides the block.** Once the hand is low over the block, seen from the head, it covers the block, and there is nothing left to line up on for the last two centimetres. That is why this gripper forgives 3 cm. + +The same thing has a bigger version, which is why fig. 01's dashed outline does not cover the whole bench: outside it, on the side away from the camera, the reaching hand and forearm come between the camera and the block, and with the hand still 8 cm away not one pixel of the block is left in the picture. The network has no memory; what it cannot see, it cannot head for. I measured the bench every 2 cm: in every cell inside the outline it places at least 19 blocks in 20; outside there is a large patch where it never places one, so the page does not let the block go there. The pad is large and flat and is fine anywhere. Figure 03 put cameras in the palms, and the stated reason is close-range pictures "when the main cameras are occluded". Here you can see why. diff --git a/content/posts/head-camera/zh.mdx b/content/posts/head-camera/zh.mdx index 997ea03..83c7524 100644 --- a/content/posts/head-camera/zh.mdx +++ b/content/posts/head-camera/zh.mdx @@ -68,7 +68,7 @@ Figure 的人形機器人在倉庫裡揀包裹,眼睛是頭上的相機。「看一眼」的兩列只練了 10 000 步:它每一回合只看一張照片,多練也不會讓它看第二眼。 +後兩列都練了 60 000 步,相機沒動時一樣好(96% 和 95%),差別只在相機歪掉之後。最後一列五個種子都在 92% 以上,這個數字得來不易:只練 20 000 步時,五個種子是 100、97、59.5、36 和 2.5。差的不是方法,是學得快慢,同樣的種子練到三倍的步數就全部到位。「看一眼」的兩列只練了 10 000 步:它每一回合只看一張照片,多練也不會讓它看第二眼。 ## 拿起來,比碰到難在哪 @@ -78,7 +78,9 @@ Figure 的人形機器人在倉庫裡揀包裹,眼睛是頭上的相機。 Date: Tue, 22 Sep 2026 17:39:26 +0800 Subject: [PATCH 17/20] docs(research): head-camera run 25, and two measurements that were never written down MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Run 25 re-measures the light-and-noise table on the current layouts (light columns identical; noise columns vary by draw, three draws recorded). RESULTS.md now records fig. 03's in-browser training time (191.7 s and 200 s, from the feat/vla session) and where the article's 88–98.5 % comes from (run 9). Co-Authored-By: Claude Opus 5 (1M context) --- docs/research/head-camera/RESULTS.md | 10 ++++++ docs/research/head-camera/run-25/output.txt | 14 ++++++++ .../head-camera/run-25/run25.test.ts.txt | 34 +++++++++++++++++++ 3 files changed, 58 insertions(+) create mode 100644 docs/research/head-camera/run-25/output.txt create mode 100644 docs/research/head-camera/run-25/run25.test.ts.txt diff --git a/docs/research/head-camera/RESULTS.md b/docs/research/head-camera/RESULTS.md index 13a158e..77edc6f 100644 --- a/docs/research/head-camera/RESULTS.md +++ b/docs/research/head-camera/RESULTS.md @@ -513,3 +513,13 @@ and `pick-fixed.json` is run 23's seed 11 (on 200 page layouts: 99.5 % untouched - The hand moves in a plane. No grasp, no descent, no gripper. - 32 × 32 only. Whether 48 × 48 buys the last few points, and what it costs in seconds, is unmeasured. - The keypoints have been measured (run 4) but not yet looked at as pictures under a turned camera. + +## Run 25 — the light-and-noise table again, and where two of the article's numbers came from (2026-09-22) + +`run-25/`: the three shipped checkpoints on the current (cut) layouts. Light columns identical to the article's table, +look-once included; the noise columns vary between draws (Math.random) and the article's single draws are low. Also +recorded now, from the feat/vla session: fig. 03 trained in headless Chromium on the M4 Pro took 191.7 s wall (its +readout said 190 s; 90 % untouched) and a second full run 200 s (84 %); the line "Page seconds are measured in node, +not yet in a browser tab" below is out of date. The article's "88 % to 98.5 %" is run 9's five seeds (98.5 / 91.0 / +88.0 / 96.0 / 92.5), trained by the research script in Node, which trains identically to the page +(`tests/head-camera/trainer.test.ts`). diff --git a/docs/research/head-camera/run-25/output.txt b/docs/research/head-camera/run-25/output.txt new file mode 100644 index 0000000..851f1bc --- /dev/null +++ b/docs/research/head-camera/run-25/output.txt @@ -0,0 +1,14 @@ +Run 25 (2026-09-22): the light-and-noise table for the three shipped checkpoints, camera untouched, on the page's +current layouts (block inside the cut BLOCK_AREA), 200 episodes a condition, layout() from mulberry32(99). The script +is run25.test.ts.txt beside this file (copy it into tests/ to run). Noise comes from Math.random, so the two noise +columns were run three times; the light columns are deterministic and came out the same every time. + +| | normal | light 80% | light 70% | light 50% | noise 0.05 (3 runs) | noise 0.15 (3 runs) | +| --- | --- | --- | --- | --- | --- | --- | +| pick-open (look once) | 71.0 | 72.0 | 58.0 | 11.0 | 19.0 / 17.0 / 22.0 | 0.5 / 2.0 / 0.5 | +| pick-fixed (keep looking) | 99.5 | 79.0 | 34.0 | 0.0 | 0.0 / 0.0 / 0.0 | 0.0 / 0.0 / 0.0 | +| pick-shaken (+ shake, noise, light) | 100.0 | 100.0 | 100.0 | 97.0 | 98.5 / 100.0 / 98.5 | 50.5 / 47.5 / 49.0 | + +Against the article's table (written before this run): the light columns match exactly, look-once's included, so the +cut corner did not change that row. The noise columns in the article (look-once 15 and 0, shaken 96.5 and 44.5) are +each one draw and sit below all three draws here. diff --git a/docs/research/head-camera/run-25/run25.test.ts.txt b/docs/research/head-camera/run-25/run25.test.ts.txt new file mode 100644 index 0000000..9d40ba6 --- /dev/null +++ b/docs/research/head-camera/run-25/run25.test.ts.txt @@ -0,0 +1,34 @@ +import { readFileSync, writeFileSync } from "node:fs"; +import { it } from "vitest"; +import { mulberry32 } from "@/lib/ml"; +import { NO_SHIFT } from "@/content/posts/head-camera/components/model"; +import { MAX_STEPS, PICK_HOME, PickPolicy, type PickState, decide, layout, pickAct, pickPicture, pickView } from "@/content/posts/head-camera/components/pick"; + +/** + * Run 25: the article's light-and-noise table for the three checkpoints the page ships, on the page's current layouts + * (block inside the cut BLOCK_AREA), 200 episodes a condition from layout() with mulberry32(99), camera untouched. + * Look-once's first picture gets the same light and noise as every later one, as the page does (pick-lab.tsx). + * The run-19 loop, with { noise, dim } passed to decide(). + */ +it("run 25", () => { + const conditions: [string, number, number][] = [["normal", 0, 1], ["light 80%", 0, 0.8], ["light 70%", 0, 0.7], ["light 50%", 0, 0.5], ["noise 0.05", 0.05, 1], ["noise 0.15", 0.15, 1]]; + const lines: string[] = []; + for (const name of ["pick-open", "pick-fixed", "pick-shaken"]) { + const policy = new PickPolicy(JSON.parse(readFileSync(`public/posts/head-camera/${name}.json`, "utf8"))); + const row: string[] = []; + for (const [label, noise, dim] of conditions) { + const rng = mulberry32(99); let wins = 0; + for (let e = 0; e < 200; e++) { + const s: PickState = { hand: [...PICK_HOME], closed: false, holding: false, ...layout(rng) }; + const belief = policy.kind === "open" ? PickPolicy.believed(policy.run(pickPicture(pickView(s, NO_SHIFT), noise, dim), []).out) : null; + for (let t = 0; t < MAX_STEPS; t++) { + const r = pickAct(s, decide(policy, s, NO_SHIFT, belief, { noise, dim }).action); + if (r === "placed") { wins++; break; } if (r === "dropped") break; + } + } + row.push(`${label} ${(wins / 2).toFixed(1)}`); + } + lines.push(`${name}: ${row.join(" | ")}`); + } + writeFileSync("docs/research/head-camera/run-25/output.txt", lines.join("\n") + "\n\n2026-09-22, Node via vitest. Noise is drawn from Math.random, so a noisy column moves by a point or two between runs.\n"); +}, 1_800_000); From 42ee26185a36192b3157e4907650578815b5bdaa Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 17:51:17 +0800 Subject: [PATCH 18/20] docs(head-camera): the failure table's noise columns as the range of three draws MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Noise comes from Math.random, and the article's single draws (look once 15 and 0, shaken 96.5 and 44.5) sit below all three draws of run 25 (docs/research/head-camera/run-25/). The columns now show 17–22 / 0–2 and 98.5–100 / 47.5–50.5, and the table says why. The light columns matched run 25 exactly and are unchanged. Co-Authored-By: Claude Opus 5 (1M context) --- content/posts/head-camera/en.mdx | 6 +++--- content/posts/head-camera/zh.mdx | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/content/posts/head-camera/en.mdx b/content/posts/head-camera/en.mdx index 94cad01..a8dddbe 100644 --- a/content/posts/head-camera/en.mdx +++ b/content/posts/head-camera/en.mdx @@ -112,13 +112,13 @@ The first extra thing in the recipe has to do with this. In training, besides sa ## How it breaks -The three models of fig. 01, given light and noise they never saw in training (200 layouts each). The sliders in fig. 01 do the same thing, and the small picture top right is exactly what the model is shown. +The three models of fig. 01, given light and noise they never saw in training (200 layouts each). The noise is random, so the noise columns are the range over three draws. The sliders in fig. 01 do the same thing, and the small picture top right is exactly what the model is shown. | | normal | light 80% | light 70% | light 50% | noise 0.05 | noise 0.15 | | --- | --- | --- | --- | --- | --- | --- | -| Look once | 71% | 72% | 58% | 11% | 15% | 0% | +| Look once | 71% | 72% | 58% | 11% | 17–22% | 0–2% | | Keep looking | 99.5% | 79% | 34% | 0% | 0% | 0% | -| + shaken, noise, light | 100% | 100% | 100% | 97% | 96.5% | 44.5% | +| + shaken, noise, light | 100% | 100% | 100% | 97% | 98.5–100% | 47.5–50.5% | The middle row is down to a third when the light drops by three tenths, and to 0% with a little noise. In its training no pixel's value ever changed, so it was free to lean on exact colours. It is the lesson of the fixed camera again: **what training never varies, the network will come to depend on.** So the second extra thing in the recipe is noise (up to 0.1) and brightness (60% to 120%) at random on every training picture. Inside that range the last row hardly notices; noise of 0.15, outside it, still breaks it. diff --git a/content/posts/head-camera/zh.mdx b/content/posts/head-camera/zh.mdx index 83c7524..e05693a 100644 --- a/content/posts/head-camera/zh.mdx +++ b/content/posts/head-camera/zh.mdx @@ -112,13 +112,13 @@ Figure 03 在手掌上加了相機,官方的理由是「主相機被擋住的 ## 它會怎麼壞 -圖 1 的三個模型,遇到訓練時沒見過的光線和雜訊(各 200 個擺法)。圖 1 的滑桿可以試同一件事,右上角的小畫面就是模型實際看到的。 +圖 1 的三個模型,遇到訓練時沒見過的光線和雜訊(各 200 個擺法)。雜訊是隨機的,雜訊欄是三次抽樣的範圍。圖 1 的滑桿可以試同一件事,右上角的小畫面就是模型實際看到的。 | | 正常 | 光 80% | 光 70% | 光 50% | 雜訊 0.05 | 雜訊 0.15 | | --- | --- | --- | --- | --- | --- | --- | -| 看一眼 | 71% | 72% | 58% | 11% | 15% | 0% | +| 看一眼 | 71% | 72% | 58% | 11% | 17–22% | 0–2% | | 持續看 | 99.5% | 79% | 34% | 0% | 0% | 0% | -| +晃相機、雜訊、亮度 | 100% | 100% | 100% | 97% | 96.5% | 44.5% | +| +晃相機、雜訊、亮度 | 100% | 100% | 100% | 97% | 98.5–100% | 47.5–50.5% | 中間那一列,光線暗三成就只剩三分之一,加一點點雜訊直接變成 0%。它訓練時每個像素的值從來沒變過,所以可以放心依賴精確的顏色。這和相機固定時是同一課:**訓練時沒變動過的東西,網路就會依賴它。** 所以配方裡多的第二樣東西,就是每張訓練畫面也隨機加雜訊(最多 0.1)、改亮度(60% 到 120%)。最後一列在這個範圍內幾乎不受影響,超出範圍的雜訊 0.15 還是會壞。 From 69543cb9c53b811458eee8908bc773273ec36ce4 Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 18:01:29 +0800 Subject: [PATCH 19/20] fix(site): thirteen visibility observers read the latest entry, not the first An IntersectionObserver batch is oldest first; under load it can carry "left" then "came back", and a callback that destructured [entry] decided the figure was off screen while it was visible, so it paused (the feat/vla session saw this as a one-in-forty e2e failure). Each callback now takes entries[entries.length - 1], as runWhenSeen already did: SLAM (2), the city, Flappy Bird, the scheduler, HydraNet, the Transformer, Lite3, diffusion (2), the path tracer's hook, the home page's number ticker and diffusion preview. tests/site/observers.test.ts keeps the pattern out. The whole e2e suite passed three times over (618 runs). Co-Authored-By: Claude Opus 5 (1M context) --- components/rt/use-tracer.ts | 2 +- components/site/diffusion-preview.tsx | 2 +- components/site/number-ticker.tsx | 2 +- .../ai-flappy-bird/components/flappy-lab.tsx | 2 +- .../city-of-agents/components/use-visible.ts | 2 +- .../diffusion-points/components/use-cloud.ts | 2 +- .../diffusion-points/components/use-fresh.ts | 2 +- .../hydranet-fruit/components/training-lab.tsx | 2 +- content/posts/lite3-walking/components/rig.ts | 2 +- content/posts/slam-2d/components/use-replay.ts | 2 +- content/posts/slam-2d/components/use-visible.ts | 2 +- .../task-scheduler/components/use-visible.ts | 2 +- .../components/training-lab.tsx | 2 +- tests/site/observers.test.ts | 16 ++++++++++++++++ 14 files changed, 29 insertions(+), 13 deletions(-) create mode 100644 tests/site/observers.test.ts diff --git a/components/rt/use-tracer.ts b/components/rt/use-tracer.ts index 678c3fa..dcaafbe 100644 --- a/components/rt/use-tracer.ts +++ b/components/rt/use-tracer.ts @@ -57,7 +57,7 @@ export function useTracer(root: RefObject, canvas: RefObject useEffect(() => { let alive = true, visible = false; - const io = new IntersectionObserver(([entry]) => (visible = entry.isIntersecting), { rootMargin: "200px" }); + const io = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => (visible = entry.isIntersecting), { rootMargin: "200px" }); if (root.current) io.observe(root.current); const worker = new Worker(new URL("./scene.worker.ts", import.meta.url), { type: "module" }); const frame = () => new Promise((resolve) => requestAnimationFrame(resolve)); diff --git a/components/site/diffusion-preview.tsx b/components/site/diffusion-preview.tsx index b87e7a0..3ab3b4f 100644 --- a/components/site/diffusion-preview.tsx +++ b/components/site/diffusion-preview.tsx @@ -109,7 +109,7 @@ export default function DiffusionPreview({ className = "block aspect-[8/5] w-ful if (want && !raf) { last = performance.now(); raf = requestAnimationFrame(frame); } else if (!want && raf) { cancelAnimationFrame(raf); raf = 0; } }; - const io = new IntersectionObserver(([e]) => { visible = e.isIntersecting; run(); }); + const io = new IntersectionObserver((entries, _observer, e = entries[entries.length - 1]) => { visible = e.isIntersecting; run(); }); io.observe(canvas); document.addEventListener("visibilitychange", run); draw(0, 0); diff --git a/components/site/number-ticker.tsx b/components/site/number-ticker.tsx index f6a4a14..bd6319e 100644 --- a/components/site/number-ticker.tsx +++ b/components/site/number-ticker.tsx @@ -20,7 +20,7 @@ export function NumberTicker({ value, pad = 0, duration = 900 }: { value: number setShown(Math.round(value * eased)); if (t < 1) raf = requestAnimationFrame(tick); }; - const seen = new IntersectionObserver(([entry]) => { + const seen = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => { if (!entry.isIntersecting) return; seen.disconnect(); setShown(0); diff --git a/content/posts/ai-flappy-bird/components/flappy-lab.tsx b/content/posts/ai-flappy-bird/components/flappy-lab.tsx index 10a2749..7a2101e 100644 --- a/content/posts/ai-flappy-bird/components/flappy-lab.tsx +++ b/content/posts/ai-flappy-bird/components/flappy-lab.tsx @@ -92,7 +92,7 @@ export function FlappyLab() { // Don't burn a core on an instrument nobody is looking at. Watch the whole instrument, // not the canvas: on a phone the controls and readouts sit a screen below it. - const io = new IntersectionObserver(([entry]) => (visible = entry.isIntersecting)); + const io = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => (visible = entry.isIntersecting)); io.observe(rootRef.current ?? canvas); paint(); if (playing) frame = requestAnimationFrame(loop); diff --git a/content/posts/city-of-agents/components/use-visible.ts b/content/posts/city-of-agents/components/use-visible.ts index 8d30ab5..2d2989e 100644 --- a/content/posts/city-of-agents/components/use-visible.ts +++ b/content/posts/city-of-agents/components/use-visible.ts @@ -7,7 +7,7 @@ export function useVisible(target: RefObject): RefObject { if (!target.current) return; - const io = new IntersectionObserver(([entry]) => { visible.current = entry.isIntersecting; }, { rootMargin: "100px" }); + const io = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => { visible.current = entry.isIntersecting; }, { rootMargin: "100px" }); io.observe(target.current); return () => io.disconnect(); }, [target]); diff --git a/content/posts/diffusion-points/components/use-cloud.ts b/content/posts/diffusion-points/components/use-cloud.ts index b6edb72..f9301bc 100644 --- a/content/posts/diffusion-points/components/use-cloud.ts +++ b/content/posts/diffusion-points/components/use-cloud.ts @@ -26,7 +26,7 @@ export function useCloud(canvas: RefObject, count: num latest.current.frame(view, dt); view.render(); }; - const seen = new IntersectionObserver(([entry]) => { + const seen = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => { visible = entry.isIntersecting; latest.current.onVisible?.(visible); }); diff --git a/content/posts/diffusion-points/components/use-fresh.ts b/content/posts/diffusion-points/components/use-fresh.ts index 7c6d7ff..d58300a 100644 --- a/content/posts/diffusion-points/components/use-fresh.ts +++ b/content/posts/diffusion-points/components/use-fresh.ts @@ -24,7 +24,7 @@ export function useFresh(target: RefObject, modelSteps: number, last.current = performance.now(); s.refresh(); }; - const seen = new IntersectionObserver(([entry]) => { visible.current = entry.isIntersecting; check(); }, { threshold: 0.2 }); + const seen = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => { visible.current = entry.isIntersecting; check(); }, { threshold: 0.2 }); if (target.current) seen.observe(target.current); const timer = window.setInterval(check, 1000); return () => { seen.disconnect(); window.clearInterval(timer); }; diff --git a/content/posts/hydranet-fruit/components/training-lab.tsx b/content/posts/hydranet-fruit/components/training-lab.tsx index 5a25aca..cf0d8bc 100644 --- a/content/posts/hydranet-fruit/components/training-lab.tsx +++ b/content/posts/hydranet-fruit/components/training-lab.tsx @@ -147,7 +147,7 @@ export function TrainingLab() { s.trainMs += performance.now() - t0; }; - const io = new IntersectionObserver(([entry]) => (visible = entry.isIntersecting)); + const io = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => (visible = entry.isIntersecting)); if (rootRef.current) io.observe(rootRef.current); frame = requestAnimationFrame(loop); return () => { diff --git a/content/posts/lite3-walking/components/rig.ts b/content/posts/lite3-walking/components/rig.ts index fbc2215..74b2e1d 100644 --- a/content/posts/lite3-walking/components/rig.ts +++ b/content/posts/lite3-walking/components/rig.ts @@ -124,7 +124,7 @@ function step(ms: number) { export function mountPanel(id: string, host: HTMLElement, knobs: Partial): () => void { const panel: Panel = { host, knobs, visible: 0 }; panels.set(id, panel); - const seen = new IntersectionObserver(([entry]) => { + const seen = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => { panel.visible = entry.intersectionRatio; if (state.status === "ready") activate(pick()); }, { threshold: [0, CLAIM_RATIO, 0.6, 0.9] }); diff --git a/content/posts/slam-2d/components/use-replay.ts b/content/posts/slam-2d/components/use-replay.ts index 84cf141..8985415 100644 --- a/content/posts/slam-2d/components/use-replay.ts +++ b/content/posts/slam-2d/components/use-replay.ts @@ -13,7 +13,7 @@ export function useReplay(near: RefObject, key: unknown, buil useEffect(() => { if (!near.current) return; - const io = new IntersectionObserver(([entry]) => { if (entry.isIntersecting) { setWanted(true); io.disconnect(); } }, { rootMargin: "1200px" }); + const io = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => { if (entry.isIntersecting) { setWanted(true); io.disconnect(); } }, { rootMargin: "1200px" }); io.observe(near.current); return () => io.disconnect(); }, [near]); diff --git a/content/posts/slam-2d/components/use-visible.ts b/content/posts/slam-2d/components/use-visible.ts index 8d30ab5..2d2989e 100644 --- a/content/posts/slam-2d/components/use-visible.ts +++ b/content/posts/slam-2d/components/use-visible.ts @@ -7,7 +7,7 @@ export function useVisible(target: RefObject): RefObject { if (!target.current) return; - const io = new IntersectionObserver(([entry]) => { visible.current = entry.isIntersecting; }, { rootMargin: "100px" }); + const io = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => { visible.current = entry.isIntersecting; }, { rootMargin: "100px" }); io.observe(target.current); return () => io.disconnect(); }, [target]); diff --git a/content/posts/task-scheduler/components/use-visible.ts b/content/posts/task-scheduler/components/use-visible.ts index 8d30ab5..2d2989e 100644 --- a/content/posts/task-scheduler/components/use-visible.ts +++ b/content/posts/task-scheduler/components/use-visible.ts @@ -7,7 +7,7 @@ export function useVisible(target: RefObject): RefObject { if (!target.current) return; - const io = new IntersectionObserver(([entry]) => { visible.current = entry.isIntersecting; }, { rootMargin: "100px" }); + const io = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => { visible.current = entry.isIntersecting; }, { rootMargin: "100px" }); io.observe(target.current); return () => io.disconnect(); }, [target]); diff --git a/content/posts/transformer-from-scratch/components/training-lab.tsx b/content/posts/transformer-from-scratch/components/training-lab.tsx index ce7d53d..9b4110b 100644 --- a/content/posts/transformer-from-scratch/components/training-lab.tsx +++ b/content/posts/transformer-from-scratch/components/training-lab.tsx @@ -109,7 +109,7 @@ export function TrainingLab() { if (++ticks % 4 === 0) paint(); }; - const io = new IntersectionObserver(([entry]) => (visible = entry.isIntersecting)); + const io = new IntersectionObserver((entries, _observer, entry = entries[entries.length - 1]) => (visible = entry.isIntersecting)); if (rootRef.current) io.observe(rootRef.current); const first = window.setTimeout(paint, 0); if (running) frame = requestAnimationFrame(loop); diff --git a/tests/site/observers.test.ts b/tests/site/observers.test.ts new file mode 100644 index 0000000..61ec534 --- /dev/null +++ b/tests/site/observers.test.ts @@ -0,0 +1,16 @@ +import { readdirSync, readFileSync, statSync } from "node:fs"; +import { join } from "node:path"; +import { expect, it } from "vitest"; + +/** + * An IntersectionObserver hands its callback a batch of entries, oldest first. Under load one batch can carry "left the + * screen" and then "came back"; a callback that reads only the first entry decides the figure is off screen while it is + * in plain sight, and the figure stops. (It made one e2e test fail about once in forty runs.) Read the last entry. + */ +it("no IntersectionObserver callback reads only the first entry of a batch", () => { + const files: string[] = []; + const walk = (dir: string) => { for (const name of readdirSync(dir)) { const p = join(dir, name); if (statSync(p).isDirectory()) walk(p); else if (/\.(ts|tsx)$/.test(name)) files.push(p); } }; + ["components", "content", "lib", "app"].forEach(walk); + const offenders = files.filter((f) => /IntersectionObserver\(\s*\(\s*\[/.test(readFileSync(f, "utf8"))); + expect(offenders).toEqual([]); +}); From 65e16ed5a792f0ccfad11e2654297b24b7f96266 Mon Sep 17 00:00:00 2001 From: PSheon Date: Tue, 22 Sep 2026 18:04:01 +0800 Subject: [PATCH 20/20] docs(handoff): bring the header up to PR #15, today's tests and branches Co-Authored-By: Claude Opus 5 (1M context) --- docs/HANDOFF.md | 21 ++++++++++++--------- 1 file changed, 12 insertions(+), 9 deletions(-) diff --git a/docs/HANDOFF.md b/docs/HANDOFF.md index cd23668..dd0ae2f 100644 --- a/docs/HANDOFF.md +++ b/docs/HANDOFF.md @@ -1,4 +1,4 @@ -# Handoff — main session, last revised 2026-09-20 +# Handoff — main session, last revised 2026-09-22 For whoever picks this up next. Read this, then the memory files under `~/.claude/projects/-Users-paul-jiang-Desktop-Paul/memory/` (they are loaded automatically, this file is not), then @@ -10,17 +10,17 @@ For whoever picks this up next. Read this, then the memory files under | | | | --- | --- | | Repo | `/Users/paul_jiang/Desktop/Paul/Blog`, GitHub `PSheon/Blog` (public) | -| Branches | `dev` is where work happens. `main` is production and moves only through a PR `dev` → `main` that Paul merges (last: PR #10, 2026-09-20) | +| Branches | `dev` is where work happens. `main` is production and moves only through a PR `dev` → `main` that Paul merges (last: PR #15, 2026-09-22) | | Production | (since 2026-09-21; DNS on Cloudflare, CNAME to Vercel, DNS only). Vercel project `paul-notebook`, deploys `main`. `paul-notebook.vercel.app` redirects 308 to it, path kept. Production env: `NEXT_PUBLIC_SITE_URL=https://blog.psheon.me` | | Dev server | `pnpm dev` on :3000 | -| Other worktree | `/Users/paul_jiang/Desktop/Paul/Blog-city`, branch `feat/city-of-agents` (PR #9 → `dev`), owned by another session. One writer per checkout | -| Tests | about 200 unit tests and 150 E2E runs (two projects: desktop, mobile), plus axe on every article. CI runs all of it on every push | +| Other worktrees | None since 2026-09-22: `feat/city-of-agents`, `feat/sche` and `feat/vla` are merged and deleted. One writer per checkout | +| Tests | 323 unit tests (2 skipped) and 206 E2E runs (two projects: desktop, mobile), plus axe on every article. CI runs all of it on every push | Published, in both languages: 001 CNN, 002 Flappy Bird, 003 trading agent, 004 Transformer, 005 HydraNet, 006 Lite3, 007 point-cloud diffusion, 008 2D SLAM, 009 city of agents, 010 a task scheduler from scratch (published 2026-09-21; -built on `feat/sche` by another session), 011 and 012 the light series (2026-09-22). On `dev`, published but not yet -released to `main`: 013 `head-camera` (merged from `feat/vla` on 2026-09-22; that branch and its worktree are gone). The earlier PCB-flip -VLA draft was dropped the same day, Paul found it dull; its simulation (arm, rasteriser, world) lives on as +built on `feat/sche` by another session), 011 and 012 the light series (PR #14), 013 `head-camera` (PR #15, merged from +`feat/vla`). All thirteen had their copy polished with Paul on 2026-09-22, item by item. The earlier PCB-flip +VLA draft (№ 014) was dropped the same day, Paul found it dull; its simulation (arm, rasteriser, world) lives on as `content/posts/head-camera/components/sim`, which 013 imports, with its tests in `tests/head-camera/`; its notes stay in `docs/research/pcb-flip-vla/`. A draft shows only in `next dev`, with a mark in the page's language ("草稿" / "DRAFT"). @@ -51,7 +51,6 @@ stays in `lib/rt` with its tests. Read before touching it: ## Waiting on Paul -- The two light articles lost `draft` on his word (2026-09-22) and go out in ONE `dev` → `main` PR that he merges. - Switch on Analytics and Speed Insights in the Vercel dashboard. Every performance number we have is simulated. - A test on a real phone. Nobody has done one. - Search Console (the verification env vars exist). The custom domain is done. Open: should `psheon.me` and `www.psheon.me` redirect to `blog.` instead of serving the site too. @@ -73,6 +72,8 @@ stays in `lib/rt` with its tests. Read before touching it: - Read the dev console once per page. - Numbers in articles must be measured and attributed (live / on my machine / offline with N seeds). Twice a claim I wrote from reasoning was wrong once measured. +- Polishing copy: list every change with the exact before and after text, numbered, and apply only what he picks. + Never touch an article's `date` or `updated`; the spread of dates is deliberate. ## Things that will bite you @@ -105,6 +106,8 @@ stays in `lib/rt` with its tests. Read before touching it: unit of work is much smaller than the budget. - Inside an `Instrument`, titles are `

`, not headings (axe `heading-order`). - Scrollable regions (tables, display maths) need `tabIndex={0}` + a name (axe). +- An `IntersectionObserver` callback gets a batch: read the LAST entry, not `([entry]) =>`, or a quick scroll + leaves a figure paused while visible. `tests/site/observers.test.ts` refuses the first-entry form. ## How a URL that does not exist is answered @@ -170,7 +173,7 @@ the per-article split. Numbers, method and what is left: `docs/research/2026-09- ## Other sessions -Sessions come and go; `ListAgents` shows who is there. Whoever owns the `Blog-city` worktree owns 009 and PR #9. Tell +Sessions come and go; `ListAgents` shows who is there. None owns a branch today. When one does, tell them when `dev` changes under them, with the SHA and the files likely to conflict. Peers cannot approve anything on Paul's behalf, and a force-push of their own branch is theirs to clear with Paul.