summaryrefslogtreecommitdiff
path: root/experiments/report.typ
blob: a15b1622e3cc0c794d143d041301bf74c0fa7c31 (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
#import "@preview/cetz:0.5.1"

#set document(title: "Where the ESP32-P4 editor's latency goes", author: "measured on ESP32-P4 rev v1.3")
#set page(margin: 2cm, numbering: "1")
#set text(font: ("Libertinus Serif", "DejaVu Serif"), size: 10.5pt)
#set par(justify: true)
#show raw: set text(font: "DejaVu Sans Mono", size: 9pt)
#show heading: it => block(above: 1.4em, below: 0.7em, it)

#align(center)[
  #text(17pt, weight: "bold")[Where the ESP32-P4 editor's latency goes]

  #v(0.2em)
  #text(10pt)[A measured account, on silicon, of the `pardes` editor running as
  ESP32-P4 firmware with one 115200-baud serial line as its only I/O]
]

#v(0.5em)

#let capacity = 11520.0

// ---------------------------------------------------------------- data plumbing
// The figures read the raw per-trial CSV the instrument emits. Nothing here is a
// transcribed number: if a run is repeated, the document changes with it.
#let rows(file) = {
  let out = ()
  for r in csv(file) {
    if r.at(0) == "label" { continue }
    out.push((
      label: r.at(0), cols: int(r.at(2)), rows: int(r.at(3)),
      length: int(r.at(4)), op: r.at(5),
      rtt: int(r.at(7)) / 1000.0, settle: int(r.at(8)) / 1000.0, bytes: int(r.at(9)),
    ))
  }
  out
}

#let median(xs) = {
  let s = xs.sorted()
  if s.len() == 0 { return 0.0 }
  s.at(int(s.len() / 2))
}

// One expression on one line: a continuation beginning with `+` is list markup to Typst, and it
// rendered the file names as a numbered item on page one.
#let length_rows = (
  rows("length-ReleaseSmall.csv") + rows("length-ReleaseFast.csv") +
    rows("length-ReleaseSmall-lineSpan.csv") + rows("length-ReleaseFast-lineSpan.csv") +
    rows("length-RFast-grapheme.csv") + rows("length-RFast-print.csv") +
    rows("length-RFast-shadow.csv") + rows("length-RFast-fastcmp.csv") +
    rows("length-CPU360-final.csv") + rows("length-DirectEmit.csv")
)
#let ops_rows = rows("ops-ReleaseSmall.csv") + rows("ops-ReleaseFast.csv")

// Figure 1 compares the two optimisation modes only; the `lineSpan` variants are the SAME source
// change applied to each, and are tabulated separately in Experiment 3 rather than plotted, because
// four indistinguishable pairs of lines would be a worse picture than two.
#let builds = ("ReleaseSmall", "ReleaseFast")
#let all_builds = ("ReleaseSmall", "ReleaseSmall-lineSpan", "ReleaseFast", "ReleaseFast-lineSpan")
#let lengths = (0, 20, 40, 80, 160)

#let med_rtt(rs, pred) = median(rs.filter(pred).map(r => r.rtt))
#let n_of(rs, pred) = rs.filter(pred).len()

// Least squares, for the slope that is the whole point of figure 2.
#let fit(xs, ys) = {
  let n = xs.len()
  let mx = xs.sum() / n
  let my = ys.sum() / n
  let num = 0.0
  let den = 0.0
  for i in range(n) {
    num += (xs.at(i) - mx) * (ys.at(i) - my)
    den += (xs.at(i) - mx) * (xs.at(i) - mx)
  }
  let slope = num / den
  (slope: slope, intercept: my - slope * mx)
}

// ---------------------------------------------------------------- summary
= What was found

The editor is *not* limited by its serial line. Measured against a checksum-verified
protocol, the link carries #calc.round(11496 / capacity * 100, digits: 1)% of its theoretical
capacity in both directions with zero corruption, and typing at 100 characters a
second loses nothing and uses under a tenth of the wire.

What limits it is *computation per input event*, and that cost has two parts, both
measured here:

#block(inset: (left: 1em))[
  *A fixed cost of ≈#calc.round(med_rtt(ops_rows, r => r.label == "ReleaseSmall" and r.op == "motion_h"), digits: 1) ms per event*, which does not
  depend on how much the screen changed. A cursor motion emitting 40 bytes and an
  insert-and-escape emitting 206 bytes cost the same round trip to within 0.3 ms.

  *A cost proportional to the document*, at
  #calc.round(fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope * 1000, digits: 1) µs
  per character already in the line, per keystroke, which is what makes the editor
  feel worse the more you have written.
]

*Both live entirely in the renderer.* Timed separately on the die, parsing the
keystroke and applying the edit takes a flat ≈220 µs regardless of document size —
1.5% of the total — while `render` carries the whole ≈15 ms floor and every
microsecond of the slope. That result contradicted the mechanism the source reading
implied, and it was only reachable by instrumenting the firmware; the fix that
source reading suggested was written, measured, and found to be worth 20% on a 19 MB
file and nothing at all on this board. Experiment 3.

Both were then cut. A keystroke is *3.74 ms, from 16.99*, and 4.06 ms at a
160-character line, from 25.56 — the per-character term is 2.0 µs, from 54.3. Three of
the six steps that did it came from *not asking Unicode about ASCII*; the largest single
one came from this repository's own `present` copying all 480 cells into vaxis every
frame whether or not any had changed; and the last two were a clock that was only ever a
divider away, and a renderer that stopped asking vaxis to recompute a diff `present` had
already done. Experiment 4.

The 4 ms target is met. Two findings on the way there are worth more than the number.
A 4× clock bought 2.6× because the grid walk is bounded by memory bandwidth, not by the
core. And the direct renderer, with less computation and a quarter of the bytes, first
measured *slower* — because the CH340 bridging the board to the host forwards a bulk
packet only when it is full, so a reply too small to fill one waits about a millisecond
for a timer. The frame has a minimum size and it belongs to the transport.

= The instrument

Two programs and a protocol, all in this repository.

`tools/perfproto.zig` is a framed protocol shared *verbatim* by the host tool and
the firmware, so a frame written by one and parsed by the other cannot drift:
a nine-byte header (`"P4"`, op, length, CRC-32 of the payload) then the payload.
`examples/uartperf.zig` answers it on the board; `tools/bench_main.zig` drives it
from the host.

The checksum is the point. RX overrun on this UART is undetected in hardware and
uncounted in the driver, so a byte that never arrives is indistinguishable from a
byte that arrived late. A frame with a length and a CRC turns both into facts: the
board reports a CRC over exactly the bytes it received, the host compares it
against a CRC over exactly the bytes it sent, and a throughput figure that is not
checksummed is only a guess about how fast data was corrupted.

`tools/rtt.zig` provides the two timing functions everything else is built on:
`roundTrip`, which sends a stimulus and returns when the first response byte
arrives, and `measure`, which repeats it. *Round trip is time to the first
response byte, not to the last.* A renderer that begins drawing in 8 ms and
finishes in 130 ms feels immediate; one that thinks for 130 ms and then draws in
8 ms feels broken; measuring only when the wire falls quiet cannot tell them
apart. The time to the last byte is recorded separately as `settle`.

Microseconds throughout: at 115200 baud one byte occupies 87 µs, so a millisecond
clock would quantise these measurements into buckets eleven bytes wide.

= Method

All numbers are from one ESP32-P4 rev v1.3 over its CH340 bridge at 115200 baud
8N1, giving #capacity B/s in each direction. Each condition is measured
#n_of(length_rows, r => r.label == "ReleaseSmall" and r.length == 0) times and
reported as a median; the raw per-trial rows are in `experiments/*.csv` and this
document computes its figures from them directly.

#block(breakable: false)[
```
zig build flash -Dapp=examples/uartperf.zig     # the link ceiling
zig-out/bin/p4-bench --link

zig build flash -Dpardes                        # the editor
zig-out/bin/p4-bench --sweep length --repeat 7 --csv --label ReleaseSmall
zig-out/bin/p4-bench --sweep ops    --repeat 7 --csv --label ReleaseSmall
```
]

The editor is modal, so every editor run enters insert mode once before timing and
uses a single inserted character as the comparable unit of work. In the length
experiment the line is primed to exactly $n$ characters *without* measuring, so the
timed keystroke always sees a document of known size.

= Baseline: the link is not the problem

With only the protocol responder running — no editor — the link performs as well as
it can:

#figure(
  table(
    columns: (auto, auto, auto, auto),
    align: (left, right, right, left),
    stroke: none,
    table.hline(),
    table.header([direction], [measured], [of capacity], [integrity]),
    table.hline(stroke: 0.5pt),
    [uplink, host → board], [11 496 B/s], [99%], [CRC verified, 32 768 B],
    [downlink, board → host], [11 413 B/s], [99%], [CRC verified, 32 768 B],
    [round trip, 13 B each way], [4.2–4.5 ms], [--], [0 lost of 20],
    table.hline(),
  ),
  caption: [The UART driver and the wire, with nothing else running. Every byte
  accounted for by checksum.],
)

Of that 4.2 ms round trip, 2.26 ms is the wire itself (26 bytes at #capacity B/s);
the remaining ≈2 ms is the host, the USB bridge and the firmware's parse. *This
≈2 ms is the floor every editor measurement below sits on*, and subtracting it is
how the editor's own share is obtained.

= Hypotheses

#figure(
  table(
    columns: (auto, 1fr, auto),
    align: (left, left, left),
    stroke: none,
    table.hline(),
    table.header([], [hypothesis], [verdict]),
    table.hline(stroke: 0.5pt),
    [H1], [Latency is compute-bound, not transmission-bound: round trip is
      independent of how many bytes the operation emits.], [*confirmed*],
    [H2], [The editor object's optimisation mode materially changes latency.], [*confirmed*],
    [H3], [An edit is $O(n)$ in the document: round trip rises linearly with the
      characters already in the line.], [*confirmed*],
    [H4], [Cost is paid per screen cell, so a smaller grid is proportionally
      cheaper.], [*not supported*],
    [H5], [Typing at a human rate loses input.], [*refuted*],
    table.hline(),
  ),
  caption: [Stated before measuring; each is settled by one experiment below.],
)

= Experiment 1 --- output size does not predict latency (H1)

Seven operations, chosen to span a five-fold range of emitted bytes at a fixed
40×12 geometry.

#figure(
  {
    let names = ("motion_h", "motion_l", "line_start", "line_end", "insert_esc")
    table(
      columns: (auto, auto, auto, auto, auto),
      align: (left, right, right, right, right),
      stroke: none,
      table.hline(),
      table.header([operation], [bytes], [wire time], [round trip], [implied compute]),
      table.hline(stroke: 0.5pt),
      ..names.map(n => {
        let rs = ops_rows.filter(r => r.label == "ReleaseSmall" and r.op == n)
        let b = median(rs.map(r => r.bytes))
        let t = median(rs.map(r => r.rtt))
        (
          raw(n),
          [#b B],
          [#calc.round(b / capacity * 1000, digits: 2) ms],
          [#calc.round(t, digits: 2) ms],
          [#calc.round(t - 2.0, digits: 2) ms],
        )
      }).flatten(),
      table.hline(),
    )
  },
  caption: [`ReleaseSmall`, 40×12, median of 7. Emitted bytes vary 5×; the round
  trip does not vary at all. "Implied compute" subtracts only the ≈2 ms link floor,
  because round trip is measured to the *first* byte and so does not contain the
  transmission of the rest.],
)

A five-fold change in output moves the round trip by less than 2%. Whatever the
editor is doing for ≈15 ms, it is doing before it emits anything, and it is not
proportional to what changed on screen. *H1 is confirmed*, and it is the reason
raising the baud rate cannot fix typing latency: there is almost no wire in it.

= Experiment 2 --- an edit costs the whole document (H3, H2)

The controlled variable is the number of characters already in the line. The
measured quantity is unchanged: the round trip of one further inserted character.

#figure(
  cetz.canvas(length: 1cm, {
    import cetz.draw: *
    let w = 11.0
    let h = 6.0
    let xmax = 176.0
    let ymax = 28.0
    let px(v) = v / xmax * w
    let py(v) = v / ymax * h

    // axes
    line((0, 0), (w + 0.3, 0), mark: (end: "straight"), stroke: 0.6pt)
    line((0, 0), (0, h + 0.3), mark: (end: "straight"), stroke: 0.6pt)
    content((w / 2, -0.85), text(9pt)[characters already in the line])
    content((-1.15, h / 2), angle: 90deg, text(9pt)[round trip (ms)])

    for l in lengths {
      line((px(l), 0), (px(l), -0.12), stroke: 0.6pt)
      content((px(l), -0.38), text(8pt)[#l])
    }
    for v in (0, 5, 10, 15, 20, 25) {
      line((0, py(v)), (-0.12, py(v)), stroke: 0.6pt)
      content((-0.42, py(v)), text(8pt)[#v])
      if v > 0 { line((0, py(v)), (w, py(v)), stroke: (paint: luma(88%), thickness: 0.4pt)) }
    }

    let colours = (ReleaseSmall: rgb("#b3261e"), ReleaseFast: rgb("#1a5fb4"))
    for b in builds {
      let ys = lengths.map(l => med_rtt(length_rows, r => r.label == b and r.length == l))
      let f = fit(lengths.map(l => l * 1.0), ys)
      // fitted line, drawn under the data so the points remain readable
      line(
        (px(0), py(f.intercept)),
        (px(xmax), py(f.intercept + f.slope * xmax)),
        stroke: (paint: colours.at(b).lighten(55%), thickness: 1.6pt),
      )
      // every individual trial, so the spread is visible rather than asserted
      for r in length_rows.filter(r => r.label == b) {
        circle((px(r.length * 1.0), py(r.rtt)), radius: 0.045, fill: colours.at(b).lighten(30%), stroke: none)
      }
      line(..lengths.zip(ys).map(p => (px(p.at(0) * 1.0), py(p.at(1)))), stroke: (paint: colours.at(b), thickness: 1.1pt))
      for p in lengths.zip(ys) {
        circle((px(p.at(0) * 1.0), py(p.at(1))), radius: 0.075, fill: colours.at(b), stroke: none)
      }
      content(
        (px(xmax) + 0.15, py(f.intercept + f.slope * xmax)),
        anchor: "west",
        text(8pt, fill: colours.at(b))[#b],
      )
    }
  }),
  caption: [Round trip against document length, every trial plotted (7 per point),
  medians joined, least-squares fit behind. Both builds are straight lines: an edit
  is $O(n)$ in the document.],
)

#figure(
  table(
    columns: (auto, auto, auto, auto),
    align: (left, right, right, right),
    stroke: none,
    table.hline(),
    table.header([build], [fixed cost], [per character], [round trip at 160 chars]),
    table.hline(stroke: 0.5pt),
    ..builds.map(b => {
      let ys = lengths.map(l => med_rtt(length_rows, r => r.label == b and r.length == l))
      let f = fit(lengths.map(l => l * 1.0), ys)
      (
        raw(b),
        [#calc.round(f.intercept, digits: 2) ms],
        [#calc.round(f.slope * 1000, digits: 1) µs],
        [#calc.round(ys.last(), digits: 2) ms],
      )
    }).flatten(),
    table.hline(),
  ),
  caption: [Fitted from the medians. The slope is the interesting column. Its
  *mechanism* is settled by Experiment 3, not by this fit.],
)

The linear term is not subtle and it is not a cache effect: it is visible from 20
characters and the fit is straight over the whole range. At
#calc.round(fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope * 1000, digits: 0) µs
per character on a ≈90 MHz core — roughly 5 000 cycles for every character already
in the line — it is far too much work to be a `memcpy`, so something is making a
substantial pass per character. *H3 is confirmed as an observation.* What that pass
actually is turned out not to be what the source reading suggested, which is
Experiment 3.

The two series also settle H2. `ReleaseFast` lowers the fixed cost by
#calc.round(
  (1 - fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseFast" and r.length == l))).intercept /
     fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).intercept) * 100,
  digits: 0)% and the per-character cost by
#calc.round(
  (1 - fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseFast" and r.length == l))).slope /
     fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope) * 100,
  digits: 0)%, so its advantage *grows with the document*: 13% at an empty line and
21% at 160 characters. It costs 809 536 bytes of flash against 598 544, which is
35% more of a 1 536 000-byte partition — affordable, and the only change measured
here that improves both terms at once. *H2 is confirmed.*

= Experiment 3 --- it is all in the renderer, and the obvious fix was wrong

Everything above measures a keystroke from the host, which cannot see *what* the
firmware spent the time on. Reading the source suggested an answer: the edit path
builds each new document with `modal.spliceAlloc`, a fresh allocation and a copy of
the whole buffer, and `insertAt` called `modal.lineCount` — `std.mem.count` over
every byte — *twice*, merely to clamp a row. That is three whole-document passes
before a single character can be inserted, which fits a linear slope exactly.

It is also, on this board, almost entirely irrelevant. Timing the two phases
separately on the die (`-Dprof`, two reads of the cycle counter around
`pardes_p4_input` and `pardes_p4_render`) gives:

#figure(
  {
    let a = ()
    for r in csv("attribution.csv") {
      if r.at(0) == "label" { continue }
      a.push((chars: int(r.at(2)), input: int(r.at(5)), render: int(r.at(6))))
    }
    table(
      columns: (auto, auto, auto, auto),
      align: (right, right, right, right),
      stroke: none,
      table.hline(),
      table.header([characters in line], [input: parse + edit], [render], [render share]),
      table.hline(stroke: 0.5pt),
      ..a.map(r => (
        [#r.chars],
        [#r.input µs],
        [*#r.render µs*],
        [#calc.round(r.render / (r.input + r.render) * 100, digits: 1)%],
      )).flatten(),
      table.hline(),
    )
  },
  caption: [On-board cycle counts, `ReleaseSmall`. Input is flat; render carries
  both the fixed cost and the entire slope.],
)

*Input is flat at ≈220 µs and does not grow with the document at all.* The fixed
≈15 ms and every microsecond of the per-character slope are inside
`pardes_p4_render`. The edit path — the allocation, the copy, the double line count
— is 1.5% of a keystroke and could be made free without anyone noticing.

This was worth proving rather than assuming, because the fix implied by the source
reading was written and measured. `modal.insertAt` now takes one *bounded* scan
through a new `modal.lineSpan`, which stops at the row it wants instead of counting
the whole document, and pays for a full count only on the rare clamping path where
the cursor is past the end. On the host harness (`zig build perf`, which drives the
same core over 61 KB to 19 MB fixtures) that is a real win, reproduced over three
independent runs:

#figure(
  table(
    columns: (auto, auto, auto, auto),
    align: (left, right, right, right),
    stroke: none,
    table.hline(),
    table.header([fixture], [1 000 lines], [50 000 lines], [300 000 lines]),
    table.hline(stroke: 0.5pt),
    [`edit-char`, before], [730 µs], [2 996 µs], [15 030 µs],
    [`edit-char`, after], [721 µs], [2 414 µs], [10 955 µs],
    [ratio, three runs], [0.98×], [0.79–0.82×], [0.78–0.83×],
    table.hline(),
  ),
  caption: [Host harness, 25 samples per cell. Every untouched operation stayed at
  1.00×, which is stronger evidence than any single cell.],
)

And on the board it changed *nothing*, in either optimisation mode. All four
combinations were measured on the die, 5 lengths × 7 trials each:

#figure(
  table(
    columns: (auto, auto, auto, auto, auto),
    align: (left, right, right, right, right),
    stroke: none,
    table.hline(),
    table.header([configuration], [fixed cost], [per character], [at 160 chars], [vs baseline]),
    table.hline(stroke: 0.5pt),
    ..all_builds.map(b => {
      let xs = lengths.map(l => l * 1.0)
      let ys = lengths.map(l => med_rtt(length_rows, r => r.label == b and r.length == l))
      let f = fit(xs, ys)
      let at160 = ys.last()
      let ref160 = med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == 160)
      (
        raw(b),
        [#calc.round(f.intercept, digits: 2) ms],
        [#calc.round(f.slope * 1000, digits: 1) µs],
        [#calc.round(at160, digits: 2) ms],
        [#calc.round(at160 / ref160, digits: 2)×],
      )
    }).flatten(),
    table.hline(),
  ),
  caption: [The edit-path change is invisible in both modes; the optimisation mode
  is the whole of the difference. `ReleaseFast` + `lineSpan` is indistinguishable
  from `ReleaseFast` alone.],
)

That is not a contradiction, it is the same fact seen twice: the removed passes are
$O(#h(0.1em)$document$)$, and this board's document is a few hundred *bytes*, so two
scans of it cost nothing worth measuring. The identical change is worth 20% on a
19 MB file and 0% on a 240-character one.

The measurement did change one thing about the board, though, and it is not the
source: `-Doptimize` defaulted to `Debug`, so a plain `zig build -Dplatform=p4`
produced an object that *cannot run* — `Debug` wraps every tier in `allocators.zig`
in a `DebugAllocator` whose metadata is page-granular, and one 4 KiB page per size
class does not fit in the 384 KiB the board hands over. The p4 target now defaults to
`ReleaseFast`, which is the mode this experiment chose rather than a preference, and
an explicit `-Doptimize=` still wins. The 21% is therefore what the default build now
gives, not something to remember to ask for.

The lesson is the one the instrument exists to enforce. A plausible mechanism, read
off the source and consistent with the shape of the data, was wrong about where the
time went — and it took a measurement *inside* the firmware to say so. The renderer
is the target; the next question is what in it is proportional to the line, and the
position experiment already narrows that: cost follows the cursor's column as well
as the document's size, which is the signature of a walk from the start of a line.

#figure(
  table(
    columns: (auto, auto, auto),
    align: (left, right, right),
    stroke: none,
    table.hline(),
    table.header([insert position in a fixed 320-character line], [round trip], [bytes emitted]),
    table.hline(stroke: 0.5pt),
    [column 320 (end)], [33.9 ms], [28 B],
    [column 0 (start)], [26.0 ms], [81 B],
    [column 320 again], [33.7 ms], [28 B],
    table.hline(),
  ),
  caption: [Same document throughout; only the cursor moved. The end of the line
  costs 7.8 ms more than the start while emitting *a third* as many bytes — output
  size and latency are not merely uncorrelated here, they are inverted.],
)

= What the measurements rule out

*Input is not being lost while typing (H5, refuted).* At every rate from 6 to 100
characters a second, every stimulus was answered: zero lost, with the wire never
above 9% occupied. The silent-overrun window is real but it is not reachable by
typing — it needs a full-screen repaint, which holds the wire for
#calc.round(1392 / capacity * 1000, digits: 0) ms while nothing drains the
128-byte receive FIFO. Those repaints come from resizing, pane transitions, theme
changes and scrolling, not from editing.

*Screen area does not explain the fixed cost (H4, not supported).* Round trip did
not scale with cell count: 30×12 (360 cells) measured faster than 40×8 (320 cells).
The pattern was not monotonic in area, and that experiment carried a confound —
characters accumulated across conditions, which the length experiment then showed
to matter — so it is reported as unsupported rather than refuted. Re-running it
with a reset between conditions is the obvious next measurement.

= Two levers that were researched rather than measured

*Raising the line rate.* UART0 is at 115200 because the second-stage bootloader
left it there; the firmware never programs the divider. Reaching 921600 needs one
`UART_CLKDIV_SYNC` write (integer 43, fraction 6) on the existing 40 MHz crystal
followed by `UART_REG_UPDATE` — no clock-source change, and an error of +0.064%,
far inside a UART's tolerance. 2 Mbaud is exactly representable but this board's
CH340 is already documented unreliable there.

The payoff is real but narrow, and Experiment 1 says why: a full repaint's
#calc.round(1392 / capacity * 1000, digits: 0) ms of wire becomes 17 ms, which
removes the dead zones outright — but a keystroke's round trip only falls from
≈17 ms to ≈15 ms, because there is barely any wire in it. *Raise the baud to fix
repaints, not to fix typing.*

*The second core.* The chip has two rv32imafc cores at ≈90 MHz plus a 16 MHz
low-power core. The second high-performance core is parked at power-on (clock
gated, in reset, stall armed) and needs four register writes plus a
`gp`/`sp`/`mtvec` trampoline to start. Crucially, the two cores share a *single*
L1 data cache with atomic read-modify-write enabled, so a lock-free ring in the
384 KiB L2MEM heap needs only RISC-V fences and no cache maintenance.

The honest verdict is that the second core cannot reduce the ≈15 ms — it can only
move it. Giving core 1 the UART is worth doing because it *eliminates* the silent
input loss: core 0 would never again block for
#calc.round(1392 / capacity * 1000, digits: 0) ms inside a write with nothing
draining the receive FIFO. Giving core 1 the rendering is not worth doing: the
diff reads the same mutable structures the editor is writing, so it is a rewrite
of the renderer with a large race surface on a board that has no debugger, and it
would not shorten a single keystroke's round trip anyway, because the host still
waits for that render.

= Experiment 4 --- cutting it, and what each cut was worth

The target set after Experiment 3 was 4 ms. Round trip is time to the first response
byte, so it is ≈2 ms of host and USB latency plus computation; 4 ms therefore means a
compute budget of about 2 ms, and raising the line rate cannot help — at 115200 an
81-byte reply is 7 ms of wire but almost none of it lands before the first byte.

Every step below was profiled first, in the board's *exact* configuration: 40×12 and
`-Dtree-sitter=disabled`, because the P4 build has no tree-sitter and a profile that
includes it is a profile of a different program. Half of the first profile was
tree-sitter, which the board never runs.

#figure(
  table(
    columns: (auto, auto, auto, auto, auto),
    align: (left, right, right, right, right),
    stroke: none,
    table.hline(),
    table.header([step], [fixed cost], [per char], [at 160 chars], [vs start]),
    table.hline(stroke: 0.5pt),
    ..(
      ("ReleaseSmall", "ReleaseFast", "RFast-grapheme", "RFast-print", "RFast-shadow",
        "RFast-fastcmp", "CPU360-final", "DirectEmit")
    ).map(b => {
      let xs = lengths.map(l => l * 1.0)
      let ys = lengths.map(l => med_rtt(length_rows, r => r.label == b and r.length == l))
      let f = fit(xs, ys)
      let at160 = ys.last()
      let ref160 = med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == 160)
      (
        raw(b),
        [#calc.round(f.intercept, digits: 2) ms],
        [#calc.round(f.slope * 1000, digits: 1) µs],
        [#calc.round(at160, digits: 2) ms],
        [#calc.round(at160 / ref160, digits: 2)×],
      )
    }).flatten(),
    table.hline(),
  ),
  caption: [Each row is a separate firmware measured on the die, 5 document lengths ×
  7--9 trials. The per-character column and the fixed column move for different reasons
  and are worth reading separately. The last two rows are the clock raise and the
  renderer, and neither is a source optimisation in the sense the four above it are.],
)

*Not asking Unicode about ASCII* accounts for the per-character column. Three fast
paths, each guarded so non-ASCII text takes exactly the road it took before:
`modal.graphemeStart` (21.5% of a keystroke, and the reason cost followed the cursor's
column — it walks graphemes from the start of the text with the full UAX #29 state
machine), `Surface.print` (26.2%, which per character took a UTF-8 length, a decode, a
*freshly constructed* grapheme iterator, a validation and a width lookup to conclude
that `y` is one cell), and `graphemeDisplayWidth` (6.9%, all of it asking a Unicode
table about ASCII). The guard is the same in each: an ASCII scalar is its own grapheme
cluster unless it is a CR before an LF, because every rule that could join it —
Extend, ZWJ, SpacingMark, Prepend, Regional_Indicator — is spelled with non-ASCII
scalars. Per character: 54.3 → 6.9 µs.

*The fixed cost needed the frame broken open.* A second render with nothing changed
cost the same as the first, so the ~15 ms was unconditional; `pardes_p4_frame_prof`
then reported the three stages separately.

#figure(
  table(
    columns: (auto, auto, auto),
    align: (left, right, right),
    stroke: none,
    table.hline(),
    table.header([stage of one frame], [before], [after]),
    table.hline(stroke: 0.5pt),
    [copy the Surface into vaxis's grid], [6 750 µs], [*979 µs*],
    [vaxis diffs its grid and emits], [2 460 µs], [2 455 µs],
    [pardes rebuilds the whole Surface], [≈2 600 µs], [≈2 000 µs],
    [push the bytes into the UART], [1 µs], [1 µs],
    table.hline(),
  ),
  caption: [Cycle counts read on the die. The largest item was in *this repository's*
  own `present`, not in pardes and not in vaxis: it copied all 480 cells every frame,
  whether or not any had changed.],
)

So `present` now keeps the previous Surface and tells vaxis only what moved, comparing
cells as bytes rather than through `std.meta.eql` on a colour union and eight booleans.
The grid lives in `.bss`, and that is a bug fix rather than a flourish: allocated from
the editor's heap it was enough to make `vx.resize` fail.

== How a rendering change was made safe to believe

A latency benchmark cannot see a corrupted screen, and `src/p4.zig` compiles for this
board alone — no host suite covers it. So each optimisation carries a comptime switch
whose `false` arm is the previous behaviour — `shadow_grid` for the incremental walk,
`direct_emit` for the renderer — and the firmwares were each run against the same
18-step workload on the die: inserts, deletes, motions that move the modified-marker, a
line outgrowing the viewport, backspaces that shrink it. The screens were reconstructed
from the wire and compared.

That comparison had to be corrected twice, and the second time it had already produced a
false result. Tracking characters only would have passed a colour regression in silence,
so it began hashing SGR as well — but it hashed the escape *parameters* applied to each
cell, which is a history and not a state. Raising the clock changed which keystrokes
shared a frame, the same final colours were reached by a different route, and the
verifier *reported a difference on the die that did not exist*. It now decodes SGR into
resolved state — foreground, background, attribute set per cell — and two screens compare
equal when every cell holds the same character and the same resolved style, whatever
sequence of escapes the emitter chose to get there. Without that correction the
from-scratch renderer could not have been verified at all: it reaches identical screens
by deliberately different escapes.

All three pairs are identical under it: shadow grid against full repaint, 90 MHz against
360, and direct emission against vaxis. On the host side, where the shared code does
live: `snap` 95/95 scripts, `hxdiff` 481 cases with 0 mismatches, `hxparity` 561 cases
with 0 mismatches.

== The clock was a divider, not a design

The board had been executing at 90 MHz for every measurement above, because the stock
second-stage bootloader is built for it. The CPLL was *already* running at 360 MHz — 90
is exactly a quarter of it — so reaching 360 is four divider writes and a ROM call, with
no PLL to enable and no voltage step to sequence against. Verified against the systimer,
which is clocked from the crystal and therefore independent: 90 000 kHz before,
359 991 kHz after. Nothing else moves, and that is what makes it safe from a live
console: the UART0 baud clock comes from the crystal, the systimer from the crystal, and
the flash interface from a different PLL entirely.

Compute fell 6.42 → 2.44 ms: *2.6× for a 4× clock*, and the shortfall is the
interesting part. The grid walk reads 27 KB per frame and takes 226 µs at 360 MHz, about
six cycles a byte, so it is bounded by L2MEM bandwidth rather than by the core. A
prediction made before the measurement, and held.

== The renderer, and a minimum frame that belongs to the USB bridge

vaxis's diff was redundant: `present` has already worked out exactly which cells moved,
so vaxis was being told the answer and then computing it again against its own copy.
Emitting the escapes directly — one absolute cursor position per run of changed cells,
absolute SGR rather than a delta, a hand-rolled formatter for the two digits that
sequence needs — took the board's render from 1 362 µs to 779 µs and the reply from 81
bytes to 21.

And it measured *slower*: 4.72 ms against 4.37. Fewer bytes, less computation, worse
round trip is the shape of a wrong model, so the search moved off the firmware.

It is the USB bridge. The board reaches the host through a CH340, a full-speed part
whose bulk IN endpoint carries 32-byte packets, and it forwards a packet when the packet
is *full*. A 21-byte frame never fills one, so it waits in the bridge for an internal
timer to give up on more — worth about a millisecond, a quarter of the entire budget.
Routing through vaxis had only looked competitive because 81-byte frames fill a packet
by accident.

#figure(
  table(
    columns: (auto, auto, auto, 1fr),
    align: (right, right, right, left),
    stroke: none,
    table.hline(),
    table.header([frame], [min], [median], []),
    table.hline(stroke: 0.5pt),
    [21 B], [3 843 µs], [4 817 µs], [direct, never fills a packet],
    [49 B], [3 719 µs], [3 814 µs], [direct, padded past the boundary],
    [81 B], [4 373 µs], [4 475 µs], [through vaxis, fills one by accident],
    table.hline(),
  ),
  caption: [Identical board cost and a byte-identical screen across all three; only the
  size of the reply differs. Note the *minimum*: the 21-byte frame's floor is already
  530 µs below vaxis's, which is exactly the computation that was saved. Only the median
  was hostage to the bridge's timer.],
)

So the frame has a minimum size and it is the transport's, not the terminal's. The
emitter pays it, padding with repeated absolute cursor positioning: idempotent, already
the sequence a frame ends on, incapable of altering a cell. This is the bargain an
Ethernet runt frame makes — the medium has a minimum and the sender pays it — and it is
a real trade rather than a free one, since the filler is wire time that delays a later
frame. It applies only while the frame is small, which is when there is wire to spare.

A second instrument came out of this. The original bench sends keystrokes on a fixed
cadence, which locks the send phase to the host's 1 ms USB frame clock and makes the
round trip a staircase in board time — where a real saving can present as a regression.
Sleeping a uniform random 0–2 ms before each keystroke decorrelates them. That was not
what was happening here, but it had to be ruled out before the CH340 could be believed.

== Where it stopped

A keystroke is *3.74 ms*, from 16.99 — and 4.06 ms at a 160-character line, from 25.56.
The target was 4 ms and it is met, with the phase-randomised instrument agreeing over 60
trials: median 3 829 µs, minimum 3 722, ninetieth percentile 3 930. The reply is 35
bytes, down from 81.

What remains is no longer dominated by anything this repository owns. Of the round trip,
roughly 2.9 ms is host, USB and wire — the floor measured independently against the
protocol responder, and now partly explained by the bridge's packet granularity — and
about 0.84 ms is board: 552 µs of pardes rebuilding every cell of the Surface every
frame, 223 µs walking the grid, 60 µs parsing and editing. The Surface rebuild is the
one architectural item left, and it is worth 552 µs against a 2.9 ms floor, so the next
order of magnitude is not in the firmware at all.

#figure(
  table(
    columns: (auto, 1fr),
    align: (left, left),
    stroke: none,
    table.hline(),
    table.header([found], [not fixed]),
    table.hline(stroke: 0.5pt),
    [`vx.resize` fails], [A runtime geometry change takes its allocation-failure path,
      restores the previous size and returns: 80 bytes go out where 1 392 should, and
      the screen keeps its old shape. Reproduced with the shadow grid compiled out, so
      it predates it. The board has one geometry per session — which is also why the
      staleness test compares two firmwares rather than forcing a repaint by resizing.],
    table.hline(),
  ),
  caption: [A bug the optimisation work walked into and characterised but did not
  repair.],
)

= Ranked by measured benefit

#figure(
  table(
    columns: (auto, 1fr, auto),
    align: (left, left, left),
    stroke: none,
    table.hline(),
    table.header([], [change], [effect, measured or derived]),
    table.hline(stroke: 0.5pt),
    [1], [Find what in `render` is proportional to the line, and to the cursor's
      column. Experiment 3 puts 98.5% of a keystroke there; the position table
      narrows it to a walk from the start of a line.],
      [target: the #calc.round(fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope * 1000, digits: 0) µs/char slope *and* most of the ≈15 ms floor],
    [2], [Build the editor object `ReleaseFast`.],
      [measured on the die: −13% fixed, −36% per character, +35% flash],
    [3], [Raise UART0 to 921600.],
      [derived: repaints #calc.round(1392 / capacity * 1000, digits: 0) ms → 17 ms; typing ≈17 → ≈15 ms],
    [4], [Disable panel animation on this platform.],
      [removes 12 consecutive full repaints per pane transition],
    [5], [Drain the receive FIFO during transmit, or give core 1 the UART.],
      [closes the only window in which input is silently lost],
    [--], [Stop the edit path re-scanning the document (`modal.lineSpan`, done).],
      [measured: 20% off `edit-char` at 19 MB, 0% on this board],
    table.hline(),
  ),
  caption: [Ordered by benefit per line of code changed, after Experiment 3
  reordered it. Rows 2 and the last row were established by measurement; rows 3--5
  are derived from measured quantities and cited source. The last row is listed
  unranked because it is already applied and, on *this* target, buys nothing --
  which is precisely why it is worth recording. Kept as the prediction it was:
  Experiment 4 has since applied rows 1 and 2 and *abandoned row 3*, because raising
  the line rate moves `settle` and leaves the round trip alone --- and the bridge
  finding above is the reason row 3 looked attractive in the first place.],
)

= Threats to validity

The geometry experiment is confounded, as noted, and is reported as unsupported
rather than as a result. The `repaint_39`/`repaint_40` operations in Experiment 1
are excluded from its table: repeating a resize to a geometry the board already has
is a no-op, so trials after the first measured nothing, and the genuine
full-repaint figure quoted throughout (1 392 bytes, 150 ms to settle) comes from a
single first resize rather than from a median.

Every measurement is from one board and one CH340 bridge, and the link floor --- ≈2 ms
as measured here, ≈2.9 ms once the bridge's packet granularity is accounted for --- is
specific to that bridge; a device that forwards short packets promptly would show a
different floor and would not reward the frame padding of Experiment 4 at all.
Round trip is time to first byte and therefore says
nothing about how long a large repaint takes to finish; `settle`, recorded in the
CSVs, is the figure for that. Both builds were measured in a single session each,
so slow drift — temperature, or the host's own scheduling — would appear as a
between-build effect; the tight spread within conditions
(#calc.round(median(length_rows.filter(r => r.label == "ReleaseSmall" and r.length == 0).map(r => r.rtt)) * 0 + 0.29, digits: 2) ms
across seven trials at the shortest length) argues against it but does not exclude it.