summaryrefslogtreecommitdiff
path: root/experiments/report.typ
blob: 0403a1a7be9ecd3c9c3c6f07d9342af790fe567a (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
#import "@preview/cetz:0.5.1"

#set document(title: "Where the ESP32-P4 editor's latency goes", author: "measured on ESP32-P4 rev v1.3")
#set page(margin: 2cm, numbering: "1")
#set text(font: ("Libertinus Serif", "DejaVu Serif"), size: 10.5pt)
#set par(justify: true)
#show raw: set text(font: "DejaVu Sans Mono", size: 9pt)
#show heading: it => block(above: 1.4em, below: 0.7em, it)

#align(center)[
  #text(17pt, weight: "bold")[Where the ESP32-P4 editor's latency goes]

  #v(0.2em)
  #text(10pt)[A measured account, on silicon, of the `pardes` editor running as
  ESP32-P4 firmware with one 115200-baud serial line as its only I/O]
]

#v(0.5em)

#let capacity = 11520.0

// ---------------------------------------------------------------- data plumbing
// The figures read the raw per-trial CSV the instrument emits. Nothing here is a
// transcribed number: if a run is repeated, the document changes with it.
#let rows(file) = {
  let out = ()
  for r in csv(file) {
    if r.at(0) == "label" { continue }
    out.push((
      label: r.at(0), cols: int(r.at(2)), rows: int(r.at(3)),
      length: int(r.at(4)), op: r.at(5),
      rtt: int(r.at(7)) / 1000.0, settle: int(r.at(8)) / 1000.0, bytes: int(r.at(9)),
    ))
  }
  out
}

#let median(xs) = {
  let s = xs.sorted()
  if s.len() == 0 { return 0.0 }
  s.at(int(s.len() / 2))
}

// One expression on one line: a continuation beginning with `+` is list markup to Typst, and it
// rendered the file names as a numbered item on page one.
#let length_rows = (
  rows("length-ReleaseSmall.csv") + rows("length-ReleaseFast.csv") +
    rows("length-ReleaseSmall-lineSpan.csv") + rows("length-ReleaseFast-lineSpan.csv") +
    rows("length-RFast-grapheme.csv") + rows("length-RFast-print.csv") +
    rows("length-RFast-shadow.csv") + rows("length-RFast-fastcmp.csv") +
    rows("length-CPU360-final.csv") + rows("length-DirectEmit.csv") +
    rows("length-WordCmp.csv")
)
#let ops_rows = rows("ops-ReleaseSmall.csv") + rows("ops-ReleaseFast.csv")

// Figure 1 compares the two optimisation modes only; the `lineSpan` variants are the SAME source
// change applied to each, and are tabulated separately in Experiment 3 rather than plotted, because
// four indistinguishable pairs of lines would be a worse picture than two.
#let builds = ("ReleaseSmall", "ReleaseFast")
#let all_builds = ("ReleaseSmall", "ReleaseSmall-lineSpan", "ReleaseFast", "ReleaseFast-lineSpan")
#let lengths = (0, 20, 40, 80, 160)

#let med_rtt(rs, pred) = median(rs.filter(pred).map(r => r.rtt))
#let n_of(rs, pred) = rs.filter(pred).len()

// Least squares, for the slope that is the whole point of figure 2.
#let fit(xs, ys) = {
  let n = xs.len()
  let mx = xs.sum() / n
  let my = ys.sum() / n
  let num = 0.0
  let den = 0.0
  for i in range(n) {
    num += (xs.at(i) - mx) * (ys.at(i) - my)
    den += (xs.at(i) - mx) * (xs.at(i) - mx)
  }
  let slope = num / den
  (slope: slope, intercept: my - slope * mx)
}

// ---------------------------------------------------------------- summary
= What was found

The editor is *not* limited by its serial line. Measured against a checksum-verified
protocol, the link carries #calc.round(11496 / capacity * 100, digits: 1)% of its theoretical
capacity in both directions with zero corruption, and typing at 100 characters a
second loses nothing and uses under a tenth of the wire.

What limits it is *computation per input event*, and that cost has two parts, both
measured here:

#block(inset: (left: 1em))[
  *A fixed cost of ≈#calc.round(med_rtt(ops_rows, r => r.label == "ReleaseSmall" and r.op == "motion_h"), digits: 1) ms per event*, which does not
  depend on how much the screen changed. A cursor motion emitting 40 bytes and an
  insert-and-escape emitting 206 bytes cost the same round trip to within 0.3 ms.

  *A cost proportional to the document*, at
  #calc.round(fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope * 1000, digits: 1) µs
  per character already in the line, per keystroke, which is what makes the editor
  feel worse the more you have written.
]

*Both live entirely in the renderer.* Timed separately on the die, parsing the
keystroke and applying the edit takes a flat ≈220 µs regardless of document size —
1.5% of the total — while `render` carries the whole ≈15 ms floor and every
microsecond of the slope. That result contradicted the mechanism the source reading
implied, and it was only reachable by instrumenting the firmware; the fix that
source reading suggested was written, measured, and found to be worth 20% on a 19 MB
file and nothing at all on this board. Experiment 3.

Both were then cut. A keystroke is *3.65 ms, from 16.99*, and the per-character term is
0.93 µs, from 54.3. Three of the seven steps that did it came from *not asking Unicode
about ASCII*; the largest single one came from this repository's own `present` copying all
480 cells into vaxis every frame whether or not any had changed; and the last three were a
clock that was only ever a divider away, a renderer that stopped asking vaxis to recompute
a diff `present` had already done, and a row comparison that turned out to be running one
byte at a time. Experiment 4.

The 4 ms target is met, over 80 trials, with the slowest of them at 3 912 µs. Three
findings on the way there are worth more than the number. A 4× clock bought 2.6×, because
the grid walk is bounded by memory rather than by the core. The direct renderer, with less
computation and a quarter of the bytes, first measured *slower* — the CH340 bridging the
board to the host forwards a bulk packet only when it is full, so a reply too small to
fill one waits about a millisecond for a timer; the frame has a minimum size and it
belongs to the transport. And the single largest remaining win was not an algorithm at
all: the firmware's biggest read ran at 3.2 cycles per byte because `std.mem.eql`
compares a byte at a time.

= The instrument

Two programs and a protocol, all in this repository.

`tools/perfproto.zig` is a framed protocol shared *verbatim* by the host tool and
the firmware, so a frame written by one and parsed by the other cannot drift:
a nine-byte header (`"P4"`, op, length, CRC-32 of the payload) then the payload.
`examples/uartperf.zig` answers it on the board; `tools/bench_main.zig` drives it
from the host.

The checksum is the point. RX overrun on this UART is undetected in hardware and
uncounted in the driver, so a byte that never arrives is indistinguishable from a
byte that arrived late. A frame with a length and a CRC turns both into facts: the
board reports a CRC over exactly the bytes it received, the host compares it
against a CRC over exactly the bytes it sent, and a throughput figure that is not
checksummed is only a guess about how fast data was corrupted.

`tools/rtt.zig` provides the two timing functions everything else is built on:
`roundTrip`, which sends a stimulus and returns when the first response byte
arrives, and `measure`, which repeats it. *Round trip is time to the first
response byte, not to the last.* A renderer that begins drawing in 8 ms and
finishes in 130 ms feels immediate; one that thinks for 130 ms and then draws in
8 ms feels broken; measuring only when the wire falls quiet cannot tell them
apart. The time to the last byte is recorded separately as `settle`.

Microseconds throughout: at 115200 baud one byte occupies 87 µs, so a millisecond
clock would quantise these measurements into buckets eleven bytes wide.

= Method

All numbers are from one ESP32-P4 rev v1.3 over its CH340 bridge at 115200 baud
8N1, giving #capacity B/s in each direction. Each condition is measured
#n_of(length_rows, r => r.label == "ReleaseSmall" and r.length == 0) times and
reported as a median; the raw per-trial rows are in `experiments/*.csv` and this
document computes its figures from them directly.

#block(breakable: false)[
```
zig build flash -Dapp=examples/uartperf.zig     # the link ceiling
zig-out/bin/p4-bench --link

zig build flash -Dpardes                        # the editor
zig-out/bin/p4-bench --sweep length --repeat 7 --csv --label ReleaseSmall
zig-out/bin/p4-bench --sweep ops    --repeat 7 --csv --label ReleaseSmall
```
]

The editor is modal, so every editor run enters insert mode once before timing and
uses a single inserted character as the comparable unit of work. In the length
experiment the line is primed to exactly $n$ characters *without* measuring, so the
timed keystroke always sees a document of known size.

= Baseline: the link is not the problem

With only the protocol responder running — no editor — the link performs as well as
it can:

#figure(
  table(
    columns: (auto, auto, auto, auto),
    align: (left, right, right, left),
    stroke: none,
    table.hline(),
    table.header([direction], [measured], [of capacity], [integrity]),
    table.hline(stroke: 0.5pt),
    [uplink, host → board], [11 496 B/s], [99%], [CRC verified, 32 768 B],
    [downlink, board → host], [11 413 B/s], [99%], [CRC verified, 32 768 B],
    [round trip, 13 B each way], [4.2–4.5 ms], [--], [0 lost of 20],
    table.hline(),
  ),
  caption: [The UART driver and the wire, with nothing else running. Every byte
  accounted for by checksum.],
)

Of that 4.2 ms round trip, 2.26 ms is the wire itself (26 bytes at #capacity B/s);
the remaining ≈2 ms is the host, the USB bridge and the firmware's parse. *This
≈2 ms is the floor every editor measurement below sits on*, and subtracting it is
how the editor's own share is obtained.

= Hypotheses

#figure(
  table(
    columns: (auto, 1fr, auto),
    align: (left, left, left),
    stroke: none,
    table.hline(),
    table.header([], [hypothesis], [verdict]),
    table.hline(stroke: 0.5pt),
    [H1], [Latency is compute-bound, not transmission-bound: round trip is
      independent of how many bytes the operation emits.], [*confirmed*],
    [H2], [The editor object's optimisation mode materially changes latency.], [*confirmed*],
    [H3], [An edit is $O(n)$ in the document: round trip rises linearly with the
      characters already in the line.], [*confirmed*],
    [H4], [Cost is paid per screen cell, so a smaller grid is proportionally
      cheaper.], [*not supported*],
    [H5], [Typing at a human rate loses input.], [*refuted*],
    table.hline(),
  ),
  caption: [Stated before measuring; each is settled by one experiment below.],
)

= Experiment 1 --- output size does not predict latency (H1)

Seven operations, chosen to span a five-fold range of emitted bytes at a fixed
40×12 geometry.

#figure(
  {
    let names = ("motion_h", "motion_l", "line_start", "line_end", "insert_esc")
    table(
      columns: (auto, auto, auto, auto, auto),
      align: (left, right, right, right, right),
      stroke: none,
      table.hline(),
      table.header([operation], [bytes], [wire time], [round trip], [implied compute]),
      table.hline(stroke: 0.5pt),
      ..names.map(n => {
        let rs = ops_rows.filter(r => r.label == "ReleaseSmall" and r.op == n)
        let b = median(rs.map(r => r.bytes))
        let t = median(rs.map(r => r.rtt))
        (
          raw(n),
          [#b B],
          [#calc.round(b / capacity * 1000, digits: 2) ms],
          [#calc.round(t, digits: 2) ms],
          [#calc.round(t - 2.0, digits: 2) ms],
        )
      }).flatten(),
      table.hline(),
    )
  },
  caption: [`ReleaseSmall`, 40×12, median of 7. Emitted bytes vary 5×; the round
  trip does not vary at all. "Implied compute" subtracts only the ≈2 ms link floor,
  because round trip is measured to the *first* byte and so does not contain the
  transmission of the rest.],
)

A five-fold change in output moves the round trip by less than 2%. Whatever the
editor is doing for ≈15 ms, it is doing before it emits anything, and it is not
proportional to what changed on screen. *H1 is confirmed*, and it is the reason
raising the baud rate cannot fix typing latency: there is almost no wire in it.

= Experiment 2 --- an edit costs the whole document (H3, H2)

The controlled variable is the number of characters already in the line. The
measured quantity is unchanged: the round trip of one further inserted character.

#figure(
  cetz.canvas(length: 1cm, {
    import cetz.draw: *
    let w = 11.0
    let h = 6.0
    let xmax = 176.0
    let ymax = 28.0
    let px(v) = v / xmax * w
    let py(v) = v / ymax * h

    // axes
    line((0, 0), (w + 0.3, 0), mark: (end: "straight"), stroke: 0.6pt)
    line((0, 0), (0, h + 0.3), mark: (end: "straight"), stroke: 0.6pt)
    content((w / 2, -0.85), text(9pt)[characters already in the line])
    content((-1.15, h / 2), angle: 90deg, text(9pt)[round trip (ms)])

    for l in lengths {
      line((px(l), 0), (px(l), -0.12), stroke: 0.6pt)
      content((px(l), -0.38), text(8pt)[#l])
    }
    for v in (0, 5, 10, 15, 20, 25) {
      line((0, py(v)), (-0.12, py(v)), stroke: 0.6pt)
      content((-0.42, py(v)), text(8pt)[#v])
      if v > 0 { line((0, py(v)), (w, py(v)), stroke: (paint: luma(88%), thickness: 0.4pt)) }
    }

    let colours = (ReleaseSmall: rgb("#b3261e"), ReleaseFast: rgb("#1a5fb4"))
    for b in builds {
      let ys = lengths.map(l => med_rtt(length_rows, r => r.label == b and r.length == l))
      let f = fit(lengths.map(l => l * 1.0), ys)
      // fitted line, drawn under the data so the points remain readable
      line(
        (px(0), py(f.intercept)),
        (px(xmax), py(f.intercept + f.slope * xmax)),
        stroke: (paint: colours.at(b).lighten(55%), thickness: 1.6pt),
      )
      // every individual trial, so the spread is visible rather than asserted
      for r in length_rows.filter(r => r.label == b) {
        circle((px(r.length * 1.0), py(r.rtt)), radius: 0.045, fill: colours.at(b).lighten(30%), stroke: none)
      }
      line(..lengths.zip(ys).map(p => (px(p.at(0) * 1.0), py(p.at(1)))), stroke: (paint: colours.at(b), thickness: 1.1pt))
      for p in lengths.zip(ys) {
        circle((px(p.at(0) * 1.0), py(p.at(1))), radius: 0.075, fill: colours.at(b), stroke: none)
      }
      content(
        (px(xmax) + 0.15, py(f.intercept + f.slope * xmax)),
        anchor: "west",
        text(8pt, fill: colours.at(b))[#b],
      )
    }
  }),
  caption: [Round trip against document length, every trial plotted (7 per point),
  medians joined, least-squares fit behind. Both builds are straight lines: an edit
  is $O(n)$ in the document.],
)

#figure(
  table(
    columns: (auto, auto, auto, auto),
    align: (left, right, right, right),
    stroke: none,
    table.hline(),
    table.header([build], [fixed cost], [per character], [round trip at 160 chars]),
    table.hline(stroke: 0.5pt),
    ..builds.map(b => {
      let ys = lengths.map(l => med_rtt(length_rows, r => r.label == b and r.length == l))
      let f = fit(lengths.map(l => l * 1.0), ys)
      (
        raw(b),
        [#calc.round(f.intercept, digits: 2) ms],
        [#calc.round(f.slope * 1000, digits: 1) µs],
        [#calc.round(ys.last(), digits: 2) ms],
      )
    }).flatten(),
    table.hline(),
  ),
  caption: [Fitted from the medians. The slope is the interesting column. Its
  *mechanism* is settled by Experiment 3, not by this fit.],
)

The linear term is not subtle and it is not a cache effect: it is visible from 20
characters and the fit is straight over the whole range. At
#calc.round(fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope * 1000, digits: 0) µs
per character on a ≈90 MHz core — roughly 5 000 cycles for every character already
in the line — it is far too much work to be a `memcpy`, so something is making a
substantial pass per character. *H3 is confirmed as an observation.* What that pass
actually is turned out not to be what the source reading suggested, which is
Experiment 3.

The two series also settle H2. `ReleaseFast` lowers the fixed cost by
#calc.round(
  (1 - fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseFast" and r.length == l))).intercept /
     fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).intercept) * 100,
  digits: 0)% and the per-character cost by
#calc.round(
  (1 - fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseFast" and r.length == l))).slope /
     fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope) * 100,
  digits: 0)%, so its advantage *grows with the document*: 13% at an empty line and
21% at 160 characters. It costs 809 536 bytes of flash against 598 544, which is
35% more of a 1 536 000-byte partition — affordable, and the only change measured
here that improves both terms at once. *H2 is confirmed.*

= Experiment 3 --- it is all in the renderer, and the obvious fix was wrong

Everything above measures a keystroke from the host, which cannot see *what* the
firmware spent the time on. Reading the source suggested an answer: the edit path
builds each new document with `modal.spliceAlloc`, a fresh allocation and a copy of
the whole buffer, and `insertAt` called `modal.lineCount` — `std.mem.count` over
every byte — *twice*, merely to clamp a row. That is three whole-document passes
before a single character can be inserted, which fits a linear slope exactly.

It is also, on this board, almost entirely irrelevant. Timing the two phases
separately on the die (`-Dprof`, two reads of the cycle counter around
`pardes_p4_input` and `pardes_p4_render`) gives:

#figure(
  {
    let a = ()
    for r in csv("attribution.csv") {
      if r.at(0) == "label" { continue }
      a.push((chars: int(r.at(2)), input: int(r.at(5)), render: int(r.at(6))))
    }
    table(
      columns: (auto, auto, auto, auto),
      align: (right, right, right, right),
      stroke: none,
      table.hline(),
      table.header([characters in line], [input: parse + edit], [render], [render share]),
      table.hline(stroke: 0.5pt),
      ..a.map(r => (
        [#r.chars],
        [#r.input µs],
        [*#r.render µs*],
        [#calc.round(r.render / (r.input + r.render) * 100, digits: 1)%],
      )).flatten(),
      table.hline(),
    )
  },
  caption: [On-board cycle counts, `ReleaseSmall`. Input is flat; render carries
  both the fixed cost and the entire slope.],
)

*Input is flat at ≈220 µs and does not grow with the document at all.* The fixed
≈15 ms and every microsecond of the per-character slope are inside
`pardes_p4_render`. The edit path — the allocation, the copy, the double line count
— is 1.5% of a keystroke and could be made free without anyone noticing.

This was worth proving rather than assuming, because the fix implied by the source
reading was written and measured. `modal.insertAt` now takes one *bounded* scan
through a new `modal.lineSpan`, which stops at the row it wants instead of counting
the whole document, and pays for a full count only on the rare clamping path where
the cursor is past the end. On the host harness (`zig build perf`, which drives the
same core over 61 KB to 19 MB fixtures) that is a real win, reproduced over three
independent runs:

#figure(
  table(
    columns: (auto, auto, auto, auto),
    align: (left, right, right, right),
    stroke: none,
    table.hline(),
    table.header([fixture], [1 000 lines], [50 000 lines], [300 000 lines]),
    table.hline(stroke: 0.5pt),
    [`edit-char`, before], [730 µs], [2 996 µs], [15 030 µs],
    [`edit-char`, after], [721 µs], [2 414 µs], [10 955 µs],
    [ratio, three runs], [0.98×], [0.79–0.82×], [0.78–0.83×],
    table.hline(),
  ),
  caption: [Host harness, 25 samples per cell. Every untouched operation stayed at
  1.00×, which is stronger evidence than any single cell.],
)

And on the board it changed *nothing*, in either optimisation mode. All four
combinations were measured on the die, 5 lengths × 7 trials each:

#figure(
  table(
    columns: (auto, auto, auto, auto, auto),
    align: (left, right, right, right, right),
    stroke: none,
    table.hline(),
    table.header([configuration], [fixed cost], [per character], [at 160 chars], [vs baseline]),
    table.hline(stroke: 0.5pt),
    ..all_builds.map(b => {
      let xs = lengths.map(l => l * 1.0)
      let ys = lengths.map(l => med_rtt(length_rows, r => r.label == b and r.length == l))
      let f = fit(xs, ys)
      let at160 = ys.last()
      let ref160 = med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == 160)
      (
        raw(b),
        [#calc.round(f.intercept, digits: 2) ms],
        [#calc.round(f.slope * 1000, digits: 1) µs],
        [#calc.round(at160, digits: 2) ms],
        [#calc.round(at160 / ref160, digits: 2)×],
      )
    }).flatten(),
    table.hline(),
  ),
  caption: [The edit-path change is invisible in both modes; the optimisation mode
  is the whole of the difference. `ReleaseFast` + `lineSpan` is indistinguishable
  from `ReleaseFast` alone.],
)

That is not a contradiction, it is the same fact seen twice: the removed passes are
$O(#h(0.1em)$document$)$, and this board's document is a few hundred *bytes*, so two
scans of it cost nothing worth measuring. The identical change is worth 20% on a
19 MB file and 0% on a 240-character one.

The measurement did change one thing about the board, though, and it is not the
source: `-Doptimize` defaulted to `Debug`, so a plain `zig build -Dplatform=p4`
produced an object that *cannot run* — `Debug` wraps every tier in `allocators.zig`
in a `DebugAllocator` whose metadata is page-granular, and one 4 KiB page per size
class does not fit in the 384 KiB the board hands over. The p4 target now defaults to
`ReleaseFast`, which is the mode this experiment chose rather than a preference, and
an explicit `-Doptimize=` still wins. The 21% is therefore what the default build now
gives, not something to remember to ask for.

The lesson is the one the instrument exists to enforce. A plausible mechanism, read
off the source and consistent with the shape of the data, was wrong about where the
time went — and it took a measurement *inside* the firmware to say so. The renderer
is the target; the next question is what in it is proportional to the line, and the
position experiment already narrows that: cost follows the cursor's column as well
as the document's size, which is the signature of a walk from the start of a line.

#figure(
  table(
    columns: (auto, auto, auto),
    align: (left, right, right),
    stroke: none,
    table.hline(),
    table.header([insert position in a fixed 320-character line], [round trip], [bytes emitted]),
    table.hline(stroke: 0.5pt),
    [column 320 (end)], [33.9 ms], [28 B],
    [column 0 (start)], [26.0 ms], [81 B],
    [column 320 again], [33.7 ms], [28 B],
    table.hline(),
  ),
  caption: [Same document throughout; only the cursor moved. The end of the line
  costs 7.8 ms more than the start while emitting *a third* as many bytes — output
  size and latency are not merely uncorrelated here, they are inverted.],
)

= What the measurements rule out

*Input is not being lost while typing (H5, refuted).* At every rate from 6 to 100
characters a second, every stimulus was answered: zero lost, with the wire never
above 9% occupied. The silent-overrun window is real but it is not reachable by
typing — it needs a full-screen repaint, which holds the wire for
#calc.round(1392 / capacity * 1000, digits: 0) ms while nothing drains the
128-byte receive FIFO. Those repaints come from resizing, pane transitions, theme
changes and scrolling, not from editing.

*Screen area does not explain the fixed cost (H4, not supported).* Round trip did
not scale with cell count: 30×12 (360 cells) measured faster than 40×8 (320 cells).
The pattern was not monotonic in area, and that experiment carried a confound —
characters accumulated across conditions, which the length experiment then showed
to matter — so it is reported as unsupported rather than refuted. Re-running it
with a reset between conditions is the obvious next measurement.

= Two levers that were researched rather than measured

*Raising the line rate.* UART0 is at 115200 because the second-stage bootloader
left it there; the firmware never programs the divider. Reaching 921600 needs one
`UART_CLKDIV_SYNC` write (integer 43, fraction 6) on the existing 40 MHz crystal
followed by `UART_REG_UPDATE` — no clock-source change, and an error of +0.064%,
far inside a UART's tolerance. 2 Mbaud is exactly representable but this board's
CH340 is already documented unreliable there.

The payoff is real but narrow, and Experiment 1 says why: a full repaint's
#calc.round(1392 / capacity * 1000, digits: 0) ms of wire becomes 17 ms, which
removes the dead zones outright — but a keystroke's round trip only falls from
≈17 ms to ≈15 ms, because there is barely any wire in it. *Raise the baud to fix
repaints, not to fix typing.*

*The second core.* The chip has two rv32imafc cores at ≈90 MHz plus a 16 MHz
low-power core. The second high-performance core is parked at power-on (clock
gated, in reset, stall armed) and needs four register writes plus a
`gp`/`sp`/`mtvec` trampoline to start. Crucially, the two cores share a *single*
L1 data cache with atomic read-modify-write enabled, so a lock-free ring in the
384 KiB L2MEM heap needs only RISC-V fences and no cache maintenance.

The honest verdict is that the second core cannot reduce the ≈15 ms — it can only
move it. Giving core 1 the UART is worth doing because it *eliminates* the silent
input loss: core 0 would never again block for
#calc.round(1392 / capacity * 1000, digits: 0) ms inside a write with nothing
draining the receive FIFO. Giving core 1 the rendering is not worth doing: the
diff reads the same mutable structures the editor is writing, so it is a rewrite
of the renderer with a large race surface on a board that has no debugger, and it
would not shorten a single keystroke's round trip anyway, because the host still
waits for that render.

= Experiment 4 --- cutting it, and what each cut was worth

The target set after Experiment 3 was 4 ms. Round trip is time to the first response
byte, so it is ≈2 ms of host and USB latency plus computation; 4 ms therefore means a
compute budget of about 2 ms, and raising the line rate cannot help — at 115200 an
81-byte reply is 7 ms of wire but almost none of it lands before the first byte.

Every step below was profiled first, in the board's *exact* configuration: 40×12 and
`-Dtree-sitter=disabled`, because the P4 build has no tree-sitter and a profile that
includes it is a profile of a different program. Half of the first profile was
tree-sitter, which the board never runs.

#figure(
  table(
    columns: (auto, auto, auto, auto, auto),
    align: (left, right, right, right, right),
    stroke: none,
    table.hline(),
    table.header([step], [fixed cost], [per char], [at 160 chars], [vs start]),
    table.hline(stroke: 0.5pt),
    ..(
      ("ReleaseSmall", "ReleaseFast", "RFast-grapheme", "RFast-print", "RFast-shadow",
        "RFast-fastcmp", "CPU360-final", "DirectEmit", "WordCmp")
    ).map(b => {
      let xs = lengths.map(l => l * 1.0)
      let ys = lengths.map(l => med_rtt(length_rows, r => r.label == b and r.length == l))
      let f = fit(xs, ys)
      let at160 = ys.last()
      let ref160 = med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == 160)
      (
        raw(b),
        [#calc.round(f.intercept, digits: 2) ms],
        [#calc.round(f.slope * 1000, digits: 1) µs],
        [#calc.round(at160, digits: 2) ms],
        [#calc.round(at160 / ref160, digits: 2)×],
      )
    }).flatten(),
    table.hline(),
  ),
  caption: [Each row is a separate firmware measured on the die, 5 document lengths ×
  7--9 trials. The per-character column and the fixed column move for different reasons
  and are worth reading separately. The last three rows are the clock raise, the renderer
  and the row comparison, and none of them is a source optimisation in the sense the four
  above them are.],
)

*Not asking Unicode about ASCII* accounts for the per-character column. Three fast
paths, each guarded so non-ASCII text takes exactly the road it took before:
`modal.graphemeStart` (21.5% of a keystroke, and the reason cost followed the cursor's
column — it walks graphemes from the start of the text with the full UAX #29 state
machine), `Surface.print` (26.2%, which per character took a UTF-8 length, a decode, a
*freshly constructed* grapheme iterator, a validation and a width lookup to conclude
that `y` is one cell), and `graphemeDisplayWidth` (6.9%, all of it asking a Unicode
table about ASCII). The guard is the same in each: an ASCII scalar is its own grapheme
cluster unless it is a CR before an LF, because every rule that could join it —
Extend, ZWJ, SpacingMark, Prepend, Regional_Indicator — is spelled with non-ASCII
scalars. Per character: 54.3 → 6.9 µs.

*The fixed cost needed the frame broken open.* A second render with nothing changed
cost the same as the first, so the ~15 ms was unconditional; `pardes_p4_frame_prof`
then reported the three stages separately.

#figure(
  table(
    columns: (auto, auto, auto),
    align: (left, right, right),
    stroke: none,
    table.hline(),
    table.header([stage of one frame], [before], [after]),
    table.hline(stroke: 0.5pt),
    [copy the Surface into vaxis's grid], [6 750 µs], [*979 µs*],
    [vaxis diffs its grid and emits], [2 460 µs], [2 455 µs],
    [pardes rebuilds the whole Surface], [≈2 600 µs], [≈2 000 µs],
    [push the bytes into the UART], [1 µs], [1 µs],
    table.hline(),
  ),
  caption: [Cycle counts read on the die. The largest item was in *this repository's*
  own `present`, not in pardes and not in vaxis: it copied all 480 cells every frame,
  whether or not any had changed.],
)

So `present` now keeps the previous Surface and tells vaxis only what moved, comparing
cells as bytes rather than through `std.meta.eql` on a colour union and eight booleans.
The grid lives in `.bss`, and that is a bug fix rather than a flourish: allocated from
the editor's heap it was enough to make `vx.resize` fail.

== How a rendering change was made safe to believe

A latency benchmark cannot see a corrupted screen, and `src/p4.zig` compiles for this
board alone — no host suite covers it. So each optimisation carries a comptime switch
whose `false` arm is the previous behaviour — `shadow_grid` for the incremental walk,
`direct_emit` for the renderer — and the firmwares were each run against the same
18-step workload on the die: inserts, deletes, motions that move the modified-marker, a
line outgrowing the viewport, backspaces that shrink it. The screens were reconstructed
from the wire and compared.

That comparison had to be corrected twice, and the second time it had already produced a
false result. Tracking characters only would have passed a colour regression in silence,
so it began hashing SGR as well — but it hashed the escape *parameters* applied to each
cell, which is a history and not a state. Raising the clock changed which keystrokes
shared a frame, the same final colours were reached by a different route, and the
verifier *reported a difference on the die that did not exist*. It now decodes SGR into
resolved state — foreground, background, attribute set per cell — and two screens compare
equal when every cell holds the same character and the same resolved style, whatever
sequence of escapes the emitter chose to get there. Without that correction the
from-scratch renderer could not have been verified at all: it reaches identical screens
by deliberately different escapes.

All three pairs are identical under it: shadow grid against full repaint, 90 MHz against
360, and direct emission against vaxis. On the host side, where the shared code does
live: `snap` 95/95 scripts, `hxdiff` 481 cases with 0 mismatches, `hxparity` 561 cases
with 0 mismatches.

== The clock was a divider, not a design

The board had been executing at 90 MHz for every measurement above, because the stock
second-stage bootloader is built for it. The CPLL was *already* running at 360 MHz — 90
is exactly a quarter of it — so reaching 360 is four divider writes and a ROM call, with
no PLL to enable and no voltage step to sequence against. Verified against the systimer,
which is clocked from the crystal and therefore independent: 90 000 kHz before,
359 991 kHz after. Nothing else moves, and that is what makes it safe from a live
console: the UART0 baud clock comes from the crystal, the systimer from the crystal, and
the flash interface from a different PLL entirely.

Compute fell 6.42 → 2.44 ms: *2.6× for a 4× clock*, and the shortfall is the
interesting part. The grid walk reads 27 KB per frame and takes 226 µs at 360 MHz, about
six cycles a byte, so it is bounded by L2MEM bandwidth rather than by the core. A
prediction made before the measurement, and held.

== The renderer, and a minimum frame that belongs to the USB bridge

vaxis's diff was redundant: `present` has already worked out exactly which cells moved,
so vaxis was being told the answer and then computing it again against its own copy.
Emitting the escapes directly — one absolute cursor position per run of changed cells,
absolute SGR rather than a delta, a hand-rolled formatter for the two digits that
sequence needs — took the board's render from 1 362 µs to 779 µs and the reply from 81
bytes to 21.

And it measured *slower*: 4.72 ms against 4.37. Fewer bytes, less computation, worse
round trip is the shape of a wrong model, so the search moved off the firmware.

It is the USB bridge. The board reaches the host through a CH340, a full-speed part
whose bulk IN endpoint carries 32-byte packets, and it forwards a packet when the packet
is *full*. A 21-byte frame never fills one, so it waits in the bridge for an internal
timer to give up on more — worth about a millisecond, a quarter of the entire budget.
Routing through vaxis had only looked competitive because 81-byte frames fill a packet
by accident.

#figure(
  table(
    columns: (auto, auto, auto, 1fr),
    align: (right, right, right, left),
    stroke: none,
    table.hline(),
    table.header([frame], [min], [median], []),
    table.hline(stroke: 0.5pt),
    [21 B], [3 843 µs], [4 817 µs], [direct, never fills a packet],
    [49 B], [3 719 µs], [3 814 µs], [direct, padded past the boundary],
    [81 B], [4 373 µs], [4 475 µs], [through vaxis, fills one by accident],
    table.hline(),
  ),
  caption: [Identical board cost and a byte-identical screen across all three; only the
  size of the reply differs. Note the *minimum*: the 21-byte frame's floor is already
  530 µs below vaxis's, which is exactly the computation that was saved. Only the median
  was hostage to the bridge's timer.],
)

So the frame has a minimum size and it is the transport's, not the terminal's. The
emitter pays it, padding with repeated absolute cursor positioning: idempotent, already
the sequence a frame ends on, incapable of altering a cell. This is the bargain an
Ethernet runt frame makes — the medium has a minimum and the sender pays it — and it is
a real trade rather than a free one, since the filler is wire time that delays a later
frame. It applies only while the frame is small, which is when there is wire to spare.

A second instrument came out of this. The original bench sends keystrokes on a fixed
cadence, which locks the send phase to the host's 1 ms USB frame clock and makes the
round trip a staircase in board time — where a real saving can present as a regression.
Sleeping a uniform random 0–2 ms before each keystroke decorrelates them. That was not
what was happening here, but it had to be ruled out before the CH340 could be believed.

== The pad target, swept

Two anecdotes disagreed about whether a bigger frame arrives sooner: padding the
cursor-positioning frame from 21 to 49 bytes made it a millisecond faster, while padding
the cursor-hiding frame from 6 to 36 made it slower. So the pad target was swept as the
only variable — same firmware, same workload, same clock.

#figure(
  table(
    columns: (auto, auto, auto, auto, 1fr),
    align: (right, right, right, right, left),
    stroke: none,
    table.hline(),
    table.header([target], [frame], [fixed], [worst], []),
    table.hline(stroke: 0.5pt),
    [0 B], [21 B], [4 743 µs], [4 906 µs], [never fills a packet],
    [16 B], [21 B], [4 817 µs], [5 079 µs], [still never fills one],
    [*32 B*], [*35 B*], [*3 799 µs*], [*4 335 µs*], [*exactly `wMaxPacketSize`*],
    [48 B], [49 B], [3 811 µs], [4 334 µs], [past it, and no better],
    [64 B], [70 B], [3 813 µs], [4 317 µs], [past it, and no better],
    table.hline(),
  ),
  caption: [Crossing the packet boundary is worth about 950 µs; going past it buys
  nothing at all. 32 is not a fitted constant — it is `wMaxPacketSize` of endpoint 0x82
  as the device reports it in its own descriptor, and the sweep is what confirms the
  descriptor is the right thing to believe.],
)

That also found a hole in the padding: it covered the branch that positions the cursor
and not the branch that hides it, so a frame which only hid the cursor was six bytes and
waited out the timer. Hiding an already-hidden cursor is as idempotent as positioning it
twice.

== The largest read in the firmware was a byte at a time

The last win was not a fast path or a clock: it was noticing that the shadow-grid diff —
two 13 KB streams, every frame, comfortably the biggest memory access the firmware makes —
ran at 3.2 cycles per byte, about four times what word-wide loads need. That is the shape
of a byte-at-a-time loop, and `std.mem.eql` was it.

Comparing a `u32` at a time took the grid walk from *223 µs to 66 µs*, 0.95 cycles per
byte, off every keystroke at every document length. The alignment test has to be made at
runtime because `Cell` is all `u8` fields and so has alignment 1: whether a row starts on
a word boundary is a property of whoever allocated the Surface rather than of the type.
The answer is bit-for-bit identical, which is what matters — the diff still rests on byte
equality implying visual equality, so it can never claim two different cells are the same.

Two other candidates were measured and *reverted*, which is the more useful half of the
result. Replacing `Surface.fill`'s per-cell writes with a row-at-a-time `@memset` and
giving `Surface.set` a one-byte store instead of a runtime-length `@memcpy` removed 52
million instructions per host run — 3.9% — and changed the cycle count on the host by
nothing and the board by nothing. Ablating the whole-surface fill priced it at 41 µs, or
1.18 cycles per byte written. Writes on this part are cheap and already pipelined; it was
the READS that were slow, and only the read was worth fixing.

== Where it stopped

A keystroke is *3.65 ms*, from 16.99. The target was 4 ms and it is met with margin: over
80 phase-randomised trials the median is 3 687 µs, the minimum 3 571 and *the maximum
3 912* — every trial under 4 ms. Across document length it holds wherever the editor
actually shows the keystroke:

#figure(
  table(
    columns: (auto, auto, auto, auto, auto, auto, auto, auto),
    align: (left, right, right, right, right, right, right, right),
    stroke: none,
    table.hline(),
    table.header([chars], [0], [20], [40], [80], [160], [320], [640]),
    table.hline(stroke: 0.5pt),
    [round trip], [3 602], [3 624], [3 638], [3 790], [3 868], [4 026], [4 192],
    [cells changed], [some], [some], [some], [some], [some], [*none*], [*none*],
    table.hline(),
  ),
  caption: [Microseconds, median of nine trials each. At 320 characters and beyond the
  line has outgrown a 40×12 viewport, the cursor is off screen and the keystroke changes
  no cell at all: the frame is 36 bytes of cursor-hide and padding. Those two columns are
  the cost of an edit that displays *nothing*.],
)

What remains is not something this repository owns. Of the round trip, about 3.0 ms is
host, USB and wire — the floor measured independently against the protocol responder, and
now partly explained by the bridge's packet granularity — and 0.55 ms is board at an empty
line, of which pardes rebuilding all 480 cells of the Surface is 436 µs, the grid walk
66 µs, and parsing and applying the edit 44 µs.

The Surface rebuild is the one architectural item left, and the two columns above say
exactly why it is architectural rather than a fast path: at 640 characters it costs 985 µs
to produce a frame in which nothing changed. Every micro-optimisation attempted against it
here — the grapheme walk, the per-cell writes, the fill — returned between nothing and
21 µs, because the cost is not in any one of those but in doing the whole frame again.
Nothing in this repository can avoid work pardes has already done.

= Experiment 5 --- how big can the window be

The 40×12 grid was never a decision about the screen. It was memory, and the constraint was
four copies of every cell: vaxis keeps a `Screen` and an `InternalScreen`, pardes keeps its
`Surface`, and the shell keeps a shadow of it to diff against. Two of those four became dead
weight the moment the emitter started writing the wire itself — allocated every session,
never read. Sizing them to *one cell* removes the memory ceiling outright: the heap now
reports 336 KB free at every geometry tried, including ones that used to fail outright.

What is left is the honest limit. Every frame walks the whole grid, so the round trip is
linear in the cell count at *0.87 µs per cell*, measured.

#figure(
  table(
    columns: (auto, auto, auto, 1fr),
    align: (left, right, right, left),
    stroke: none,
    table.hline(),
    table.header([geometry], [cells], [round trip], []),
    table.hline(stroke: 0.5pt),
    [40×12], [480], [3 628 µs], [the old ceiling],
    [*56×14*], [*784*], [*3 930 µs*], [*the new default: +63% area, +40% width*],
    [56×16], [896], [3 965 µs], [over 4 ms on the phase-randomised instrument],
    [60×18], [1 080], [4 114 µs], [],
    [64×20], [1 280], [4 281 µs], [],
    [80×24], [1 920], [4 809 µs], [the classic terminal, one flag away],
    [100×30], [3 000], [5 743 µs], [],
    [120×36], [4 320], [6 923 µs], [],
    [140×42], [5 880], [8 310 µs], [the largest that runs at all],
    [160×48], [7 680], [---], [links, then traps at boot],
    [200×60], [12 000], [---], [does not link],
    table.hline(),
  ),
  caption: [`-Dp4-cols` / `-Dp4-rows`, because none of this is a constant. The two failures at
  the bottom are different failures and both are worth naming: 200×60 is refused by the
  linker — `.bss will not fit in region l2mem, overflowed by 76 036 bytes`, that `.bss` being
  the shadow grid, sized at compile time — while 160×48 links and then traps, the same
  pressure arriving at runtime as a collision instead of as a diagnostic. Neither is a heap
  problem any more, which is the surprise: the heap has 336 KB spare while `.bss` runs out.],
)

= Three bugs the latency work walked straight past

All three were invisible to every number in this document, and two of them had been there
since the port began.

*Keystrokes were being lost.* Reported as "a key is stuck and is only sent when I send a new
event". It was neither stuck nor late. The loop is read, apply, render, write, and the write
blocks while the transmit FIFO is full — real backpressure, and it should stay, because half
an escape sequence leaves the host terminal in the wrong colour for the rest of the session.
But nothing drained the RECEIVE FIFO during that wait, and that FIFO is 128 bytes, or 11 ms
of wire. Measured by counting what the firmware's loop actually took off the UART: a 200-byte
burst arrived as 197, a 300 as 257, a 600 as 478. Draining the receiver inside the wait fixes
the first window; a second one — applying the input, which costs 44 µs a keystroke on an empty
line and 63 µs at 640 characters — needed the loop to hand the editor eight bytes at a time
instead of a hundred and twenty-eight. With both, 4 096 bytes in a single write arrive intact,
and past that the loss is *counted* rather than silent.

*Every escape sequence was being shredded.* `vaxis.Parser` resolves a buffer containing
nothing but `0x1b` as the Escape key — deliberately, and correctly for a terminal, where the
kernel hands over a whole sequence in one read. This wire hands over one byte at a time, 87 µs
apart, so the first byte of every sequence arrived alone and was resolved as Escape and the
rest arrived as ordinary keys. A mouse click came through as ten key presses, and the `0`
among them is "go to column zero" in normal mode — which is exactly where the cursor kept
landing, and why this looked at first like a coordinate bug. Arrow keys, function keys and the
host's own in-band resize reports were all being taken apart the same way. The shell now holds
a lone ESC for 10 ms, two orders of magnitude longer than the wire needs and imperceptible to
the person pressing it.

Finding it took instrumenting the ABI to print the tag of every event the shell applied. Ten
`key_press` where one `mouse` belonged is not something any amount of reading the coordinate
arithmetic would have shown, and it had been read twice.

*The mouse was never enabled.* With the sequences intact, that was one line: the shell already
handled mouse events, because it mirrors the tty shell. Spelled out rather than taken from
`vx.setMouseMode`, which asks for `1002;1003;1004;1006` — 1003 being any-motion tracking, a
report per cell the pointer crosses with no button held. On this line that is dozens of
15-byte reports for one sweep, arriving as input the editor must parse while it paints, and
arriving whether anyone wants it or not. The console thins drag reports to 25 a second on the
way in, newest-wins, which is right for motion and only for motion: where the pointer passed
through is not information an editor can use.


#figure(
  table(
    columns: (auto, 1fr),
    align: (left, left),
    stroke: none,
    table.hline(),
    table.header([found], [outcome]),
    table.hline(stroke: 0.5pt),
    [`vx.resize` fails], [A runtime geometry change took its allocation-failure path, restored
      the previous size and returned: 80 bytes went out where 1 392 should, and the screen kept
      its old shape. Reproduced with the shadow grid compiled out, so it predated it, and left
      unrepaired through three experiments --- which is also why the staleness test compares two
      firmwares rather than forcing a repaint by resizing.

      *Fixed, and not on purpose.* The failing allocation was vaxis's two grids, and the emitter
      stopped needing them; there is nothing left to reallocate, so the path that failed is gone
      rather than repaired. Verified on the die: an in-band report re-lays the screen from 56
      columns to 30 and back. The pleasant kind of consequence, and worth recording as a
      consequence rather than as a fix --- nobody set out to repair it.],
    table.hline(),
  ),
  caption: [A bug that three experiments characterised and did not repair, removed by a change
  aimed at something else entirely.],
)

= Ranked by measured benefit

#figure(
  table(
    columns: (auto, 1fr, auto),
    align: (left, left, left),
    stroke: none,
    table.hline(),
    table.header([], [change], [effect, measured or derived]),
    table.hline(stroke: 0.5pt),
    [1], [Find what in `render` is proportional to the line, and to the cursor's
      column. Experiment 3 puts 98.5% of a keystroke there; the position table
      narrows it to a walk from the start of a line.],
      [target: the #calc.round(fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope * 1000, digits: 0) µs/char slope *and* most of the ≈15 ms floor],
    [2], [Build the editor object `ReleaseFast`.],
      [measured on the die: −13% fixed, −36% per character, +35% flash],
    [3], [Raise UART0 to 921600.],
      [derived: repaints #calc.round(1392 / capacity * 1000, digits: 0) ms → 17 ms; typing ≈17 → ≈15 ms],
    [4], [Disable panel animation on this platform.],
      [removes 12 consecutive full repaints per pane transition],
    [5], [Drain the receive FIFO during transmit, or give core 1 the UART.],
      [closes the only window in which input is silently lost],
    [--], [Stop the edit path re-scanning the document (`modal.lineSpan`, done).],
      [measured: 20% off `edit-char` at 19 MB, 0% on this board],
    table.hline(),
  ),
  caption: [Ordered by benefit per line of code changed, after Experiment 3
  reordered it. Rows 2 and the last row were established by measurement; rows 3--5
  are derived from measured quantities and cited source. The last row is listed
  unranked because it is already applied and, on *this* target, buys nothing --
  which is precisely why it is worth recording. Kept as the prediction it was:
  Experiment 4 has since applied rows 1 and 2 and left row 3 unapplied. Row 3 deserves a
  correction rather than a dismissal: it was abandoned because a higher line rate moves
  `settle` and leaves the round trip alone, and that reasoning assumed the first byte
  reaches the host as soon as it is sent. The bridge finding says otherwise --- delivery
  waits for 32 bytes, which is 278 µs of wire at 115200 and would be 35 µs at 921600, so
  the raise is worth roughly 240 µs of the round trip after all. It stays unapplied
  because one 115200-baud line is the premise of this port, not a free variable, and
  meeting the target by changing the link would be answering a different question.],
)

= Threats to validity

The geometry experiment is confounded, as noted, and is reported as unsupported
rather than as a result. The `repaint_39`/`repaint_40` operations in Experiment 1
are excluded from its table: repeating a resize to a geometry the board already has
is a no-op, so trials after the first measured nothing, and the genuine
full-repaint figure quoted throughout (1 392 bytes, 150 ms to settle) comes from a
single first resize rather than from a median.

Every measurement is from one board and one CH340 bridge, and the link floor --- ≈2 ms
as measured here, ≈2.9 ms once the bridge's packet granularity is accounted for --- is
specific to that bridge; a device that forwards short packets promptly would show a
different floor and would not reward the frame padding of Experiment 4 at all.
Round trip is time to first byte and therefore says
nothing about how long a large repaint takes to finish; `settle`, recorded in the
CSVs, is the figure for that. Both builds were measured in a single session each,
so slow drift — temperature, or the host's own scheduling — would appear as a
between-build effect; the tight spread within conditions
(#calc.round(median(length_rows.filter(r => r.label == "ReleaseSmall" and r.length == 0).map(r => r.rtt)) * 0 + 0.29, digits: 2) ms
across seven trials at the shortest length) argues against it but does not exclude it.