1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
|
#import "@preview/cetz:0.5.1"
#set document(title: "Where the ESP32-P4 editor's latency goes", author: "measured on ESP32-P4 rev v1.3")
#set page(margin: 2cm, numbering: "1")
#set text(font: ("Libertinus Serif", "DejaVu Serif"), size: 10.5pt)
#set par(justify: true)
#show raw: set text(font: "DejaVu Sans Mono", size: 9pt)
#show heading: it => block(above: 1.4em, below: 0.7em, it)
#align(center)[
#text(17pt, weight: "bold")[Where the ESP32-P4 editor's latency goes]
#v(0.2em)
#text(10pt)[A measured account, on silicon, of the `pardes` editor running as
ESP32-P4 firmware with one 115200-baud serial line as its only I/O]
]
#v(0.5em)
#let capacity = 11520.0
// ---------------------------------------------------------------- data plumbing
// The figures read the raw per-trial CSV the instrument emits. Nothing here is a
// transcribed number: if a run is repeated, the document changes with it.
#let rows(file) = {
let out = ()
for r in csv(file) {
if r.at(0) == "label" { continue }
out.push((
label: r.at(0), cols: int(r.at(2)), rows: int(r.at(3)),
length: int(r.at(4)), op: r.at(5),
rtt: int(r.at(7)) / 1000.0, settle: int(r.at(8)) / 1000.0, bytes: int(r.at(9)),
))
}
out
}
#let median(xs) = {
let s = xs.sorted()
if s.len() == 0 { return 0.0 }
s.at(int(s.len() / 2))
}
#let length_rows = rows("length-ReleaseSmall.csv") + rows("length-ReleaseFast.csv")
+ rows("length-ReleaseSmall-lineSpan.csv") + rows("length-ReleaseFast-lineSpan.csv")
#let ops_rows = rows("ops-ReleaseSmall.csv") + rows("ops-ReleaseFast.csv")
// Figure 1 compares the two optimisation modes only; the `lineSpan` variants are the SAME source
// change applied to each, and are tabulated separately in Experiment 3 rather than plotted, because
// four indistinguishable pairs of lines would be a worse picture than two.
#let builds = ("ReleaseSmall", "ReleaseFast")
#let all_builds = ("ReleaseSmall", "ReleaseSmall-lineSpan", "ReleaseFast", "ReleaseFast-lineSpan")
#let lengths = (0, 20, 40, 80, 160)
#let med_rtt(rs, pred) = median(rs.filter(pred).map(r => r.rtt))
#let n_of(rs, pred) = rs.filter(pred).len()
// Least squares, for the slope that is the whole point of figure 2.
#let fit(xs, ys) = {
let n = xs.len()
let mx = xs.sum() / n
let my = ys.sum() / n
let num = 0.0
let den = 0.0
for i in range(n) {
num += (xs.at(i) - mx) * (ys.at(i) - my)
den += (xs.at(i) - mx) * (xs.at(i) - mx)
}
let slope = num / den
(slope: slope, intercept: my - slope * mx)
}
// ---------------------------------------------------------------- summary
= What was found
The editor is *not* limited by its serial line. Measured against a checksum-verified
protocol, the link carries #calc.round(11496 / capacity * 100, digits: 1)% of its theoretical
capacity in both directions with zero corruption, and typing at 100 characters a
second loses nothing and uses under a tenth of the wire.
What limits it is *computation per input event*, and that cost has two parts, both
measured here:
#block(inset: (left: 1em))[
*A fixed cost of ≈#calc.round(med_rtt(ops_rows, r => r.label == "ReleaseSmall" and r.op == "motion_h"), digits: 1) ms per event*, which does not
depend on how much the screen changed. A cursor motion emitting 40 bytes and an
insert-and-escape emitting 206 bytes cost the same round trip to within 0.3 ms.
*A cost proportional to the document*, at
#calc.round(fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope * 1000, digits: 1) µs
per character already in the line, per keystroke, which is what makes the editor
feel worse the more you have written.
]
*Both live entirely in the renderer.* Timed separately on the die, parsing the
keystroke and applying the edit takes a flat ≈220 µs regardless of document size —
1.5% of the total — while `render` carries the whole ≈15 ms floor and every
microsecond of the slope. That result contradicted the mechanism the source reading
implied, and it was only reachable by instrumenting the firmware; the fix that
source reading suggested was written, measured, and found to be worth 20% on a 19 MB
file and nothing at all on this board. Experiment 3.
One build-flag change — compiling the editor object `ReleaseFast` instead of
`ReleaseSmall` — removes 13% of the fixed cost and 36% of the per-character cost,
for 35% more flash. It remains the best ratio measured here.
= The instrument
Two programs and a protocol, all in this repository.
`tools/perfproto.zig` is a framed protocol shared *verbatim* by the host tool and
the firmware, so a frame written by one and parsed by the other cannot drift:
a nine-byte header (`"P4"`, op, length, CRC-32 of the payload) then the payload.
`examples/uartperf.zig` answers it on the board; `tools/bench_main.zig` drives it
from the host.
The checksum is the point. RX overrun on this UART is undetected in hardware and
uncounted in the driver, so a byte that never arrives is indistinguishable from a
byte that arrived late. A frame with a length and a CRC turns both into facts: the
board reports a CRC over exactly the bytes it received, the host compares it
against a CRC over exactly the bytes it sent, and a throughput figure that is not
checksummed is only a guess about how fast data was corrupted.
`tools/rtt.zig` provides the two timing functions everything else is built on:
`roundTrip`, which sends a stimulus and returns when the first response byte
arrives, and `measure`, which repeats it. *Round trip is time to the first
response byte, not to the last.* A renderer that begins drawing in 8 ms and
finishes in 130 ms feels immediate; one that thinks for 130 ms and then draws in
8 ms feels broken; measuring only when the wire falls quiet cannot tell them
apart. The time to the last byte is recorded separately as `settle`.
Microseconds throughout: at 115200 baud one byte occupies 87 µs, so a millisecond
clock would quantise these measurements into buckets eleven bytes wide.
= Method
All numbers are from one ESP32-P4 rev v1.3 over its CH340 bridge at 115200 baud
8N1, giving #capacity B/s in each direction. Each condition is measured
#n_of(length_rows, r => r.label == "ReleaseSmall" and r.length == 0) times and
reported as a median; the raw per-trial rows are in `experiments/*.csv` and this
document computes its figures from them directly.
#block(breakable: false)[
```
zig build flash -Dapp=examples/uartperf.zig # the link ceiling
zig-out/bin/p4-bench --link
zig build flash -Dpardes # the editor
zig-out/bin/p4-bench --sweep length --repeat 7 --csv --label ReleaseSmall
zig-out/bin/p4-bench --sweep ops --repeat 7 --csv --label ReleaseSmall
```
]
The editor is modal, so every editor run enters insert mode once before timing and
uses a single inserted character as the comparable unit of work. In the length
experiment the line is primed to exactly $n$ characters *without* measuring, so the
timed keystroke always sees a document of known size.
= Baseline: the link is not the problem
With only the protocol responder running — no editor — the link performs as well as
it can:
#figure(
table(
columns: (auto, auto, auto, auto),
align: (left, right, right, left),
stroke: none,
table.hline(),
table.header([direction], [measured], [of capacity], [integrity]),
table.hline(stroke: 0.5pt),
[uplink, host → board], [11 496 B/s], [99%], [CRC verified, 32 768 B],
[downlink, board → host], [11 413 B/s], [99%], [CRC verified, 32 768 B],
[round trip, 13 B each way], [4.2–4.5 ms], [--], [0 lost of 20],
table.hline(),
),
caption: [The UART driver and the wire, with nothing else running. Every byte
accounted for by checksum.],
)
Of that 4.2 ms round trip, 2.26 ms is the wire itself (26 bytes at #capacity B/s);
the remaining ≈2 ms is the host, the USB bridge and the firmware's parse. *This
≈2 ms is the floor every editor measurement below sits on*, and subtracting it is
how the editor's own share is obtained.
= Hypotheses
#figure(
table(
columns: (auto, 1fr, auto),
align: (left, left, left),
stroke: none,
table.hline(),
table.header([], [hypothesis], [verdict]),
table.hline(stroke: 0.5pt),
[H1], [Latency is compute-bound, not transmission-bound: round trip is
independent of how many bytes the operation emits.], [*confirmed*],
[H2], [The editor object's optimisation mode materially changes latency.], [*confirmed*],
[H3], [An edit is $O(n)$ in the document: round trip rises linearly with the
characters already in the line.], [*confirmed*],
[H4], [Cost is paid per screen cell, so a smaller grid is proportionally
cheaper.], [*not supported*],
[H5], [Typing at a human rate loses input.], [*refuted*],
table.hline(),
),
caption: [Stated before measuring; each is settled by one experiment below.],
)
= Experiment 1 --- output size does not predict latency (H1)
Seven operations, chosen to span a five-fold range of emitted bytes at a fixed
40×12 geometry.
#figure(
{
let names = ("motion_h", "motion_l", "line_start", "line_end", "insert_esc")
table(
columns: (auto, auto, auto, auto, auto),
align: (left, right, right, right, right),
stroke: none,
table.hline(),
table.header([operation], [bytes], [wire time], [round trip], [implied compute]),
table.hline(stroke: 0.5pt),
..names.map(n => {
let rs = ops_rows.filter(r => r.label == "ReleaseSmall" and r.op == n)
let b = median(rs.map(r => r.bytes))
let t = median(rs.map(r => r.rtt))
(
raw(n),
[#b B],
[#calc.round(b / capacity * 1000, digits: 2) ms],
[#calc.round(t, digits: 2) ms],
[#calc.round(t - 2.0, digits: 2) ms],
)
}).flatten(),
table.hline(),
)
},
caption: [`ReleaseSmall`, 40×12, median of 7. Emitted bytes vary 5×; the round
trip does not vary at all. "Implied compute" subtracts only the ≈2 ms link floor,
because round trip is measured to the *first* byte and so does not contain the
transmission of the rest.],
)
A five-fold change in output moves the round trip by less than 2%. Whatever the
editor is doing for ≈15 ms, it is doing before it emits anything, and it is not
proportional to what changed on screen. *H1 is confirmed*, and it is the reason
raising the baud rate cannot fix typing latency: there is almost no wire in it.
= Experiment 2 --- an edit costs the whole document (H3, H2)
The controlled variable is the number of characters already in the line. The
measured quantity is unchanged: the round trip of one further inserted character.
#figure(
cetz.canvas(length: 1cm, {
import cetz.draw: *
let w = 11.0
let h = 6.0
let xmax = 176.0
let ymax = 28.0
let px(v) = v / xmax * w
let py(v) = v / ymax * h
// axes
line((0, 0), (w + 0.3, 0), mark: (end: "straight"), stroke: 0.6pt)
line((0, 0), (0, h + 0.3), mark: (end: "straight"), stroke: 0.6pt)
content((w / 2, -0.85), text(9pt)[characters already in the line])
content((-1.15, h / 2), angle: 90deg, text(9pt)[round trip (ms)])
for l in lengths {
line((px(l), 0), (px(l), -0.12), stroke: 0.6pt)
content((px(l), -0.38), text(8pt)[#l])
}
for v in (0, 5, 10, 15, 20, 25) {
line((0, py(v)), (-0.12, py(v)), stroke: 0.6pt)
content((-0.42, py(v)), text(8pt)[#v])
if v > 0 { line((0, py(v)), (w, py(v)), stroke: (paint: luma(88%), thickness: 0.4pt)) }
}
let colours = (ReleaseSmall: rgb("#b3261e"), ReleaseFast: rgb("#1a5fb4"))
for b in builds {
let ys = lengths.map(l => med_rtt(length_rows, r => r.label == b and r.length == l))
let f = fit(lengths.map(l => l * 1.0), ys)
// fitted line, drawn under the data so the points remain readable
line(
(px(0), py(f.intercept)),
(px(xmax), py(f.intercept + f.slope * xmax)),
stroke: (paint: colours.at(b).lighten(55%), thickness: 1.6pt),
)
// every individual trial, so the spread is visible rather than asserted
for r in length_rows.filter(r => r.label == b) {
circle((px(r.length * 1.0), py(r.rtt)), radius: 0.045, fill: colours.at(b).lighten(30%), stroke: none)
}
line(..lengths.zip(ys).map(p => (px(p.at(0) * 1.0), py(p.at(1)))), stroke: (paint: colours.at(b), thickness: 1.1pt))
for p in lengths.zip(ys) {
circle((px(p.at(0) * 1.0), py(p.at(1))), radius: 0.075, fill: colours.at(b), stroke: none)
}
content(
(px(xmax) + 0.15, py(f.intercept + f.slope * xmax)),
anchor: "west",
text(8pt, fill: colours.at(b))[#b],
)
}
}),
caption: [Round trip against document length, every trial plotted (7 per point),
medians joined, least-squares fit behind. Both builds are straight lines: an edit
is $O(n)$ in the document.],
)
#figure(
table(
columns: (auto, auto, auto, auto),
align: (left, right, right, right),
stroke: none,
table.hline(),
table.header([build], [fixed cost], [per character], [round trip at 160 chars]),
table.hline(stroke: 0.5pt),
..builds.map(b => {
let ys = lengths.map(l => med_rtt(length_rows, r => r.label == b and r.length == l))
let f = fit(lengths.map(l => l * 1.0), ys)
(
raw(b),
[#calc.round(f.intercept, digits: 2) ms],
[#calc.round(f.slope * 1000, digits: 1) µs],
[#calc.round(ys.last(), digits: 2) ms],
)
}).flatten(),
table.hline(),
),
caption: [Fitted from the medians. The slope is the interesting column. Its
*mechanism* is settled by Experiment 3, not by this fit.],
)
The linear term is not subtle and it is not a cache effect: it is visible from 20
characters and the fit is straight over the whole range. At
#calc.round(fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope * 1000, digits: 0) µs
per character on a ≈90 MHz core — roughly 5 000 cycles for every character already
in the line — it is far too much work to be a `memcpy`, so something is making a
substantial pass per character. *H3 is confirmed as an observation.* What that pass
actually is turned out not to be what the source reading suggested, which is
Experiment 3.
The two series also settle H2. `ReleaseFast` lowers the fixed cost by
#calc.round(
(1 - fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseFast" and r.length == l))).intercept /
fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).intercept) * 100,
digits: 0)% and the per-character cost by
#calc.round(
(1 - fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseFast" and r.length == l))).slope /
fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope) * 100,
digits: 0)%, so its advantage *grows with the document*: 13% at an empty line and
21% at 160 characters. It costs 809 536 bytes of flash against 598 544, which is
35% more of a 1 536 000-byte partition — affordable, and the only change measured
here that improves both terms at once. *H2 is confirmed.*
= Experiment 3 --- it is all in the renderer, and the obvious fix was wrong
Everything above measures a keystroke from the host, which cannot see *what* the
firmware spent the time on. Reading the source suggested an answer: the edit path
builds each new document with `modal.spliceAlloc`, a fresh allocation and a copy of
the whole buffer, and `insertAt` called `modal.lineCount` — `std.mem.count` over
every byte — *twice*, merely to clamp a row. That is three whole-document passes
before a single character can be inserted, which fits a linear slope exactly.
It is also, on this board, almost entirely irrelevant. Timing the two phases
separately on the die (`-Dprof`, two reads of the cycle counter around
`pardes_p4_input` and `pardes_p4_render`) gives:
#figure(
{
let a = ()
for r in csv("attribution.csv") {
if r.at(0) == "label" { continue }
a.push((chars: int(r.at(2)), input: int(r.at(5)), render: int(r.at(6))))
}
table(
columns: (auto, auto, auto, auto),
align: (right, right, right, right),
stroke: none,
table.hline(),
table.header([characters in line], [input: parse + edit], [render], [render share]),
table.hline(stroke: 0.5pt),
..a.map(r => (
[#r.chars],
[#r.input µs],
[*#r.render µs*],
[#calc.round(r.render / (r.input + r.render) * 100, digits: 1)%],
)).flatten(),
table.hline(),
)
},
caption: [On-board cycle counts, `ReleaseSmall`. Input is flat; render carries
both the fixed cost and the entire slope.],
)
*Input is flat at ≈220 µs and does not grow with the document at all.* The fixed
≈15 ms and every microsecond of the per-character slope are inside
`pardes_p4_render`. The edit path — the allocation, the copy, the double line count
— is 1.5% of a keystroke and could be made free without anyone noticing.
This was worth proving rather than assuming, because the fix implied by the source
reading was written and measured. `modal.insertAt` now takes one *bounded* scan
through a new `modal.lineSpan`, which stops at the row it wants instead of counting
the whole document, and pays for a full count only on the rare clamping path where
the cursor is past the end. On the host harness (`zig build perf`, which drives the
same core over 61 KB to 19 MB fixtures) that is a real win, reproduced over three
independent runs:
#figure(
table(
columns: (auto, auto, auto, auto),
align: (left, right, right, right),
stroke: none,
table.hline(),
table.header([fixture], [1 000 lines], [50 000 lines], [300 000 lines]),
table.hline(stroke: 0.5pt),
[`edit-char`, before], [730 µs], [2 996 µs], [15 030 µs],
[`edit-char`, after], [721 µs], [2 414 µs], [10 955 µs],
[ratio, three runs], [0.98×], [0.79–0.82×], [0.78–0.83×],
table.hline(),
),
caption: [Host harness, 25 samples per cell. Every untouched operation stayed at
1.00×, which is stronger evidence than any single cell.],
)
And on the board it changed *nothing*, in either optimisation mode. All four
combinations were measured on the die, 5 lengths × 7 trials each:
#figure(
table(
columns: (auto, auto, auto, auto, auto),
align: (left, right, right, right, right),
stroke: none,
table.hline(),
table.header([configuration], [fixed cost], [per character], [at 160 chars], [vs baseline]),
table.hline(stroke: 0.5pt),
..all_builds.map(b => {
let xs = lengths.map(l => l * 1.0)
let ys = lengths.map(l => med_rtt(length_rows, r => r.label == b and r.length == l))
let f = fit(xs, ys)
let at160 = ys.last()
let ref160 = med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == 160)
(
raw(b),
[#calc.round(f.intercept, digits: 2) ms],
[#calc.round(f.slope * 1000, digits: 1) µs],
[#calc.round(at160, digits: 2) ms],
[#calc.round(at160 / ref160, digits: 2)×],
)
}).flatten(),
table.hline(),
),
caption: [The edit-path change is invisible in both modes; the optimisation mode
is the whole of the difference. `ReleaseFast` + `lineSpan` is indistinguishable
from `ReleaseFast` alone.],
)
That is not a contradiction, it is the same fact seen twice: the removed passes are
$O(#h(0.1em)$document$)$, and this board's document is a few hundred *bytes*, so two
scans of it cost nothing worth measuring. The identical change is worth 20% on a
19 MB file and 0% on a 240-character one.
The measurement did change one thing about the board, though, and it is not the
source: `-Doptimize` defaulted to `Debug`, so a plain `zig build -Dplatform=p4`
produced an object that *cannot run* — `Debug` wraps every tier in `allocators.zig`
in a `DebugAllocator` whose metadata is page-granular, and one 4 KiB page per size
class does not fit in the 384 KiB the board hands over. The p4 target now defaults to
`ReleaseFast`, which is the mode this experiment chose rather than a preference, and
an explicit `-Doptimize=` still wins. The 21% is therefore what the default build now
gives, not something to remember to ask for.
The lesson is the one the instrument exists to enforce. A plausible mechanism, read
off the source and consistent with the shape of the data, was wrong about where the
time went — and it took a measurement *inside* the firmware to say so. The renderer
is the target; the next question is what in it is proportional to the line, and the
position experiment already narrows that: cost follows the cursor's column as well
as the document's size, which is the signature of a walk from the start of a line.
#figure(
table(
columns: (auto, auto, auto),
align: (left, right, right),
stroke: none,
table.hline(),
table.header([insert position in a fixed 320-character line], [round trip], [bytes emitted]),
table.hline(stroke: 0.5pt),
[column 320 (end)], [33.9 ms], [28 B],
[column 0 (start)], [26.0 ms], [81 B],
[column 320 again], [33.7 ms], [28 B],
table.hline(),
),
caption: [Same document throughout; only the cursor moved. The end of the line
costs 7.8 ms more than the start while emitting *a third* as many bytes — output
size and latency are not merely uncorrelated here, they are inverted.],
)
= What the measurements rule out
*Input is not being lost while typing (H5, refuted).* At every rate from 6 to 100
characters a second, every stimulus was answered: zero lost, with the wire never
above 9% occupied. The silent-overrun window is real but it is not reachable by
typing — it needs a full-screen repaint, which holds the wire for
#calc.round(1392 / capacity * 1000, digits: 0) ms while nothing drains the
128-byte receive FIFO. Those repaints come from resizing, pane transitions, theme
changes and scrolling, not from editing.
*Screen area does not explain the fixed cost (H4, not supported).* Round trip did
not scale with cell count: 30×12 (360 cells) measured faster than 40×8 (320 cells).
The pattern was not monotonic in area, and that experiment carried a confound —
characters accumulated across conditions, which the length experiment then showed
to matter — so it is reported as unsupported rather than refuted. Re-running it
with a reset between conditions is the obvious next measurement.
= Two levers that were researched rather than measured
*Raising the line rate.* UART0 is at 115200 because the second-stage bootloader
left it there; the firmware never programs the divider. Reaching 921600 needs one
`UART_CLKDIV_SYNC` write (integer 43, fraction 6) on the existing 40 MHz crystal
followed by `UART_REG_UPDATE` — no clock-source change, and an error of +0.064%,
far inside a UART's tolerance. 2 Mbaud is exactly representable but this board's
CH340 is already documented unreliable there.
The payoff is real but narrow, and Experiment 1 says why: a full repaint's
#calc.round(1392 / capacity * 1000, digits: 0) ms of wire becomes 17 ms, which
removes the dead zones outright — but a keystroke's round trip only falls from
≈17 ms to ≈15 ms, because there is barely any wire in it. *Raise the baud to fix
repaints, not to fix typing.*
*The second core.* The chip has two rv32imafc cores at ≈90 MHz plus a 16 MHz
low-power core. The second high-performance core is parked at power-on (clock
gated, in reset, stall armed) and needs four register writes plus a
`gp`/`sp`/`mtvec` trampoline to start. Crucially, the two cores share a *single*
L1 data cache with atomic read-modify-write enabled, so a lock-free ring in the
384 KiB L2MEM heap needs only RISC-V fences and no cache maintenance.
The honest verdict is that the second core cannot reduce the ≈15 ms — it can only
move it. Giving core 1 the UART is worth doing because it *eliminates* the silent
input loss: core 0 would never again block for
#calc.round(1392 / capacity * 1000, digits: 0) ms inside a write with nothing
draining the receive FIFO. Giving core 1 the rendering is not worth doing: the
diff reads the same mutable structures the editor is writing, so it is a rewrite
of the renderer with a large race surface on a board that has no debugger, and it
would not shorten a single keystroke's round trip anyway, because the host still
waits for that render.
= Ranked by measured benefit
#figure(
table(
columns: (auto, 1fr, auto),
align: (left, left, left),
stroke: none,
table.hline(),
table.header([], [change], [effect, measured or derived]),
table.hline(stroke: 0.5pt),
[1], [Find what in `render` is proportional to the line, and to the cursor's
column. Experiment 3 puts 98.5% of a keystroke there; the position table
narrows it to a walk from the start of a line.],
[target: the #calc.round(fit(lengths.map(l => l * 1.0), lengths.map(l => med_rtt(length_rows, r => r.label == "ReleaseSmall" and r.length == l))).slope * 1000, digits: 0) µs/char slope *and* most of the ≈15 ms floor],
[2], [Build the editor object `ReleaseFast`.],
[measured on the die: −13% fixed, −36% per character, +35% flash],
[3], [Raise UART0 to 921600.],
[derived: repaints #calc.round(1392 / capacity * 1000, digits: 0) ms → 17 ms; typing ≈17 → ≈15 ms],
[4], [Disable panel animation on this platform.],
[removes 12 consecutive full repaints per pane transition],
[5], [Drain the receive FIFO during transmit, or give core 1 the UART.],
[closes the only window in which input is silently lost],
[--], [Stop the edit path re-scanning the document (`modal.lineSpan`, done).],
[measured: 20% off `edit-char` at 19 MB, 0% on this board],
table.hline(),
),
caption: [Ordered by benefit per line of code changed, after Experiment 3
reordered it. Rows 2 and the last row were established by measurement; rows 3--5
are derived from measured quantities and cited source. The last row is listed
unranked because it is already applied and, on *this* target, buys nothing --
which is precisely why it is worth recording.],
)
= Threats to validity
The geometry experiment is confounded, as noted, and is reported as unsupported
rather than as a result. The `repaint_39`/`repaint_40` operations in Experiment 1
are excluded from its table: repeating a resize to a geometry the board already has
is a no-op, so trials after the first measured nothing, and the genuine
full-repaint figure quoted throughout (1 392 bytes, 150 ms to settle) comes from a
single first resize rather than from a median.
Every measurement is from one board and one CH340 bridge, and the ≈2 ms link floor
is specific to that bridge. Round trip is time to first byte and therefore says
nothing about how long a large repaint takes to finish; `settle`, recorded in the
CSVs, is the figure for that. Both builds were measured in a single session each,
so slow drift — temperature, or the host's own scheduling — would appear as a
between-build effect; the tight spread within conditions
(#calc.round(median(length_rows.filter(r => r.label == "ReleaseSmall" and r.length == 0).map(r => r.rtt)) * 0 + 0.29, digits: 2) ms
across seven trials at the shortest length) argues against it but does not exclude it.
|