1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
1525
1526
1527
1528
1529
1530
1531
1532
1533
1534
1535
1536
1537
1538
1539
1540
1541
1542
1543
1544
1545
1546
1547
1548
1549
1550
1551
1552
1553
1554
1555
1556
1557
1558
1559
1560
1561
1562
1563
1564
1565
1566
1567
1568
1569
1570
1571
1572
1573
1574
1575
1576
1577
1578
1579
1580
1581
1582
1583
1584
1585
1586
1587
1588
1589
1590
1591
1592
1593
1594
1595
1596
1597
1598
1599
1600
1601
1602
1603
1604
1605
1606
1607
1608
1609
1610
1611
1612
1613
1614
1615
1616
1617
1618
1619
1620
1621
1622
1623
1624
1625
1626
1627
1628
1629
1630
1631
1632
1633
1634
1635
1636
1637
1638
1639
1640
1641
1642
1643
1644
1645
1646
1647
1648
1649
1650
1651
1652
1653
1654
1655
1656
1657
1658
1659
1660
1661
1662
1663
1664
1665
1666
1667
1668
1669
1670
1671
1672
1673
1674
1675
1676
1677
1678
1679
1680
1681
1682
1683
1684
1685
1686
1687
1688
1689
1690
1691
1692
1693
1694
1695
1696
1697
1698
1699
1700
1701
1702
1703
1704
1705
1706
1707
1708
1709
1710
1711
1712
1713
1714
1715
1716
1717
1718
1719
1720
1721
1722
1723
1724
1725
1726
1727
1728
1729
1730
1731
1732
1733
1734
1735
1736
1737
1738
1739
1740
1741
1742
1743
1744
1745
1746
1747
1748
1749
1750
1751
1752
1753
1754
1755
1756
1757
1758
1759
1760
1761
1762
1763
1764
1765
1766
1767
1768
1769
1770
1771
1772
1773
1774
1775
1776
1777
1778
1779
1780
1781
1782
1783
1784
1785
1786
1787
1788
1789
1790
1791
1792
1793
1794
1795
1796
1797
1798
1799
1800
1801
1802
1803
1804
1805
1806
1807
1808
1809
1810
1811
1812
1813
1814
1815
1816
1817
1818
1819
1820
1821
1822
1823
1824
1825
1826
1827
1828
1829
1830
1831
1832
1833
1834
1835
1836
1837
1838
1839
1840
1841
1842
1843
1844
1845
1846
1847
1848
1849
1850
1851
1852
1853
1854
1855
1856
1857
1858
1859
1860
1861
1862
1863
1864
1865
1866
1867
1868
1869
1870
1871
1872
1873
1874
1875
1876
1877
1878
1879
1880
1881
1882
1883
1884
1885
1886
1887
1888
1889
1890
1891
1892
1893
1894
1895
1896
1897
1898
1899
1900
1901
1902
1903
1904
1905
1906
1907
1908
1909
1910
1911
1912
1913
1914
1915
1916
1917
1918
1919
1920
1921
1922
1923
1924
1925
1926
1927
1928
1929
1930
1931
1932
1933
1934
1935
1936
1937
1938
1939
1940
1941
1942
1943
1944
1945
1946
1947
1948
1949
1950
1951
1952
1953
1954
1955
1956
1957
1958
1959
1960
1961
1962
1963
1964
1965
1966
1967
1968
1969
1970
1971
1972
1973
1974
1975
1976
1977
1978
1979
1980
1981
1982
1983
1984
1985
1986
1987
1988
1989
1990
1991
1992
1993
1994
1995
1996
1997
1998
1999
2000
2001
2002
2003
2004
2005
2006
2007
2008
2009
2010
2011
2012
2013
2014
2015
2016
2017
2018
2019
2020
2021
2022
2023
2024
2025
2026
2027
2028
2029
2030
2031
2032
2033
2034
2035
2036
2037
2038
2039
2040
2041
2042
2043
2044
2045
2046
2047
2048
2049
2050
2051
2052
2053
2054
2055
2056
2057
2058
2059
2060
2061
2062
2063
2064
2065
2066
2067
2068
2069
2070
2071
2072
2073
2074
2075
2076
2077
2078
2079
2080
2081
2082
2083
2084
2085
2086
2087
2088
2089
2090
2091
2092
2093
2094
2095
2096
2097
2098
2099
2100
2101
2102
2103
2104
2105
2106
2107
2108
2109
2110
2111
2112
2113
2114
2115
2116
2117
2118
2119
2120
2121
2122
2123
2124
2125
2126
2127
2128
2129
2130
2131
2132
2133
2134
2135
2136
2137
2138
2139
2140
2141
2142
2143
2144
2145
2146
2147
2148
2149
2150
2151
2152
2153
2154
2155
2156
2157
2158
2159
2160
2161
2162
2163
2164
2165
2166
2167
2168
2169
2170
2171
2172
2173
2174
2175
2176
2177
2178
2179
2180
2181
2182
2183
2184
2185
2186
2187
2188
2189
2190
2191
2192
2193
2194
2195
2196
2197
2198
2199
2200
2201
2202
2203
2204
2205
2206
2207
2208
2209
2210
2211
2212
2213
2214
2215
2216
2217
2218
2219
2220
2221
2222
2223
2224
2225
2226
2227
2228
2229
2230
2231
2232
2233
2234
2235
2236
2237
2238
2239
2240
2241
2242
2243
2244
2245
2246
2247
2248
2249
2250
2251
2252
2253
2254
2255
2256
2257
2258
2259
2260
2261
2262
2263
2264
2265
2266
2267
2268
2269
2270
2271
2272
2273
2274
2275
2276
2277
2278
2279
2280
2281
2282
2283
2284
2285
2286
2287
2288
2289
2290
2291
2292
2293
2294
2295
2296
2297
2298
2299
2300
2301
2302
2303
2304
2305
2306
2307
2308
2309
2310
2311
2312
2313
2314
2315
2316
2317
2318
2319
2320
2321
2322
2323
2324
2325
2326
2327
2328
2329
2330
2331
2332
2333
2334
2335
2336
2337
2338
2339
2340
2341
2342
2343
2344
2345
2346
2347
2348
2349
2350
2351
2352
2353
2354
2355
2356
2357
2358
2359
2360
2361
2362
2363
2364
2365
2366
2367
2368
2369
2370
2371
2372
2373
2374
2375
2376
2377
2378
2379
2380
2381
2382
2383
2384
2385
2386
2387
2388
2389
2390
2391
2392
2393
2394
2395
2396
2397
2398
2399
2400
2401
2402
2403
2404
2405
2406
2407
2408
2409
2410
2411
2412
2413
2414
2415
2416
2417
2418
2419
2420
2421
2422
2423
2424
2425
2426
2427
2428
2429
2430
2431
2432
2433
2434
2435
2436
2437
2438
2439
2440
2441
2442
2443
2444
2445
2446
2447
2448
2449
2450
2451
2452
2453
2454
2455
2456
2457
2458
2459
2460
2461
2462
2463
2464
2465
2466
2467
2468
2469
2470
2471
2472
2473
2474
2475
2476
2477
2478
2479
2480
2481
2482
2483
2484
2485
2486
2487
2488
2489
2490
2491
2492
2493
2494
2495
2496
2497
2498
2499
2500
2501
2502
2503
2504
2505
2506
2507
2508
2509
2510
2511
2512
2513
2514
2515
2516
2517
2518
2519
2520
2521
2522
2523
2524
2525
2526
2527
2528
2529
2530
2531
2532
2533
2534
2535
2536
2537
2538
2539
2540
2541
2542
2543
2544
2545
2546
2547
2548
2549
2550
2551
2552
2553
2554
2555
2556
2557
2558
2559
2560
2561
2562
2563
2564
2565
2566
2567
2568
2569
2570
2571
2572
2573
2574
2575
2576
2577
2578
2579
2580
2581
2582
2583
2584
2585
2586
2587
2588
2589
2590
2591
2592
2593
2594
2595
2596
2597
2598
2599
2600
2601
2602
2603
2604
2605
2606
2607
2608
2609
2610
2611
2612
2613
2614
2615
2616
2617
2618
2619
2620
2621
2622
2623
2624
2625
2626
2627
2628
2629
2630
2631
2632
2633
2634
2635
2636
2637
2638
2639
2640
2641
2642
2643
2644
2645
2646
2647
2648
2649
2650
2651
2652
2653
2654
2655
2656
2657
2658
2659
2660
2661
2662
2663
2664
2665
2666
2667
2668
2669
2670
2671
2672
2673
2674
2675
2676
2677
2678
2679
2680
2681
2682
2683
2684
2685
2686
2687
2688
2689
2690
2691
2692
2693
2694
2695
2696
2697
2698
2699
2700
2701
2702
2703
2704
2705
2706
2707
2708
2709
2710
2711
2712
2713
2714
2715
2716
2717
2718
2719
2720
2721
2722
2723
2724
2725
2726
2727
2728
2729
2730
2731
2732
2733
2734
2735
2736
2737
2738
2739
2740
2741
2742
2743
2744
2745
2746
2747
2748
2749
2750
2751
2752
2753
2754
2755
2756
2757
2758
2759
2760
2761
2762
2763
2764
2765
2766
2767
2768
2769
2770
2771
2772
2773
2774
2775
2776
2777
2778
2779
2780
2781
2782
2783
2784
2785
2786
2787
2788
2789
2790
2791
2792
2793
2794
2795
2796
2797
2798
2799
2800
2801
2802
2803
2804
2805
2806
2807
2808
2809
2810
2811
2812
2813
2814
2815
2816
2817
2818
2819
2820
2821
2822
2823
2824
2825
2826
2827
2828
2829
2830
2831
2832
2833
2834
2835
2836
2837
2838
2839
2840
2841
2842
2843
2844
2845
2846
2847
2848
2849
2850
2851
2852
2853
2854
2855
2856
2857
2858
2859
2860
2861
2862
2863
2864
2865
2866
2867
2868
2869
2870
2871
2872
2873
2874
2875
2876
2877
2878
2879
2880
2881
2882
2883
2884
2885
2886
2887
2888
2889
2890
2891
2892
2893
2894
2895
2896
2897
2898
2899
2900
2901
2902
2903
|
//! A minimal IPv4 stack: Ethernet, ARP, IPv4, ICMP echo, UDP, a DHCP client, one TCP client and
//! HTTP GET. This is what replaces lwIP.
//!
//! Two functions drive everything and nothing else touches the outside world:
//!
//! stack.onFrame(frame) a received Ethernet frame, headers and all
//! stack.tick(now_ms) time passing, in milliseconds, from anywhere the caller likes
//!
//! and one callback carries frames out (`send`, supplied to `init`). There is no `std.Io`, no
//! allocator, no clock read and no hidden thread. That is not minimalism for its own sake: it is
//! what makes the whole stack testable on the host, where a "network" is a test function that hands
//! `onFrame` bytes it wrote by hand and reads back whatever `send` was given. Every protocol
//! behaviour in this file is exercised that way in `ip_test.zig`, including retransmission - which
//! on a real timer would be a flaky test and here is two calls to `tick`.
//!
//! **Everything is statically sized.** `Stack` is one struct with fixed buffers inside it; there is
//! no allocator, not even a `FixedBufferAllocator`, because nothing here has a lifetime that an
//! arena would model better than a field does. `@sizeOf(Stack)` is asserted at compile time below
//! (`footprint`) so the number cannot drift silently against the ~128 KB of L2MEM the image has.
//!
//! Wire formats are matched field by field against the lwIP this replaces, and every one is cited:
//! ESP-IDF v6.0.2 carries lwIP at `components/lwip/lwip/src/`, and the packed structs in
//! `include/lwip/prot/*.h` are the reference for offsets, and its `.c` files for behaviour. Where
//! this stack deliberately differs from lwIP, the comment says so and why.
//!
//! ## What this is not
//!
//! * No IPv6, no TCP listen/accept, no IP fragmentation or reassembly, no TLS. Out of scope.
//! * No congestion control. TCP sends at most one unacknowledged segment at a time (see `Tcp`),
//! which is a fixed window of one and therefore needs no congestion window, no slow start and
//! no fast recovery. It is also slow. For an HTTP GET of a few kilobytes over Wi-Fi that is the
//! right trade; for bulk transfer it is not, and nothing here pretends otherwise.
//! * No VLAN tags, no 802.1Q. A tagged frame is dropped as an unknown ethertype.
//! * No transfer coding but `identity` and `chunked`. Anything else - `gzip`, `deflate`, a
//! stack of them - is rejected with `error.UnsupportedTransferEncoding` rather than handed
//! back with its framing bytes still in it.
//! * DNS resolves A records only, one query at a time, over the UDP already here, with no
//! cache. `resolve` follows `httpGet`'s protocol exactly: start, `error.WouldBlock`, the
//! caller drives `tick`/`onFrame`, call again with the same name.
const std = @import("std");
const assert = std.debug.assert;
// =============================================================================== sizing
//
// The whole static footprint, in one place. Every buffer in `Stack` is one of these.
/// Ethernet MTU: the largest IP datagram that fits in one frame.
pub const mtu: usize = 1500;
/// Ethernet header: 6 destination + 6 source + 2 ethertype. lwIP `prot/ethernet.h:89`
/// (`SIZEOF_ETH_HDR`, with its optional `ETH_PAD_SIZE` at zero).
pub const eth_hlen: usize = 14;
/// The largest frame this stack will build or accept, excluding the FCS the MAC appends.
pub const frame_max: usize = eth_hlen + mtu;
/// ARP cache entries. Four is enough for the gateway, one peer, and two strangers, which is the
/// whole population a single-connection HTTP client on a home /24 ever needs to address.
pub const arp_cache_len: usize = 4;
/// Bytes of HTTP response head (status line plus headers) that may be buffered while waiting for
/// the blank line. Exceeding this fails the request rather than truncating silently.
///
/// 2048, raised from 1024 against a measurement rather than a guess. A real response from the site
/// this stack was pointed at - Cloudflare in front of GitHub Pages - carries **1043 bytes** of head:
/// 26 header lines, of which `Report-To` alone is 254 bytes and `Nel`, `X-Fastly-Request-ID`,
/// `X-GitHub-Request-Id` and `alt-svc` are another 200 between them. At 1024 the request failed with
/// `HttpHeadersTooLong` after the body had already been negotiated, 19 bytes short.
///
/// Modern CDN responses simply have large heads, and 1 KB is not a realistic ceiling for one. 2 KB
/// leaves about a kilobyte of margin over the measured case; the failure remains a named error
/// rather than truncation, so a head that exceeds even this is still diagnosable rather than silently
/// wrong.
pub const http_head_max: usize = 2048;
/// Bytes of chunked *framing* - one chunk's extension parameters, or the whole trailer section -
/// tolerated before the response is failed. Framing is skipped rather than stored, so this bounds
/// work and not memory: without it a peer that streams `;a=b` forever, or trailer lines forever,
/// is a request that never ends and never errors. 512 is generous; a real trailer section is one
/// or two short lines.
pub const http_framing_max: usize = 512;
/// The longest host name `resolve` will encode into a DNS question, in dotted text. RFC 1035 2.3.4
/// allows 255; this stack holds the encoded question in `Stack` for the duration of the query, and
/// 64 covers every name a device that fetches one URL will ever ask for. A longer one is
/// `error.NameTooLong`, never a silently truncated question.
pub const dns_name_max: usize = 64;
/// The encoded question that `dns_name_max` produces. Encoding turns `a.b` into `1a1b0`: one
/// length byte per label plus the root label, which for a name with no trailing dot is exactly
/// two bytes more than the text. RFC 1035 4.1.2.
pub const dns_qname_max: usize = dns_name_max + 2;
/// Bytes of outbound TCP payload held for retransmission. This is sized for one HTTP request line
/// plus headers; there is no streaming send, so it is also the hard limit on request size.
pub const tcp_tx_max: usize = 512;
/// The receive window this stack advertises, in bytes, when it has that much room to consume into.
/// One MSS: a peer that fills the window gets a segment acknowledged before it may send another.
pub const tcp_window: u16 = 1460;
/// TCP MSS offered in the SYN. 1500 - 20 (IP) - 20 (TCP).
pub const tcp_mss: u16 = 1460;
/// RFC 1122 4.2.2.6: a peer that sends no MSS option is assumed to accept 536.
pub const tcp_default_mss: u16 = 536;
/// Initial retransmission timeout. RFC 6298 2.1 specifies 1 s for a connection with no RTT sample,
/// and this stack never takes an RTT sample - see `Tcp.rto_ms`.
pub const tcp_rto_initial_ms: u32 = 1000;
/// Retransmission timeout ceiling. RFC 6298 5.7 allows any value at or above 60 s; 16 s is chosen
/// against a device whose whole reason to exist is one short request.
pub const tcp_rto_max_ms: u32 = 16_000;
/// Retransmissions of the same segment before the connection is abandoned with `error.TimedOut`.
/// With the backoff above that is 1+2+4+8+16+16 = 47 s of trying.
pub const tcp_max_retries: u8 = 6;
/// TIME_WAIT duration. RFC 793 says 2*MSL, conventionally 240 s. Two seconds is what this uses:
/// holding a connection block for four minutes on a part with 128 KB of RAM to protect a
/// port number that this stack increments on every connect is the wrong trade. The risk it drops is
/// a late duplicate segment from the *previous* incarnation of the same 4-tuple being accepted into
/// a new one, and incrementing the local port already makes a repeat 4-tuple require 16,384
/// connections first.
pub const tcp_time_wait_ms: u32 = 2000;
/// How long a half-closed connection waits for the peer's FIN before the block is released. RFC
/// 793 has no such timer and a connection may legitimately sit in FIN-WAIT-2 forever; Linux uses
/// 60 s for the same reason this uses 10 s - a peer that has our FIN and never answers is a peer
/// that is gone, and the one connection block here is not worth holding for it.
pub const tcp_fin_wait2_ms: u32 = 10_000;
/// DHCP retransmission backoff, in milliseconds, indexed by attempt. RFC 2131 4.1 asks for
/// randomised exponential backoff starting at 4 s; this starts at 2 s because the first DHCP
/// exchange is on the critical path of every boot, and does not randomise because there is one
/// client on this board and the collision RFC 2131 is avoiding is between many.
const dhcp_backoff_ms = [_]u32{ 2_000, 4_000, 8_000, 16_000, 32_000, 64_000 };
/// Minimum length of the BOOTP/DHCP message this stack transmits, in UDP payload bytes. RFC 951
/// fixed BOOTP messages at 300 bytes and relay agents in the field still expect at least that
/// much; lwIP pads the same way through its fixed-size `struct dhcp_msg`
/// (`prot/dhcp.h:63-91`: 236 + 4 cookie + `DHCP_OPTIONS_LEN` 68 = 308).
const dhcp_min_msg_len: usize = 300;
/// DNS retransmission backoff, in milliseconds, indexed by attempt. RFC 1035 4.2.1 leaves the
/// timer to the implementation; this is BIND's classic 1 s doubling, and the array length is the
/// try count, so the whole exchange is bounded at 1+2+4 = 7 s and then `error.TimedOut`. The
/// transaction id is *not* redrawn between tries: a slow first answer must still be accepted.
const dns_backoff_ms = [_]u32{ 1_000, 2_000, 4_000 };
/// Compression pointers followed while skipping one name (RFC 1035 4.1.4). This is the bound that
/// makes a hostile message terminate: see `dnsSkipName`, where the argument is written out.
const dns_max_jumps: u8 = 16;
// =============================================================== addresses and enumerations
pub const Mac = [6]u8;
pub const Ip4 = [4]u8;
pub const mac_broadcast: Mac = @splat(0xff);
pub const ip_any: Ip4 = @splat(0x00);
pub const ip_broadcast: Ip4 = @splat(0xff);
/// Ethernet type field values. lwIP `prot/ieee.h:52-85` (`enum lwip_ieee_eth_type`).
pub const EtherType = enum(u16) {
ip4 = 0x0800,
arp = 0x0806,
vlan = 0x8100,
ip6 = 0x86dd,
_,
};
/// IP header protocol numbers. lwIP `prot/ip.h:46-50`.
pub const Protocol = enum(u8) {
icmp = 1,
tcp = 6,
udp = 17,
_,
};
// =============================================================================== checksum
//
// One implementation for IPv4, ICMP, UDP and TCP. The last two prepend a pseudo-header, which is
// the only difference between them: the arithmetic is identical, so it is written once.
/// The Internet checksum of RFC 1071: the one's complement of the one's complement sum of the
/// data taken as 16-bit big-endian words, with a zero byte appended if the length is odd.
///
/// Incremental, because TCP and UDP checksum a pseudo-header, a header and a payload that are
/// three separate buffers and never adjacent in memory. Feeding them in sequence must give the
/// same answer as checksumming the concatenation, which is why `half` exists: a chunk of odd
/// length leaves the high byte of a word owed, and the next chunk's first byte completes it.
/// Getting that wrong is invisible until a payload happens to have odd length, which for HTTP is
/// most of the time.
pub const Checksum = struct {
/// Accumulated 16-bit words. Deferring the end-around carry is safe for any length this
/// stack can produce: 32 bits absorbs 65,536 words, and the largest thing checksummed here is
/// 1,500 bytes.
sum: u32 = 0,
/// High byte of a 16-bit word whose low byte has not arrived yet.
half: ?u8 = null,
pub fn update(self: *Checksum, bytes: []const u8) void {
var b = bytes;
if (self.half) |hi| {
if (b.len == 0) return;
self.sum += (@as(u32, hi) << 8) | b[0];
self.half = null;
b = b[1..];
}
var i: usize = 0;
while (i + 1 < b.len) : (i += 2) self.sum += std.mem.readInt(u16, b[i..][0..2], .big);
if (i < b.len) self.half = b[i];
}
/// Feed a big-endian 16-bit value, for the pseudo-header fields that are not in any buffer.
pub fn update16(self: *Checksum, v: u16) void {
var tmp: [2]u8 = undefined;
std.mem.writeInt(u16, &tmp, v, .big);
self.update(&tmp);
}
/// The checksum as it goes on the wire. RFC 1071 1: an odd-length buffer is padded with a
/// zero byte, which the fold below does implicitly by shifting the owed byte up.
pub fn final(self: Checksum) u16 {
var s = self.sum;
if (self.half) |hi| s += @as(u32, hi) << 8;
while (s >> 16 != 0) s = (s & 0xffff) + (s >> 16);
return ~@as(u16, @truncate(s));
}
};
/// The Internet checksum of one contiguous buffer.
pub fn checksum(bytes: []const u8) u16 {
var c: Checksum = .{};
c.update(bytes);
return c.final();
}
/// The TCP/UDP pseudo-header of RFC 793 3.1: source address, destination address, a zero byte, the
/// protocol number and the transport length. Not transmitted; only checksummed.
fn pseudoHeader(c: *Checksum, src: Ip4, dst: Ip4, proto: Protocol, len: u16) void {
c.update(&src);
c.update(&dst);
c.update16(@intFromEnum(proto)); // the zero byte and the protocol byte, as one word
c.update16(len);
}
/// A checksum for a UDP or TCP segment: pseudo-header, then the segment with its own checksum
/// field already zeroed.
fn transportChecksum(src: Ip4, dst: Ip4, proto: Protocol, segment: []const u8) u16 {
var c: Checksum = .{};
pseudoHeader(&c, src, dst, proto, @intCast(segment.len));
c.update(segment);
return c.final();
}
/// Verify a received transport checksum. A UDP datagram may carry zero, meaning "not computed"
/// (RFC 768); TCP may not.
fn transportChecksumOk(src: Ip4, dst: Ip4, proto: Protocol, segment: []const u8, field: u16) bool {
if (proto == .udp and field == 0) return true;
// Summing a segment that already contains its own checksum yields 0 (or, equivalently, the
// sum before complementing is 0xffff). RFC 1071 1.
return transportChecksum(src, dst, proto, segment) == 0;
}
/// A transmitted UDP checksum of zero would be read as "not computed", so RFC 768 requires it be
/// sent as the equivalent 0xffff instead. TCP has no such rule and no such ambiguity.
pub fn udpChecksumOnWire(c: u16) u16 {
return if (c == 0) 0xffff else c;
}
// ============================================================== unaligned big-endian access
//
// `std.mem.readInt`/`writeInt` with an explicit endianness at every single field. Never a shift and
// an or: a network header written by hand is where byte order goes wrong, and it goes wrong
// silently, on one field, in a way that looks like a hardware problem.
inline fn rd16(b: []const u8, off: usize) u16 {
return std.mem.readInt(u16, b[off..][0..2], .big);
}
inline fn rd32(b: []const u8, off: usize) u32 {
return std.mem.readInt(u32, b[off..][0..4], .big);
}
inline fn wr16(b: []u8, off: usize, v: u16) void {
std.mem.writeInt(u16, b[off..][0..2], v, .big);
}
inline fn wr32(b: []u8, off: usize, v: u32) void {
std.mem.writeInt(u32, b[off..][0..4], v, .big);
}
inline fn rdIp(b: []const u8, off: usize) Ip4 {
return b[off..][0..4].*;
}
inline fn wrIp(b: []u8, off: usize, v: Ip4) void {
b[off..][0..4].* = v;
}
inline fn rdMac(b: []const u8, off: usize) Mac {
return b[off..][0..6].*;
}
inline fn wrMac(b: []u8, off: usize, v: Mac) void {
b[off..][0..6].* = v;
}
// ============================================================================ header offsets
//
// Byte offsets rather than packed structs. `extern struct` would need `align(1)` on every field and
// a byte-swap on every access on this little-endian part, and the offsets are what the RFCs and
// lwIP's headers actually state, so this is the form that can be checked against them by eye.
/// lwIP `prot/ethernet.h:76-83` (`struct eth_hdr`).
const eth = struct {
const dst = 0;
const src = 6;
const ethertype = 12;
};
/// lwIP `prot/etharp.h:86-96` (`struct etharp_hdr`), `SIZEOF_ETHARP_HDR` 28 at `:102`.
const arp = struct {
const hwtype = 0;
const proto = 2;
const hwlen = 4;
const protolen = 5;
const opcode = 6;
const sha = 8; // sender hardware address
const spa = 14; // sender protocol address
const tha = 18; // target hardware address
const tpa = 24; // target protocol address
const len = 28;
/// lwIP `prot/iana.h:54` (`LWIP_IANA_HWTYPE_ETHERNET`).
const hwtype_ethernet: u16 = 1;
/// lwIP `prot/etharp.h:105-108` (`enum etharp_opcode`).
const op_request: u16 = 1;
const op_reply: u16 = 2;
};
/// lwIP `prot/ip4.h:73-97` (`struct ip_hdr`), `IP_HLEN` 20 at `:64`.
const ip4 = struct {
const v_hl = 0;
const tos = 1;
const total_len = 2;
const id = 4;
const frag = 6;
const ttl = 8;
const proto = 9;
const chksum = 10;
const src = 12;
const dst = 16;
const hlen = 20;
/// lwIP `prot/ip4.h:84-87`.
const flag_df: u16 = 0x4000;
const flag_mf: u16 = 0x2000;
const offset_mask: u16 = 0x1fff;
};
/// lwIP `prot/icmp.h:89-95` (`struct icmp_echo_hdr`).
const icmp = struct {
const type_ = 0;
const code = 1;
const chksum = 2;
const id = 4;
const seq = 6;
const hlen = 8;
/// lwIP `prot/icmp.h:46,50`.
const echo_reply: u8 = 0;
const echo_request: u8 = 8;
};
/// lwIP `prot/udp.h:53-58` (`struct udp_hdr`), `UDP_HLEN` 8 at `:46`.
const udp = struct {
const src_port = 0;
const dst_port = 2;
const len = 4;
const chksum = 6;
const hlen = 8;
};
/// lwIP `prot/tcp.h:56-65` (`struct tcp_hdr`), `TCP_HLEN` 20 at `:47`.
const tcp = struct {
const src_port = 0;
const dst_port = 2;
const seq = 4;
const ack = 8;
/// Top four bits are the header length in 32-bit words; the low six are the flags.
/// lwIP `prot/tcp.h:85-87`.
const hdrlen_flags = 12;
const window = 14;
const chksum = 16;
const urgent = 18;
const hlen = 20;
/// lwIP `prot/tcp.h:72-81`.
const fin: u8 = 0x01;
const syn: u8 = 0x02;
const rst: u8 = 0x04;
const psh: u8 = 0x08;
const ack_f: u8 = 0x10;
const urg: u8 = 0x20;
/// RFC 793 3.1: kind 2, length 4, then the 16-bit MSS.
const opt_end: u8 = 0;
const opt_nop: u8 = 1;
const opt_mss: u8 = 2;
};
/// lwIP `prot/dhcp.h:50-91` (`struct dhcp_msg`) and `:51-56` for the offsets named there.
const dhcp = struct {
const op = 0;
const htype = 1;
const hlen = 2;
const hops = 3;
const xid = 4;
const secs = 8;
const flags = 10;
const ciaddr = 12;
const yiaddr = 16;
const siaddr = 20;
const giaddr = 24;
const chaddr = 28;
const sname = 44; // DHCP_SNAME_OFS
const file = 108; // DHCP_FILE_OFS
const cookie = 236; // DHCP_MSG_LEN
const options = 240; // DHCP_OPTIONS_OFS = DHCP_MSG_LEN + 4
/// lwIP `prot/dhcp.h:116-117`.
const bootrequest: u8 = 1;
const bootreply: u8 = 2;
/// lwIP `prot/dhcp.h:120-127`.
const discover: u8 = 1;
const offer: u8 = 2;
const request: u8 = 3;
const ack: u8 = 5;
const nak: u8 = 6;
/// lwIP `prot/dhcp.h:129`.
const magic_cookie: u32 = 0x63825363;
/// RFC 2131 figure 2: the top bit of `flags` asks the server to broadcast its reply.
const flag_broadcast: u16 = 0x8000;
/// lwIP `prot/dhcp.h:134-165`. Only the ones this client uses.
const opt_pad: u8 = 0;
const opt_subnet_mask: u8 = 1;
const opt_router: u8 = 3;
const opt_dns: u8 = 6;
const opt_hostname: u8 = 12;
const opt_requested_ip: u8 = 50;
const opt_lease_time: u8 = 51;
const opt_overload: u8 = 52;
const opt_msg_type: u8 = 53;
const opt_server_id: u8 = 54;
const opt_param_list: u8 = 55;
const opt_max_msg_size: u8 = 57;
const opt_t1: u8 = 58;
const opt_t2: u8 = 59;
const opt_end: u8 = 255;
/// lwIP `prot/iana.h:66-68`.
const server_port: u16 = 67;
const client_port: u16 = 68;
};
/// RFC 1035 4.1. There is no lwIP reference for this one: lwIP's resolver is `core/dns.c`, which
/// builds the same header out of its own `struct dns_hdr` (`core/dns.c:180-190`) - the offsets
/// below are the RFC's, and `dns.c` is only a cross-check.
const dns = struct {
// 4.1.1 header, six 16-bit fields.
const id = 0;
const flags = 2;
const qdcount = 4;
const ancount = 6;
const nscount = 8;
const arcount = 10;
const hlen = 12;
/// 4.1.1: QR is the top bit of `flags`, RD is bit 8, RCODE the bottom four bits.
const flag_qr: u16 = 0x8000;
const flag_rd: u16 = 0x0100;
const rcode_mask: u16 = 0x000f;
/// RCODE 3, "name error": the name authoritatively does not exist. RFC 1035 4.1.1.
const rcode_name_error: u16 = 3;
/// 4.1.4: the two top bits of a length byte set means the rest is a 14-bit offset.
const ptr_mask: u8 = 0xc0;
/// 2.3.4: a label is at most 63 bytes, which is also why 0x40 and 0x80 are free to be flags.
const label_max: u8 = 63;
/// 3.2.2 TYPE and 3.2.4 CLASS. Only the two this stack looks at, plus CNAME, which is not
/// followed but must be stepped over: a name behind a CNAME chain answers with the chain and
/// the A record together, and a resolver that stops at the first record finds the CNAME.
const type_a: u16 = 1;
const type_cname: u16 = 5;
const class_in: u16 = 1;
/// 3.2.1: TYPE(2) CLASS(2) TTL(4) RDLENGTH(2) after the name.
const rr_fixed = 10;
/// lwIP `prot/iana.h:64` (`LWIP_IANA_PORT_DNS`).
const port: u16 = 53;
};
// ============================================================================== ARP cache
const ArpEntry = struct {
ip: Ip4 = ip_any,
mac: Mac = @splat(0),
/// `tick`'s clock at the last hit or update. Zero means the entry is empty.
stamp_ms: u64 = 0,
inline fn valid(self: ArpEntry) bool {
return self.stamp_ms != 0;
}
};
/// Entries older than this are treated as absent and re-resolved. lwIP's default is 300 s
/// (`ARP_TMR_INTERVAL` 1000 ms x `ARP_MAXAGE` 300, `core/ipv4/etharp.c`); the same here.
const arp_max_age_ms: u64 = 300_000;
/// How often an unanswered ARP request is repeated while `httpGet` waits for a MAC address.
const arp_retry_ms: u64 = 1000;
/// ARP requests sent for one destination before `httpGet` gives up with `error.HostUnreachable`.
const arp_max_tries: u8 = 5;
// ================================================================================ DHCP state
pub const DhcpState = enum {
/// `dhcpStart` has not been called, or `setStatic` has taken over.
off,
/// DISCOVER sent, waiting for an OFFER.
selecting,
/// REQUEST sent, waiting for an ACK.
requesting,
/// Bound, lease held, T1 not yet reached.
bound,
/// Past T1: unicast REQUEST to the server that granted the lease.
renewing,
/// Past T2: broadcast REQUEST to any server.
rebinding,
};
const Dhcp = struct {
state: DhcpState = .off,
xid: u32 = 0,
/// The address the server offered, held between OFFER and ACK.
offered: Ip4 = ip_any,
/// Option 54 from the OFFER, echoed in the REQUEST and unicast to when renewing.
server: Ip4 = ip_any,
/// Option 51, seconds. `0xffff_ffff` is an infinite lease (RFC 2131 3.3).
lease_s: u32 = 0,
/// Absolute deadlines derived from the lease at bind time, in `tick`'s milliseconds.
t1_ms: u64 = 0,
t2_ms: u64 = 0,
expire_ms: u64 = 0,
/// When the next DISCOVER/REQUEST retransmission is due, and how many have gone out.
retry_ms: u64 = 0,
tries: u8 = 0,
/// `tick`'s clock when acquisition began, for the `secs` field.
started_ms: u64 = 0,
};
// ================================================================================ TCP state
pub const TcpState = enum {
closed,
/// Waiting for the peer's MAC address before the SYN can be built.
arp_wait,
syn_sent,
established,
/// Our FIN is sent; the peer has not FINed.
fin_wait_1,
fin_wait_2,
/// The peer FINed first and we have replied with our own FIN.
last_ack,
time_wait,
};
const Tcp = struct {
state: TcpState = .closed,
peer_ip: Ip4 = ip_any,
peer_port: u16 = 0,
local_port: u16 = 0,
/// Initial send sequence number. The SYN occupies `iss`; request data occupies
/// `iss+1 .. iss+1+tx_len`; a FIN occupies `iss+1+tx_len`. Every offset in this struct is
/// derived from that one layout, which is why there is no separate "unacked offset".
iss: u32 = 0,
/// Oldest sequence number not yet acknowledged by the peer.
snd_una: u32 = 0,
/// Next sequence number to send.
snd_nxt: u32 = 0,
/// The peer's advertised window.
snd_wnd: u32 = 0,
/// The peer's MSS, from its SYN's option or RFC 1122's default.
snd_mss: u16 = tcp_default_mss,
/// Set once a FIN has been queued behind the request data.
fin_queued: bool = false,
/// Set once the peer's FIN has been received in order. Receiving it does not by itself move
/// `state`, so that `tcpSendData` stays the only thing that changes it.
peer_fin: bool = false,
/// Next sequence number expected from the peer.
rcv_nxt: u32 = 0,
/// Retransmission deadline in `tick`'s milliseconds, and the current timeout. `rto_ms` doubles
/// on every retransmission and is never reduced by an RTT measurement, because this stack
/// takes none: with a single segment in flight and a fixed backoff there is nothing an RTT
/// estimator would change except the first timeout, and 1 s is already RFC 6298's answer for
/// that case.
rto_deadline_ms: u64 = 0,
rto_ms: u32 = tcp_rto_initial_ms,
retries: u8 = 0,
/// When TIME_WAIT ends.
close_deadline_ms: u64 = 0,
/// The request bytes, held for retransmission until acknowledged.
tx: [tcp_tx_max]u8 = undefined,
tx_len: usize = 0,
/// Sequence number of the first byte of `tx`.
inline fn dataStart(self: Tcp) u32 {
return self.iss +% 1;
}
/// Sequence number one past the last byte of `tx`.
inline fn dataEnd(self: Tcp) u32 {
return self.iss +% 1 +% @as(u32, @intCast(self.tx_len));
}
};
/// Sequence-number comparison. TCP sequence numbers wrap, so they are compared by the sign of the
/// difference and never by `<`. RFC 1982 serial arithmetic; the classic bug this avoids is a
/// connection that stalls forever once the sequence space crosses 2^32.
inline fn seqLt(a: u32, b: u32) bool {
return @as(i32, @bitCast(a -% b)) < 0;
}
inline fn seqLe(a: u32, b: u32) bool {
return @as(i32, @bitCast(a -% b)) <= 0;
}
inline fn seqGt(a: u32, b: u32) bool {
return seqLt(b, a);
}
inline fn seqGe(a: u32, b: u32) bool {
return seqLe(b, a);
}
// =============================================================================== HTTP state
pub const HttpError = error{
/// The request is in flight. Call `tick`, feed frames to `onFrame`, and call `httpGet` again
/// with the same arguments. This is the only "error" a healthy request returns.
WouldBlock,
/// `httpGet` was called with different arguments while a request was in flight.
Busy,
/// The peer's MAC address could not be resolved.
HostUnreachable,
/// The peer sent RST.
ConnectionReset,
/// The peer FINed or vanished before the body was complete.
ConnectionClosed,
/// Retransmissions exhausted.
TimedOut,
/// The status line, the headers, or a chunked body's trailer section exceeded its budget
/// (`http_head_max`, `http_framing_max`).
HttpHeadersTooLong,
/// The status line was not `HTTP/1.x SSS`.
HttpMalformed,
/// A chunked body's framing was not RFC 7230 4.1: a size with no hex digits, a size that
/// overflows `usize`, or a CRLF that was not where the grammar puts it. Distinct from
/// `HttpMalformed` because the two point at different halves of the response, and on a board
/// with one UART the error name is the whole diagnosis.
HttpChunkMalformed,
/// `Transfer-Encoding` was present and was neither `identity` nor `chunked`.
UnsupportedTransferEncoding,
/// The body did not fit in the caller's `out` buffer.
StreamTooLong,
/// The request line and headers did not fit in `tcp_tx_max`, or `path` is unusable.
RequestTooLong,
/// `httpGet` was called before the stack had an address.
NoAddress,
};
const HttpPhase = enum { idle, head, body, complete, failed };
const Http = struct {
phase: HttpPhase = .idle,
/// Valid when `phase == .failed`.
err: HttpError = error.WouldBlock,
/// The caller's output buffer, borrowed for the duration of the request. Recorded rather than
/// copied, so the caller must not move or resize it between `httpGet` calls; the identity check
/// in `httpGet` catches the common way of getting that wrong.
out: []u8 = &.{},
out_len: usize = 0,
/// The request being served, kept so a re-entrant `httpGet` can be told apart from a new one.
/// The hash covers the path *and* the `Host:` name, which is what makes two requests to the
/// same address for the same path but different virtual hosts distinguishable - and they must
/// be, or the second silently rides on the first's connection. A hash rather than the strings
/// themselves because `Stack` has a 4 KiB budget and the strings are the caller's, alive for
/// the duration of the call only.
req_host: Ip4 = ip_any,
req_port: u16 = 0,
req_hash: u64 = 0,
/// Status line and headers, accumulated until the blank line.
head: [http_head_max]u8 = undefined,
head_len: usize = 0,
status: u16 = 0,
/// `null` means the response had no usable `Content-Length`, so the body ends at the peer's
/// FIN - or, when `chunked`, at the zero-length chunk.
content_length: ?usize = null,
// ------------------------------------------------------- RFC 7230 4.1 chunked decoding
//
// Four fields hold the whole position in the chunked grammar, because a segment boundary may
// fall between any two bytes of it and the decoder has to resume from exactly here.
/// `Transfer-Encoding: chunked` was in force on this response.
chunked: bool = false,
chunk: ChunkState = .size,
/// In `.size`, the hexadecimal size accumulated so far; in `.data`, the bytes of this chunk
/// still to come. The two are the same number, which is why one field serves both: the size
/// read is the count remaining the instant the header ends.
chunk_left: usize = 0,
/// At least one hex digit has been seen in the size being read. RFC 7230 4.1 is `1*HEXDIG`,
/// so an empty size is malformed - and without this flag a stray CRLF reads as a chunk of
/// length zero, which is the terminator, which ends the body early and looks like success.
chunk_digit: bool = false,
/// Framing bytes consumed in the extension or trailer section now being skipped, against
/// `http_framing_max`.
chunk_skip: u16 = 0,
};
/// Where the chunked decoder is in RFC 7230 4.1's grammar:
///
/// chunked-body = *chunk last-chunk trailer-part CRLF
/// chunk = chunk-size [ chunk-ext ] CRLF chunk-data CRLF
/// last-chunk = 1*("0") [ chunk-ext ] CRLF
///
/// Every terminal in that grammar that can be split by a segment boundary is a state, including
/// the two halves of each CRLF. That is not pedantry: a 1,460-byte segment ends wherever the
/// server's writes happen to end, and "the CR arrived and the LF did not" is a case that happens.
const ChunkState = enum {
/// Reading hex digits of `chunk-size`.
size,
/// A `;` was seen: skipping `chunk-ext` to the CR that ends the header.
ext,
/// The CR of the chunk header is in; its LF must follow.
size_lf,
/// Copying `chunk_left` more bytes of `chunk-data` into the caller's `out`.
data,
/// The data is in; the CR of the CRLF that closes the chunk must follow.
data_cr,
/// ...and its LF.
data_lf,
/// At the first byte of a trailer line - or of the CRLF that ends the whole body.
trailer,
/// Inside a trailer line, skipping to its CR.
trailer_line,
/// The LF of a trailer line.
trailer_lf,
/// The LF of the final empty line. The response is complete after it, and not before.
end_lf,
};
// ================================================================================ DNS state
pub const DnsError = error{
/// The query is in flight. Call `tick`, feed frames to `onFrame`, and call `resolve` again
/// with the same name. This is the only "error" a healthy query returns.
WouldBlock,
/// `resolve` was called with a different name while a query was in flight. One query is
/// outstanding at a time; the caller must finish or abandon the first.
Busy,
/// `resolve` was called before the stack had an address of its own to send from.
NoAddress,
/// No resolver: DHCP supplied none and `setDnsServer` was not called.
NoDnsServer,
/// The name was empty, had an empty label, or had a label over 63 bytes. RFC 1035 2.3.4.
NameInvalid,
/// The name was longer than `dns_name_max`.
NameTooLong,
/// `dns_backoff_ms.len` queries went out and nothing came back.
TimedOut,
/// The server said the name does not exist (RCODE 3), or answered with no A record in it -
/// a CNAME chain leading nowhere, or an AAAA-only name. Both mean the same thing to a stack
/// that speaks IPv4 only.
NameNotFound,
/// The server answered with a non-zero RCODE other than name-error: SERVFAIL, REFUSED.
DnsRefused,
/// A response that matched the id and the question could not be parsed: a name that runs off
/// the end, a compression pointer that goes forward or loops, an RDLENGTH past the message.
/// Responses that do *not* match the id and question are ignored rather than reported, so
/// this is the server or an attacker who already guessed both, never stray traffic.
DnsMalformed,
};
const DnsPhase = enum { idle, waiting, done, failed };
const DnsQuery = struct {
phase: DnsPhase = .idle,
/// Valid when `phase == .failed`.
err: DnsError = error.WouldBlock,
/// The question, in RFC 1035 4.1.2 wire form, root label included. Held rather than
/// re-encoded because it is needed in three places: to build each retransmission from `tick`,
/// to compare against the question echoed in a response, and to tell a re-entrant `resolve`
/// from a new one. Comparing the encoded form is what makes the last two exact.
qname: [dns_qname_max]u8 = undefined,
qname_len: u8 = 0,
/// The transaction id, held across retransmissions so a slow first answer still matches.
id: u16 = 0,
/// The ephemeral source port, redrawn per query. Together with `id` that is 32 bits an
/// off-path spoofer has to guess, which is the whole of what plain DNS offers.
local_port: u16 = 0,
tries: u8 = 0,
retry_ms: u64 = 0,
/// Valid when `phase == .done`.
result: Ip4 = ip_any,
};
// ================================================================================= counters
//
// Not statistics for their own sake: the first hardware bring-up of this stack will be a board that
// either answers a ping or does not, with no debugger and one UART. These are what turns "nothing
// happens" into "1,204 frames arrived, 1,204 were dropped, and the checksum counter is zero", which
// says the frames are not for us rather than that the checksum code is broken.
pub const Counters = struct {
rx_frames: u32 = 0,
rx_dropped: u32 = 0,
tx_frames: u32 = 0,
tx_dropped: u32 = 0,
arp_rx: u32 = 0,
arp_tx: u32 = 0,
icmp_echo: u32 = 0,
udp_rx: u32 = 0,
dhcp_rx: u32 = 0,
dhcp_tx: u32 = 0,
tcp_rx: u32 = 0,
tcp_tx: u32 = 0,
tcp_retx: u32 = 0,
tcp_rst_rx: u32 = 0,
/// Queries sent, retransmissions among them, and responses that matched the outstanding
/// query's id and question. `dns_tx > dns_rx` with `dns_retx` climbing is a resolver that is
/// not answering; `dns_rx == 0` with `udp_rx` climbing is an answer arriving and being
/// rejected, which is a different bug in a different place.
dns_tx: u32 = 0,
dns_retx: u32 = 0,
dns_rx: u32 = 0,
/// Frames discarded because a checksum did not verify. A non-zero value here with a working
/// link means a bug in this file or a broken SDIO transfer, not a network problem.
checksum_bad: u32 = 0,
};
// ==================================================================================== Stack
pub const Stack = struct {
// ------------------------------------------------------------------ identity and route
mac: Mac,
/// `null` until DHCP binds or `setStatic` is called.
addr: ?Ip4 = null,
mask: Ip4 = ip_any,
gw: Ip4 = ip_any,
/// The resolver, from DHCP option 6 or `setDnsServer`. `null` means nothing to ask.
dns: ?Ip4 = null,
/// Where frames go. Called synchronously from `onFrame`, `tick` and `httpGet`; the slice is
/// borrowed for the duration of the call and must be copied if the transport needs it later.
///
/// Note the absence of a context pointer: the interface this slice implements specifies
/// `*const fn ([]const u8) void`, so a callee needing state has to reach it some other way.
send: *const fn (frame: []const u8) void,
// ------------------------------------------------------------------------------ clock
/// The last value handed to `tick`. `onFrame` needs a timestamp for the ARP cache and takes it
/// from here rather than reading a clock, which is what keeps this file free of any hardware
/// dependency at all.
now_ms: u64 = 0,
// ------------------------------------------------------------------------------ state
arp_cache: [arp_cache_len]ArpEntry = @splat(.{}),
/// Pending ARP resolution for the TCP peer: deadline, tries.
arp_retry_ms: u64 = 0,
arp_tries: u8 = 0,
dhcp: Dhcp = .{},
tcp: Tcp = .{},
http: Http = .{},
/// The one outstanding DNS query. Named `query` and not `dns`, which is the server's address.
query: DnsQuery = .{},
counters: Counters = .{},
/// IPv4 identification field. Incremented per datagram. Nothing here fragments, so this only
/// has to be non-constant for the benefit of middleboxes and packet captures.
ip_id: u16 = 0,
/// Mixed into transaction ids, initial sequence numbers and ephemeral ports. There is no
/// hardware RNG in this file's reach, so this is seeded from the MAC and stirred by every
/// `tick` value observed - which for the two uses here (not colliding with a previous
/// incarnation of the same connection, and not matching a stale DHCP reply) is sufficient.
/// It is emphatically *not* a source of security-relevant randomness.
entropy: u64,
/// The single transmit staging buffer. Every frame this stack sends is built here and handed to
/// `send` before the next one starts, so one is enough - and `send` is documented as borrowing.
tx: [frame_max]u8 = undefined,
/// The total static footprint of one `Stack`, asserted so the number in the report cannot rot.
pub const footprint = @sizeOf(Stack);
// =========================================================================== lifecycle
/// A single struct-literal return, deliberately: result-location semantics then construct the
/// buffers in the caller's storage instead of memcpy-ing several kilobytes off a stack that is
/// 8 KB by default on this target.
pub fn init(mac: [6]u8, send: *const fn (frame: []const u8) void) Stack {
return .{
.mac = mac,
.send = send,
.entropy = std.hash.Wyhash.hash(0x4200_cafe, &mac),
};
}
/// Stir and draw. Not random; see `entropy`.
fn draw(self: *Stack) u32 {
self.entropy = self.entropy *% 6364136223846793005 +% 1442695040888963407;
return @truncate(self.entropy >> 32);
}
// ============================================================================= address
/// The configured address, or `null` if there is none yet.
pub fn ip(self: *Stack) ?[4]u8 {
return self.addr;
}
pub fn netmask(self: *Stack) Ip4 {
return self.mask;
}
pub fn gateway(self: *Stack) Ip4 {
return self.gw;
}
/// The resolver `resolve` will ask: the first server DHCP offered (option 6), or whatever
/// `setDnsServer` last set. `null` means `resolve` will answer `error.NoDnsServer`.
pub fn dnsServer(self: *Stack) ?Ip4 {
return self.dns;
}
/// Override the resolver. Only needed on a network whose DHCP server offers none, or when
/// configuring statically: the ordinary path is a lease that carries option 6, which
/// `dhcpBind` already stores, and a caller that does nothing gets that.
///
/// Any query in flight is abandoned: it was addressed to the old server and its answer would
/// now be rejected as coming from the wrong source.
pub fn setDnsServer(self: *Stack, addr: Ip4) void {
self.dns = addr;
self.query.phase = .idle;
}
pub fn dhcpState(self: *Stack) DhcpState {
return self.dhcp.state;
}
pub fn tcpState(self: *Stack) TcpState {
return self.tcp.state;
}
/// The status code of the last response whose head was parsed. Zero before that.
pub fn httpStatus(self: *Stack) u16 {
return self.http.status;
}
/// Configure statically and stop any DHCP activity. This is the path the first hardware test
/// takes: it makes the board reachable without a working DHCP client, so an ARP or ping
/// failure means the SDIO transport or the association is wrong rather than this file.
pub fn setStatic(self: *Stack, addr: [4]u8, mask: [4]u8, gw: [4]u8) void {
self.dhcp = .{};
self.addr = addr;
self.mask = mask;
self.gw = gw;
// Any query in flight was sent from the old address, so its answer is addressed to a
// station that no longer exists. `dns` itself is left alone: a resolver learnt from a
// previous lease is still the right one to ask on the same wire.
self.query.phase = .idle;
self.announce();
}
/// Is this address ours, or one everybody on the wire is meant to hear?
fn forUs(self: *Stack, dst: Ip4) bool {
if (std.mem.eql(u8, &dst, &ip_broadcast)) return true;
const a = self.addr orelse return false;
if (std.mem.eql(u8, &dst, &a)) return true;
// Subnet broadcast: host bits all ones.
var i: usize = 0;
while (i < 4) : (i += 1) {
if (dst[i] | self.mask[i] != 0xff) return false;
if (dst[i] & self.mask[i] != a[i] & self.mask[i]) return false;
}
return true;
}
fn onLink(self: *Stack, dst: Ip4) bool {
const a = self.addr orelse return true; // unconfigured: everything is a direct neighbour
var i: usize = 0;
while (i < 4) : (i += 1) {
if ((dst[i] ^ a[i]) & self.mask[i] != 0) return false;
}
return true;
}
// ================================================================== frame construction
fn emitFrame(self: *Stack, len: usize) void {
if (len > frame_max) {
self.counters.tx_dropped += 1;
return;
}
// Ethernet's 60-byte minimum (64 with FCS) is padded by the MAC, and ESP-Hosted's slave
// hands the frame to the C6's Wi-Fi MAC, which does the same. Nothing is padded here.
self.counters.tx_frames += 1;
self.send(self.tx[0..len]);
}
fn ethHeader(self: *Stack, dst: Mac, ethertype: EtherType) void {
wrMac(&self.tx, eth.dst, dst);
wrMac(&self.tx, eth.src, self.mac);
wr16(&self.tx, eth.ethertype, @intFromEnum(ethertype));
}
/// Build and send an IPv4 datagram whose payload the caller has already written to
/// `self.tx[eth_hlen + ip4.hlen ..]`. Returns false if the destination's MAC is unknown, in
/// which case an ARP request has been sent and the datagram is dropped.
///
/// Dropping rather than queueing is lwIP's `ETHARP_SUPPORT_STATIC_ENTRIES`-less behaviour minus
/// its one-packet queue (`core/ipv4/etharp.c`, `etharp_query`). Nothing here needs the queue:
/// DHCP is broadcast, ICMP replies go to a peer whose MAC just arrived in the request, and TCP
/// resolves the peer before the SYN is built (`TcpState.arp_wait`).
fn emitIp(self: *Stack, src: Ip4, dst: Ip4, proto: Protocol, payload_len: usize) bool {
assert(payload_len <= mtu - ip4.hlen);
const total: u16 = @intCast(ip4.hlen + payload_len);
const dst_mac = self.routeMac(dst) orelse {
self.counters.tx_dropped += 1;
return false;
};
self.ethHeader(dst_mac, .ip4);
const h = self.tx[eth_hlen..][0..ip4.hlen];
h[ip4.v_hl] = 0x45; // IPv4, 5 words of header, no options
h[ip4.tos] = 0;
wr16(h, ip4.total_len, total);
wr16(h, ip4.id, self.ip_id);
self.ip_id +%= 1;
// DF set: this stack neither fragments what it sends nor reassembles what it receives, so
// saying so is more useful than letting a router fragment a datagram we cannot rebuild.
wr16(h, ip4.frag, ip4.flag_df);
h[ip4.ttl] = 64; // RFC 1122 3.2.1.7 recommends 64
h[ip4.proto] = @intFromEnum(proto);
wr16(h, ip4.chksum, 0);
wrIp(h, ip4.src, src);
wrIp(h, ip4.dst, dst);
wr16(h, ip4.chksum, checksum(h));
self.emitFrame(eth_hlen + total);
return true;
}
/// The MAC a datagram for `dst` must be sent to: broadcast for a broadcast address, the peer
/// itself if it is on-link, otherwise the gateway. `null` means unresolved, and an ARP request
/// has been sent.
fn routeMac(self: *Stack, dst: Ip4) ?Mac {
if (std.mem.eql(u8, &dst, &ip_broadcast)) return mac_broadcast;
if (self.addr != null) {
// Subnet broadcast.
var all_ones = true;
var i: usize = 0;
while (i < 4) : (i += 1) {
if (dst[i] | self.mask[i] != 0xff) all_ones = false;
}
if (all_ones and self.onLink(dst)) return mac_broadcast;
}
const next = if (self.onLink(dst)) dst else self.gw;
if (self.arpLookup(next)) |m| return m;
self.arpRequest(next);
return null;
}
// ================================================================================= ARP
/// An entry's timestamp is *not* refreshed by a lookup, only by an ARP packet from that host.
/// Refreshing on use looks like a cheap optimisation and is a real bug: an entry kept alive by
/// our own traffic is never re-resolved, so a gateway whose MAC changes - VRRP failover, a
/// replaced router, a roam to a different AP with a different BSSID-derived address - is never
/// noticed, and every frame goes to a MAC that no longer answers. Ageing out after
/// `arp_max_age_ms` of no ARP traffic from that host costs one dropped segment and a
/// retransmission; getting it wrong costs the connection.
fn arpLookup(self: *Stack, target: Ip4) ?Mac {
for (&self.arp_cache) |*e| {
if (!e.valid()) continue;
if (self.now_ms -% e.stamp_ms > arp_max_age_ms) {
e.stamp_ms = 0;
continue;
}
if (std.mem.eql(u8, &e.ip, &target)) return e.mac;
}
return null;
}
/// Insert or refresh. `insert` false means "update only if already known", which is how a
/// four-entry cache survives a busy /24: every ARP request on the segment is a broadcast, and a
/// cache that admitted all of them would evict the gateway within seconds. lwIP draws the same
/// line with `ETHARP_FLAG_TRY_HARD` (`core/ipv4/etharp.c`, `etharp_update_arp_entry`).
fn arpStore(self: *Stack, target: Ip4, hw: Mac, insert: bool) void {
if (std.mem.eql(u8, &target, &ip_any)) return;
if (std.mem.eql(u8, &target, &ip_broadcast)) return;
for (&self.arp_cache) |*e| {
if (e.valid() and std.mem.eql(u8, &e.ip, &target)) {
e.mac = hw;
e.stamp_ms = self.now_ms;
return;
}
}
if (!insert) return;
// Free slot, else the least recently used.
var victim: *ArpEntry = &self.arp_cache[0];
for (&self.arp_cache) |*e| {
if (!e.valid()) {
victim = e;
break;
}
if (e.stamp_ms < victim.stamp_ms) victim = e;
}
victim.* = .{ .ip = target, .mac = hw, .stamp_ms = self.now_ms };
}
fn arpEmit(self: *Stack, opcode: u16, target_ip: Ip4, target_mac: Mac, dst_mac: Mac, spa: Ip4) void {
self.ethHeader(dst_mac, .arp);
const h = self.tx[eth_hlen..][0..arp.len];
wr16(h, arp.hwtype, arp.hwtype_ethernet);
wr16(h, arp.proto, @intFromEnum(EtherType.ip4));
h[arp.hwlen] = 6;
h[arp.protolen] = 4;
wr16(h, arp.opcode, opcode);
wrMac(h, arp.sha, self.mac);
wrIp(h, arp.spa, spa);
wrMac(h, arp.tha, target_mac);
wrIp(h, arp.tpa, target_ip);
self.counters.arp_tx += 1;
self.emitFrame(eth_hlen + arp.len);
}
fn arpRequest(self: *Stack, target: Ip4) void {
// RFC 826: the target hardware address of a request is "don't care"; zero is conventional.
self.arpEmit(arp.op_request, target, @splat(0), mac_broadcast, self.addr orelse ip_any);
}
/// Gratuitous ARP: a broadcast request for our own address, which every listener treats as
/// "this MAC now owns this IP". Sent when an address is acquired, so the gateway and the AP
/// learn us without waiting to need us. RFC 5227 2.3.
fn announce(self: *Stack) void {
const a = self.addr orelse return;
self.arpEmit(arp.op_request, a, @splat(0), mac_broadcast, a);
}
fn arpInput(self: *Stack, body: []const u8) void {
if (body.len < arp.len) {
self.counters.rx_dropped += 1;
return;
}
// RFC 826 "Packet Reception", exactly the four checks lwIP makes at
// `core/ipv4/etharp.c:656-659`.
if (rd16(body, arp.hwtype) != arp.hwtype_ethernet or
rd16(body, arp.proto) != @intFromEnum(EtherType.ip4) or
body[arp.hwlen] != 6 or body[arp.protolen] != 4)
{
self.counters.rx_dropped += 1;
return;
}
self.counters.arp_rx += 1;
const spa = rdIp(body, arp.spa);
const sha = rdMac(body, arp.sha);
const tpa = rdIp(body, arp.tpa);
const for_us = if (self.addr) |a| std.mem.eql(u8, &tpa, &a) else false;
// Learn the sender. Admitted to a free slot only when the packet was addressed to us -
// either a request we must answer or the reply to a request we sent.
self.arpStore(spa, sha, for_us);
if (rd16(body, arp.opcode) == arp.op_request and for_us) {
// A reply goes back to the requester, not to the broadcast address.
self.arpEmit(arp.op_reply, spa, sha, sha, self.addr.?);
}
}
// =============================================================================== input
/// A received Ethernet frame. Everything this stack does in response happens before this
/// returns, including any frame it sends.
pub fn onFrame(self: *Stack, frame: []const u8) void {
self.counters.rx_frames += 1;
if (frame.len < eth_hlen or frame.len > frame_max) {
self.counters.rx_dropped += 1;
return;
}
const dst = rdMac(frame, eth.dst);
// The C6's MAC filter should already have done this, but a promiscuous or misconfigured
// transport would otherwise have this stack answering ARP for other stations.
if (!std.mem.eql(u8, &dst, &self.mac) and !std.mem.eql(u8, &dst, &mac_broadcast)) {
self.counters.rx_dropped += 1;
return;
}
const body = frame[eth_hlen..];
switch (@as(EtherType, @enumFromInt(rd16(frame, eth.ethertype)))) {
.arp => self.arpInput(body),
.ip4 => self.ip4Input(body),
// .vlan lands here: an 802.1Q tag would need the 4-byte shim skipped and the real
// ethertype read from behind it. Nothing on this board tags frames, so it is dropped
// rather than half-handled.
else => self.counters.rx_dropped += 1,
}
}
fn ip4Input(self: *Stack, body: []const u8) void {
if (body.len < ip4.hlen) {
self.counters.rx_dropped += 1;
return;
}
if (body[ip4.v_hl] >> 4 != 4) {
self.counters.rx_dropped += 1;
return;
}
const hlen = @as(usize, body[ip4.v_hl] & 0x0f) * 4;
if (hlen < ip4.hlen or hlen > body.len) {
self.counters.rx_dropped += 1;
return;
}
if (checksum(body[0..hlen]) != 0) {
self.counters.checksum_bad += 1;
return;
}
const total = rd16(body, ip4.total_len);
if (total < hlen or total > body.len) {
// Shorter than claimed: truncated. Longer than claimed happens legitimately - a
// 60-byte minimum-length Ethernet frame padding a 28-byte datagram - and is handled by
// trusting `total` below, but a frame shorter than its own IP header claims is junk.
self.counters.rx_dropped += 1;
return;
}
const frag = rd16(body, ip4.frag);
if (frag & (ip4.flag_mf | ip4.offset_mask) != 0) {
// A fragment. Reassembly is out of scope, and accepting the first fragment as a whole
// datagram would be worse than dropping it.
self.counters.rx_dropped += 1;
return;
}
const src = rdIp(body, ip4.src);
const dst = rdIp(body, ip4.dst);
const proto: Protocol = @enumFromInt(body[ip4.proto]);
const payload = body[hlen..total];
if (!self.forUs(dst)) {
// One exception, and it is the reason DHCP works at all: a server may unicast its
// OFFER or ACK to the address it is about to grant, which is not yet ours, at a MAC
// that is. RFC 2131 4.1 permits exactly this. So while unbound, UDP is let through to
// the demultiplexer, which will only match the DHCP client port.
const dhcp_pending = self.addr == null and self.dhcp.state != .off;
if (!(dhcp_pending and proto == .udp)) {
self.counters.rx_dropped += 1;
return;
}
}
switch (proto) {
.icmp => self.icmpInput(src, dst, payload),
.udp => self.udpInput(src, dst, payload),
.tcp => self.tcpInput(src, dst, payload),
else => self.counters.rx_dropped += 1,
}
}
// ================================================================================ ICMP
fn icmpInput(self: *Stack, src: Ip4, dst: Ip4, payload: []const u8) void {
if (payload.len < icmp.hlen) {
self.counters.rx_dropped += 1;
return;
}
// ICMP has no pseudo-header (RFC 792): the checksum covers the message alone.
if (checksum(payload) != 0) {
self.counters.checksum_bad += 1;
return;
}
if (payload[icmp.type_] != icmp.echo_request) {
// Destination-unreachable and time-exceeded carry useful information that nothing here
// consumes; a stack with no routing decisions to revise has nothing to do with them.
self.counters.rx_dropped += 1;
return;
}
// A request addressed to the broadcast address is answered from our own address only; a
// reply sourced from a broadcast address is malformed and some hosts treat it as an attack.
if (self.addr == null) return;
if (payload.len > mtu - ip4.hlen) {
// Would need fragmenting to answer. `ping -s 1473` from the development host lands
// here as a fragmented request and is already dropped above; this covers the rest.
self.counters.rx_dropped += 1;
return;
}
_ = dst;
const out = self.tx[eth_hlen + ip4.hlen ..][0..payload.len];
@memcpy(out, payload);
out[icmp.type_] = icmp.echo_reply;
out[icmp.code] = 0;
wr16(out, icmp.chksum, 0);
wr16(out, icmp.chksum, checksum(out));
self.counters.icmp_echo += 1;
_ = self.emitIp(self.addr.?, src, .icmp, payload.len);
}
// ================================================================================= UDP
fn udpInput(self: *Stack, src: Ip4, dst: Ip4, payload: []const u8) void {
if (payload.len < udp.hlen) {
self.counters.rx_dropped += 1;
return;
}
const ulen = rd16(payload, udp.len);
if (ulen < udp.hlen or ulen > payload.len) {
self.counters.rx_dropped += 1;
return;
}
const datagram = payload[0..ulen];
if (!transportChecksumOk(src, dst, .udp, datagram, rd16(datagram, udp.chksum))) {
self.counters.checksum_bad += 1;
return;
}
self.counters.udp_rx += 1;
const sport = rd16(datagram, udp.src_port);
const dport = rd16(datagram, udp.dst_port);
const data = datagram[udp.hlen..];
if (dport == dhcp.client_port) {
self.dhcpInput(src, data);
} else if (self.query.phase == .waiting and dport == self.query.local_port) {
self.dnsInput(src, sport, data);
} else {
// No sockets, so nothing else has a port. A real stack would answer with ICMP port
// unreachable; announcing which ports are closed is of no use to this device.
self.counters.rx_dropped += 1;
}
}
/// Send a UDP datagram. `src` may be `0.0.0.0`, which DHCP needs before it has an address.
fn emitUdp(self: *Stack, src: Ip4, sport: u16, dst: Ip4, dport: u16, data_len: usize) bool {
const seg_len = udp.hlen + data_len;
assert(seg_len <= mtu - ip4.hlen);
const seg = self.tx[eth_hlen + ip4.hlen ..][0..seg_len];
wr16(seg, udp.src_port, sport);
wr16(seg, udp.dst_port, dport);
wr16(seg, udp.len, @intCast(seg_len));
wr16(seg, udp.chksum, 0);
wr16(seg, udp.chksum, udpChecksumOnWire(transportChecksum(src, dst, .udp, seg)));
return self.emitIp(src, dst, .udp, seg_len);
}
// ================================================================================ DHCP
/// Begin acquiring an address. Idempotent while an acquisition is in progress; a call while
/// bound restarts from DISCOVER.
pub fn dhcpStart(self: *Stack) void {
self.addr = null;
self.mask = ip_any;
self.gw = ip_any;
self.dns = null;
// The resolver is gone with the lease, so anything in flight to it is abandoned rather
// than left to time out against a server this stack no longer believes in.
self.query.phase = .idle;
self.dhcp = .{
.state = .selecting,
.xid = self.draw(),
.started_ms = self.now_ms,
};
self.dhcpSend(dhcp.discover);
self.dhcp.tries = 1;
self.dhcp.retry_ms = self.now_ms + dhcp_backoff_ms[0];
}
/// Options are appended through this so a length byte can never be written by hand.
const OptWriter = struct {
buf: []u8,
i: usize = 0,
fn raw(self: *OptWriter, code: u8, value: []const u8) void {
assert(value.len <= 255);
assert(self.i + 2 + value.len <= self.buf.len);
self.buf[self.i] = code;
self.buf[self.i + 1] = @intCast(value.len);
@memcpy(self.buf[self.i + 2 ..][0..value.len], value);
self.i += 2 + value.len;
}
fn byte(self: *OptWriter, code: u8, v: u8) void {
self.raw(code, &[_]u8{v});
}
fn word(self: *OptWriter, code: u8, v: u16) void {
var t: [2]u8 = undefined;
std.mem.writeInt(u16, &t, v, .big);
self.raw(code, &t);
}
fn address(self: *OptWriter, code: u8, v: Ip4) void {
self.raw(code, &v);
}
fn end(self: *OptWriter) void {
assert(self.i < self.buf.len);
self.buf[self.i] = dhcp.opt_end;
self.i += 1;
}
};
/// Build and send one DHCP message. The RFC 2131 4.3.6 table is what decides which fields are
/// set: it is the part of DHCP that servers actually enforce, and getting `ciaddr` or the
/// server identifier wrong produces a NAK rather than an error message.
fn dhcpSend(self: *Stack, kind: u8) void {
const msg = self.tx[eth_hlen + ip4.hlen + udp.hlen ..][0..dhcp_min_msg_len];
@memset(msg, 0);
const renewing = self.dhcp.state == .renewing;
const rebinding = self.dhcp.state == .rebinding;
// RENEWING and REBINDING carry the bound address in `ciaddr` and no requested-IP option;
// SELECTING and REQUESTING carry zero and use option 50. lwIP makes the same distinction
// at `core/ipv4/dhcp.c:2026-2030`.
const use_ciaddr = renewing or rebinding;
msg[dhcp.op] = dhcp.bootrequest;
msg[dhcp.htype] = @intCast(arp.hwtype_ethernet);
msg[dhcp.hlen] = 6;
msg[dhcp.hops] = 0;
wr32(msg, dhcp.xid, self.dhcp.xid);
wr16(msg, dhcp.secs, @intCast(@min(0xffff, (self.now_ms -% self.dhcp.started_ms) / 1000)));
// Ask the server to broadcast its reply. lwIP clears this flag
// (`core/ipv4/dhcp.c:2024-2025`: "we don't need the broadcast flag since we can receive
// unicast traffic before being fully configured"), and so can this stack - `ip4Input` has
// the explicit exemption for it. The flag is set anyway because a broadcast reply is the
// path with the fewest ways to fail on first bring-up: it needs no ARP entry at the server,
// no unicast-to-unconfigured-host handling in the AP, and no exemption in this file.
wr16(msg, dhcp.flags, dhcp.flag_broadcast);
if (use_ciaddr) wrIp(msg, dhcp.ciaddr, self.addr orelse ip_any);
@memcpy(msg[dhcp.chaddr..][0..6], &self.mac);
wr32(msg, dhcp.cookie, dhcp.magic_cookie);
var o: OptWriter = .{ .buf = msg[dhcp.options..] };
o.byte(dhcp.opt_msg_type, kind);
// RFC 2131 3.5: the maximum message size we can reassemble. One MTU minus the headers,
// which for this stack is also the largest datagram it can receive at all.
o.word(dhcp.opt_max_msg_size, @intCast(mtu - ip4.hlen - udp.hlen));
if (kind == dhcp.request and !use_ciaddr) {
o.address(dhcp.opt_requested_ip, self.dhcp.offered);
o.address(dhcp.opt_server_id, self.dhcp.server);
}
// RFC 2131 4.3.6: a REQUEST in RENEWING/REBINDING must not carry a server identifier.
o.raw(dhcp.opt_param_list, &[_]u8{
dhcp.opt_subnet_mask,
dhcp.opt_router,
dhcp.opt_dns,
dhcp.opt_lease_time,
dhcp.opt_t1,
dhcp.opt_t2,
});
o.raw(dhcp.opt_hostname, "esp32p4");
o.end();
// Everything past the END option stays zero: RFC 2131 4.1 pads with option 0.
const src = if (use_ciaddr) (self.addr orelse ip_any) else ip_any;
// RENEWING unicasts to the server that granted the lease; every other message is broadcast
// (RFC 2131 4.3.6, 4.4.5).
const dst = if (renewing) self.dhcp.server else ip_broadcast;
self.counters.dhcp_tx += 1;
_ = self.emitUdp(src, dhcp.client_port, dst, dhcp.server_port, dhcp_min_msg_len);
}
/// One parsed option, or the end of the list.
const Opt = struct { code: u8, value: []const u8 };
/// Walk a DHCP option list. Stops at END, at a truncated option, or at the end of the buffer -
/// a malformed length must not walk off the datagram, which is the classic DHCP parser bug.
fn dhcpOption(body: []const u8, want: u8) ?[]const u8 {
if (body.len <= dhcp.options) return null;
var i: usize = dhcp.options;
while (i < body.len) {
const code = body[i];
if (code == dhcp.opt_end) return null;
if (code == dhcp.opt_pad) {
i += 1;
continue;
}
if (i + 2 > body.len) return null;
const len = body[i + 1];
if (i + 2 + len > body.len) return null;
if (code == want) return body[i + 2 ..][0..len];
i += 2 + len;
}
return null;
}
fn dhcpOptionIp(body: []const u8, want: u8) ?Ip4 {
const v = dhcpOption(body, want) orelse return null;
if (v.len < 4) return null;
return v[0..4].*;
}
fn dhcpOptionU32(body: []const u8, want: u8) ?u32 {
const v = dhcpOption(body, want) orelse return null;
if (v.len != 4) return null;
return std.mem.readInt(u32, v[0..4], .big);
}
fn dhcpInput(self: *Stack, src: Ip4, body: []const u8) void {
if (self.dhcp.state == .off) return;
if (body.len < dhcp.options) {
self.counters.rx_dropped += 1;
return;
}
if (body[dhcp.op] != dhcp.bootreply) return;
if (rd32(body, dhcp.cookie) != dhcp.magic_cookie) return;
if (rd32(body, dhcp.xid) != self.dhcp.xid) return;
// The reply must be about our hardware address, not a relayed one for someone else.
if (body[dhcp.hlen] != 6 or !std.mem.eql(u8, body[dhcp.chaddr..][0..6], &self.mac)) return;
const kind_opt = dhcpOption(body, dhcp.opt_msg_type) orelse return;
if (kind_opt.len != 1) return;
self.counters.dhcp_rx += 1;
switch (kind_opt[0]) {
dhcp.offer => {
if (self.dhcp.state != .selecting) return;
self.dhcp.offered = rdIp(body, dhcp.yiaddr);
if (std.mem.eql(u8, &self.dhcp.offered, &ip_any)) return;
// Option 54 is how the REQUEST names which offer it accepts. A server that omits
// it is out of spec; `siaddr` is the best fallback, and the sender is the last.
self.dhcp.server = dhcpOptionIp(body, dhcp.opt_server_id) orelse blk: {
const s = rdIp(body, dhcp.siaddr);
break :blk if (std.mem.eql(u8, &s, &ip_any)) src else s;
};
self.dhcp.state = .requesting;
self.dhcpSend(dhcp.request);
self.dhcp.tries = 1;
self.dhcp.retry_ms = self.now_ms + dhcp_backoff_ms[0];
},
dhcp.ack => {
switch (self.dhcp.state) {
.requesting, .renewing, .rebinding => {},
else => return,
}
const granted = rdIp(body, dhcp.yiaddr);
if (std.mem.eql(u8, &granted, &ip_any)) return;
self.dhcpBind(body, granted, src);
},
dhcp.nak => {
switch (self.dhcp.state) {
.requesting, .renewing, .rebinding => {},
else => return,
}
// RFC 2131 4.4.5: a NAK sends the client back to INIT. The lease is gone, so the
// address goes with it - continuing to use it would be squatting.
self.dhcpStart();
},
else => {},
}
}
fn dhcpBind(self: *Stack, body: []const u8, granted: Ip4, src: Ip4) void {
self.addr = granted;
self.mask = dhcpOptionIp(body, dhcp.opt_subnet_mask) orelse .{ 255, 255, 255, 0 };
self.gw = dhcpOptionIp(body, dhcp.opt_router) orelse ip_any;
self.dns = dhcpOptionIp(body, dhcp.opt_dns);
if (dhcpOptionIp(body, dhcp.opt_server_id)) |s| self.dhcp.server = s else if (std.mem.eql(u8, &self.dhcp.server, &ip_any)) {
self.dhcp.server = src;
}
// RFC 2131 3.3. A server that sends no lease time is out of spec; an hour is a safe
// assumption, being short enough that a wrong guess self-corrects.
const lease = dhcpOptionU32(body, dhcp.opt_lease_time) orelse 3600;
self.dhcp.lease_s = lease;
if (lease == 0xffff_ffff) {
// Infinite lease: never renew.
self.dhcp.t1_ms = std.math.maxInt(u64);
self.dhcp.t2_ms = std.math.maxInt(u64);
self.dhcp.expire_ms = std.math.maxInt(u64);
} else {
// The server may state T1 and T2 itself; otherwise lwIP's derivation, which is RFC
// 2131 4.4.5's: half the lease, and seven eighths of it
// (`core/ipv4/dhcp.c:757` and `:766`).
const t1 = dhcpOptionU32(body, dhcp.opt_t1) orelse lease / 2;
const t2 = dhcpOptionU32(body, dhcp.opt_t2) orelse (lease / 8) * 7;
const base = self.now_ms;
self.dhcp.t1_ms = base + @as(u64, @min(t1, lease)) * 1000;
self.dhcp.t2_ms = base + @as(u64, @min(t2, lease)) * 1000;
self.dhcp.expire_ms = base + @as(u64, lease) * 1000;
}
self.dhcp.state = .bound;
self.dhcp.tries = 0;
self.dhcp.retry_ms = 0;
self.announce();
}
fn dhcpTick(self: *Stack) void {
// T1 while bound: start renewing. RFC 2131 4.4.5 requires a fresh transaction id, and the
// first REQUEST goes out on this same tick rather than one backoff later - a state change
// that transmits nothing is how a lease quietly expires while the client thinks it is
// renewing.
if (self.dhcp.state == .bound and self.now_ms >= self.dhcp.t1_ms) {
self.dhcp.state = .renewing;
self.dhcp.xid = self.draw();
self.dhcp.started_ms = self.now_ms;
self.dhcp.tries = 0;
self.dhcp.retry_ms = 0;
}
switch (self.dhcp.state) {
.off, .bound => return,
.selecting, .requesting, .renewing, .rebinding => {},
}
if (self.dhcp.state == .renewing and self.now_ms >= self.dhcp.t2_ms) {
// T2: the granting server is not answering. Ask anyone.
self.dhcp.state = .rebinding;
self.dhcp.tries = 0;
self.dhcp.retry_ms = 0;
}
if ((self.dhcp.state == .renewing or self.dhcp.state == .rebinding) and
self.now_ms >= self.dhcp.expire_ms)
{
// The lease is over. Give up the address before asking again: keeping it would mean
// using an address the server may already have given away.
self.dhcpStart();
return;
}
if (self.dhcp.retry_ms != 0 and self.now_ms < self.dhcp.retry_ms) return;
const kind: u8 = if (self.dhcp.state == .selecting) dhcp.discover else dhcp.request;
self.dhcpSend(kind);
const idx = @min(self.dhcp.tries, dhcp_backoff_ms.len - 1);
self.dhcp.retry_ms = self.now_ms + dhcp_backoff_ms[idx];
if (self.dhcp.tries < 255) self.dhcp.tries += 1;
}
// ================================================================================= DNS
//
// One question, QTYPE=A, QCLASS=IN, recursion desired, over the UDP above. No cache, no
// search list, no NS or SOA handling, no TCP fallback on a truncated answer: this resolves
// the one name a device that fetches one URL has to resolve, and says so with a named error
// when it cannot.
//
// The hard part of DNS parsing is not the header, it is that a name in a resource record may
// be a compression pointer into anywhere earlier in the message (RFC 1035 4.1.4). A parser
// that follows those without a bound hangs on a message that points at itself, and such a
// message costs an attacker two bytes. `dnsSkipName` is where that is dealt with.
/// Encode a dotted name into RFC 1035 4.1.2 wire form: each label prefixed with its length,
/// terminated by the zero-length root label. Returns the encoded length.
///
/// `out` must be at least `dns_qname_max`, which the length check below makes sufficient: a
/// name of `n` text bytes with no trailing dot encodes to exactly `n + 2`.
fn dnsEncodeName(name: []const u8, out: []u8) DnsError!usize {
assert(out.len >= dns_qname_max);
if (name.len > dns_name_max) return error.NameTooLong;
// A trailing dot is the root label written out, and `example.com.` names the same node as
// `example.com`. Everything after it - an empty final label - is not.
var rest = name;
if (rest.len != 0 and rest[rest.len - 1] == '.') rest = rest[0 .. rest.len - 1];
if (rest.len == 0) return error.NameInvalid;
var o: usize = 0;
var labels = std.mem.splitScalar(u8, rest, '.');
while (labels.next()) |label| {
// An empty label inside a name (`a..b`, or a leading dot) is not a name.
if (label.len == 0 or label.len > dns.label_max) return error.NameInvalid;
out[o] = @intCast(label.len);
@memcpy(out[o + 1 ..][0..label.len], label);
o += 1 + label.len;
}
out[o] = 0;
return o + 1;
}
/// Compare an encoded name against the question we asked, ASCII-case-insensitively. RFC 4343:
/// label comparison ignores case, and a resolver is entitled to answer `0X4200.CAFE` to a
/// question about `0x4200.cafe`. Length bytes are 0-63 and so are never touched by the fold.
fn dnsQNameEql(a: []const u8, b: []const u8) bool {
if (a.len != b.len) return false;
for (a, b) |x, y| if (std.ascii.toLower(x) != std.ascii.toLower(y)) return false;
return true;
}
/// Step over the name at `start` and return the offset of the byte after it - which for a
/// name that ends in a compression pointer is two bytes after the pointer, *not* wherever the
/// pointer led. `null` means the name is unparseable and the message is to be rejected.
///
/// **Why this terminates.** Two independent bounds, because one of them is not enough:
///
/// * A pointer must point strictly backwards (`target < here`). That alone is the check
/// most implementations stop at, and it is *not* sufficient: after jumping back the walk
/// moves forward again over labels, so a pointer at offset 12 to offset 10 and a label at
/// 10 that is two bytes long lands back at 12, and the pair loops forever with every
/// individual jump going backwards.
/// * So the jumps themselves are counted, and `dns_max_jumps` of them ends the name. That
/// is the bound that actually holds: the loop below does at most `dns_max_jumps` jumps
/// and, between them, walks labels whose lengths are positive, so it visits at most
/// `dns_max_jumps * msg.len` bytes and stops. A legitimate answer uses one jump per name.
fn dnsSkipName(msg: []const u8, start: usize) ?usize {
var i = start;
var jumps: u8 = 0;
// The offset after the name in the *message*, fixed by the first pointer taken.
var after: ?usize = null;
while (true) {
if (i >= msg.len) return null;
const len = msg[i];
if (len & dns.ptr_mask == dns.ptr_mask) {
if (i + 1 >= msg.len) return null;
const target = (@as(usize, len & 0x3f) << 8) | msg[i + 1];
if (after == null) after = i + 2;
if (target >= i) return null;
jumps += 1;
if (jumps > dns_max_jumps) return null;
i = target;
continue;
}
// 0x40 and 0x80 are the reserved label types of RFC 1035 4.1.4 / RFC 6891; neither is
// something this stack can skip a known number of bytes past, so neither is accepted.
if (len & dns.ptr_mask != 0) return null;
if (len == 0) return after orelse i + 1;
i += 1 + @as(usize, len);
if (i > msg.len) return null;
}
}
/// Build and send the query held in `self.query`. Called for the first transmission and for
/// every retransmission, from the same fields, so the two cannot drift apart.
fn dnsSend(self: *Stack) void {
const src = self.addr orelse return;
const server = self.dns orelse return;
const qn_len: usize = self.query.qname_len;
const msg_len = dns.hlen + qn_len + 4;
const msg = self.tx[eth_hlen + ip4.hlen + udp.hlen ..][0..msg_len];
wr16(msg, dns.id, self.query.id);
// RD only. Not AD, not CD, not EDNS0: this asks a recursive resolver for one A record and
// has nothing to validate with.
wr16(msg, dns.flags, dns.flag_rd);
wr16(msg, dns.qdcount, 1);
wr16(msg, dns.ancount, 0);
wr16(msg, dns.nscount, 0);
wr16(msg, dns.arcount, 0);
@memcpy(msg[dns.hlen..][0..qn_len], self.query.qname[0..qn_len]);
wr16(msg, dns.hlen + qn_len, dns.type_a);
wr16(msg, dns.hlen + qn_len + 2, dns.class_in);
self.counters.dns_tx += 1;
_ = self.emitUdp(src, self.query.local_port, server, dns.port, msg_len);
}
fn dnsFail(self: *Stack, e: DnsError) void {
self.query.phase = .failed;
self.query.err = e;
}
/// A datagram to the port the outstanding query was sent from. Called from inside `onFrame`.
///
/// Everything that does not match the query is *ignored*, not failed: on a real network the
/// port this query owns will collect late answers to previous queries, scans, and whatever
/// else is loose on the segment, and any of those failing the query would be a denial of
/// service that costs one packet. Only a response that matches the source, the id and the
/// question can decide the query - and then it decides it either way.
fn dnsInput(self: *Stack, src: Ip4, sport: u16, msg: []const u8) void {
const server = self.dns orelse return;
if (!std.mem.eql(u8, &src, &server)) return;
if (sport != dns.port) return;
if (msg.len < dns.hlen) return;
if (rd16(msg, dns.id) != self.query.id) return;
const flags = rd16(msg, dns.flags);
if (flags & dns.flag_qr == 0) return; // a query, not a response
if (rd16(msg, dns.qdcount) != 1) return;
// The question, echoed. A server that answers a different question - or an attacker who
// guessed the id and the port but not the name - is not answering this.
const qn = self.query.qname[0..self.query.qname_len];
var off = dns.hlen + qn.len + 4;
if (msg.len < off) return;
if (!dnsQNameEql(msg[dns.hlen..][0..qn.len], qn)) return;
if (rd16(msg, dns.hlen + qn.len) != dns.type_a) return;
if (rd16(msg, dns.hlen + qn.len + 2) != dns.class_in) return;
self.counters.dns_rx += 1;
const rcode = flags & dns.rcode_mask;
if (rcode != 0) {
self.dnsFail(if (rcode == dns.rcode_name_error) error.NameNotFound else error.DnsRefused);
return;
}
// Walk the answer section and take the first A record. Walking rather than reading the
// first record is what makes a CNAME chain work: `0x4200.cafe` may answer with the CNAME
// and the A together, in that order, and a resolver that reads answer[0] gets a name.
var left = rd16(msg, dns.ancount);
while (left != 0) : (left -= 1) {
off = dnsSkipName(msg, off) orelse {
self.dnsFail(error.DnsMalformed);
return;
};
if (off + dns.rr_fixed > msg.len) {
self.dnsFail(error.DnsMalformed);
return;
}
const rtype = rd16(msg, off);
const rclass = rd16(msg, off + 2);
const rdlen: usize = rd16(msg, off + 8);
off += dns.rr_fixed;
if (off + rdlen > msg.len) {
self.dnsFail(error.DnsMalformed);
return;
}
if (rtype == dns.type_a and rclass == dns.class_in and rdlen == 4) {
self.query.result = rdIp(msg, off);
self.query.phase = .done;
return;
}
off += rdlen;
}
// A well-formed answer with no A record in it: NODATA, or a CNAME chain this stack will
// not chase a second query down.
self.dnsFail(error.NameNotFound);
}
fn dnsTick(self: *Stack) void {
if (self.query.phase != .waiting) return;
if (self.now_ms < self.query.retry_ms) return;
if (self.query.tries >= dns_backoff_ms.len) {
self.dnsFail(error.TimedOut);
return;
}
self.query.retry_ms = self.now_ms + dns_backoff_ms[self.query.tries];
self.query.tries += 1;
self.counters.dns_retx += 1;
self.dnsSend();
}
/// Resolve `name` to an IPv4 address.
///
/// **The protocol is `httpGet`'s, deliberately.** There is no clock and no `std.Io` here, so
/// there is nothing for a blocking call to block on: the first call sends the query and
/// returns `error.WouldBlock`, and the caller drives `tick` and `onFrame` and calls again
/// with the same name until an address or a real error comes back.
///
/// const addr = while (true) {
/// stack.tick(hal.systimer.millis());
/// while (transport.next()) |frame| stack.onFrame(frame);
/// if (stack.resolve("0x4200.cafe")) |a| break a
/// else |e| if (e != error.WouldBlock) return e;
/// };
///
/// The wait is bounded whether or not the caller bounds it: `dns_backoff_ms` retransmits
/// three times over 7 s and then answers `error.TimedOut`. Nothing here waits forever, and
/// the retransmissions happen in `tick`, so a caller that ticks and polls rarely still gets
/// them on time.
///
/// One query is outstanding at a time. A call naming something else while one is in flight is
/// `error.Busy`; a call naming something else after one has finished starts a new query,
/// which is what makes the loop above safe to write for two names in a row.
pub fn resolve(self: *Stack, name: []const u8) DnsError!Ip4 {
// Encoded first, and compared in encoded form: `0x4200.cafe`, `0x4200.cafe.` and
// `0X4200.CAFE` are one name, and a caller that spells it differently between two polls
// of the same loop must not get `error.Busy` for it.
var wire: [dns_qname_max]u8 = undefined;
const wire_len = try dnsEncodeName(name, &wire);
const same = self.query.qname_len == wire_len and
dnsQNameEql(self.query.qname[0..wire_len], wire[0..wire_len]);
switch (self.query.phase) {
.idle => {},
.waiting => {
if (!same) return error.Busy;
return error.WouldBlock;
},
// A finished query for this name is collected and the slot released. A finished query
// for a different name falls through and is replaced.
.done => if (same) {
self.query.phase = .idle;
return self.query.result;
},
.failed => if (same) {
self.query.phase = .idle;
return self.query.err;
},
}
if (self.addr == null) return error.NoAddress;
if (self.dns == null) return error.NoDnsServer;
@memcpy(self.query.qname[0..wire_len], wire[0..wire_len]);
self.query.qname_len = @intCast(wire_len);
self.query.id = @truncate(self.draw());
// RFC 6335's dynamic range, as `tcpConnect` uses. Redrawn per query so a late answer to
// the previous one cannot be mistaken for this one even if the id happens to repeat.
self.query.local_port = @intCast(49152 + self.draw() % (65535 - 49152 + 1));
self.query.result = ip_any;
self.query.err = error.WouldBlock;
self.query.phase = .waiting;
self.query.tries = 1;
self.query.retry_ms = self.now_ms + dns_backoff_ms[0];
self.dnsSend();
return error.WouldBlock;
}
// ================================================================================= TCP
//
// One connection, client side only, one unacknowledged segment at a time. The send side is a
// single static buffer holding the whole request, so "retransmission" is always "send from
// `snd_una` again" and there is no retransmission queue. The receive side has no reassembly
// buffer at all: a segment that is not the next one expected is answered with a duplicate ACK
// and dropped. Three of those is a fast-retransmit signal to any modern peer, so the common
// case of one lost segment costs a round trip rather than an RTO - but a reordered segment
// costs a retransmission that a reassembly buffer would have avoided. That is the price of not
// having one, and on a Wi-Fi link where reordering is rare it is the right price.
fn emitTcp(self: *Stack, flags: u8, seq: u32, data: []const u8, mss_opt: bool) bool {
const src = self.addr orelse return false;
const opt_len: usize = if (mss_opt) 4 else 0;
const seg_len = tcp.hlen + opt_len + data.len;
assert(seg_len <= mtu - ip4.hlen);
const seg = self.tx[eth_hlen + ip4.hlen ..][0..seg_len];
wr16(seg, tcp.src_port, self.tcp.local_port);
wr16(seg, tcp.dst_port, self.tcp.peer_port);
wr32(seg, tcp.seq, seq);
wr32(seg, tcp.ack, self.tcp.rcv_nxt);
const words: u16 = @intCast((tcp.hlen + opt_len) / 4);
wr16(seg, tcp.hdrlen_flags, (words << 12) | flags);
wr16(seg, tcp.window, self.rcvWindow());
wr16(seg, tcp.chksum, 0);
wr16(seg, tcp.urgent, 0);
if (mss_opt) {
seg[tcp.hlen] = tcp.opt_mss;
seg[tcp.hlen + 1] = 4;
wr16(seg, tcp.hlen + 2, tcp_mss);
}
if (data.len != 0) @memcpy(seg[tcp.hlen + opt_len ..], data);
wr16(seg, tcp.chksum, transportChecksum(src, self.tcp.peer_ip, .tcp, seg));
self.counters.tcp_tx += 1;
return self.emitIp(src, self.tcp.peer_ip, .tcp, seg_len);
}
/// The window to advertise: real back-pressure, not a constant. Everything accepted is consumed
/// synchronously into the HTTP head buffer or the caller's `out`, so the window is whatever
/// room is left there, capped at one MSS. Advertising a fixed window while the consumer had no
/// room left would turn "the response is bigger than your buffer" into a silently dropped
/// segment and an RTO storm.
fn rcvWindow(self: *Stack) u16 {
const room: usize = switch (self.http.phase) {
.head => (http_head_max - self.http.head_len) + self.http.out.len,
// Chunked framing - the CRLF closing each chunk, the zero-length chunk, the trailer
// section and the final CRLF - is consumed and discarded rather than delivered, so it
// needs window that `out` does not account for. Without this a body that exactly
// fills `out` closes the window before its own terminator can arrive, and the request
// stalls until the RTO gives up on a peer that is behaving perfectly.
.body => (self.http.out.len - self.http.out_len) +
@as(usize, if (self.http.chunked) http_framing_max + 16 else 0),
else => tcp_window,
};
return @intCast(@min(room, tcp_window));
}
/// Reset the connection and fail the request. RST is sent unless the peer sent one.
fn tcpAbort(self: *Stack, err: HttpError, send_rst: bool) void {
if (send_rst and self.tcp.state != .closed and self.tcp.state != .arp_wait) {
_ = self.emitTcp(tcp.rst | tcp.ack_f, self.tcp.snd_nxt, &.{}, false);
}
self.tcp.state = .closed;
if (self.http.phase == .head or self.http.phase == .body) {
self.http.phase = .failed;
self.http.err = err;
}
}
/// Set up the connection block for a fresh connect. Written field by field on purpose: the
/// obvious `self.tcp = .{ ... }` would assign `tx` from the struct's `undefined` default, which
/// in a safe build overwrites the request bytes with 0xAA, and in a release build memsets half
/// a kilobyte for nothing.
fn tcpConnect(self: *Stack, peer: Ip4, port: u16) void {
// A fresh ephemeral port every time. RFC 6335's dynamic range is 49152-65535, and moving
// through it is what makes the short TIME_WAIT above safe.
const span: u32 = 65535 - 49152 + 1;
const iss = self.draw();
self.tcp.state = .arp_wait;
self.tcp.peer_ip = peer;
self.tcp.peer_port = port;
self.tcp.local_port = @intCast(49152 + self.draw() % span);
self.tcp.iss = iss;
self.tcp.snd_una = iss;
self.tcp.snd_nxt = iss;
self.tcp.snd_wnd = 0;
self.tcp.snd_mss = tcp_default_mss;
self.tcp.fin_queued = false;
self.tcp.peer_fin = false;
self.tcp.rcv_nxt = 0;
self.tcp.rto_deadline_ms = 0;
self.tcp.rto_ms = tcp_rto_initial_ms;
self.tcp.retries = 0;
self.tcp.close_deadline_ms = 0;
self.tcp.tx_len = 0;
self.arp_tries = 0;
self.arp_retry_ms = 0;
}
fn tcpSendSyn(self: *Stack) void {
self.tcp.state = .syn_sent;
self.tcp.snd_nxt = self.tcp.iss +% 1;
_ = self.emitTcp(tcp.syn, self.tcp.iss, &.{}, true);
self.armRto();
}
fn armRto(self: *Stack) void {
self.tcp.rto_deadline_ms = self.now_ms + self.tcp.rto_ms;
}
/// Send as much of the request as the peer's window and MSS allow, then the FIN if the whole
/// request has gone out. One segment in flight, so this sends at most one segment per call.
///
/// Only `established` sends: receiving the peer's FIN does not move the state, it sets
/// `peer_fin`, so this stays the single place that decides what goes on the wire and the state
/// only ever changes when a FIN of ours actually leaves.
fn tcpSendData(self: *Stack) void {
if (self.tcp.state != .established) return;
// Nothing outstanding is the precondition for sending: this is the fixed window of one.
if (seqLt(self.tcp.snd_una, self.tcp.snd_nxt)) return;
const end = self.tcp.dataEnd();
if (seqLt(self.tcp.snd_nxt, end)) {
const off: usize = self.tcp.snd_nxt -% self.tcp.dataStart();
const remaining = self.tcp.tx_len - off;
const window: usize = self.tcp.snd_una +% self.tcp.snd_wnd -% self.tcp.snd_nxt;
const n = @min(@min(remaining, self.tcp.snd_mss), @max(window, 1));
// PSH on the last segment of the request: the peer's application should see it without
// waiting for more. RFC 793 has no requirement here; every HTTP server expects it.
const last = off + n == self.tcp.tx_len;
const flags: u8 = tcp.ack_f | (if (last) tcp.psh else 0);
const seq = self.tcp.snd_nxt;
self.tcp.snd_nxt = seq +% @as(u32, @intCast(n));
_ = self.emitTcp(flags, seq, self.tcp.tx[off..][0..n], false);
self.armRto();
return;
}
if (self.tcp.fin_queued and self.tcp.snd_nxt == end) {
const seq = self.tcp.snd_nxt;
self.tcp.snd_nxt = seq +% 1;
_ = self.emitTcp(tcp.fin | tcp.ack_f, seq, &.{}, false);
self.armRto();
// RFC 793's FIN-WAIT-1 if we closed first, its CLOSING/LAST-ACK if the peer did. Both
// of the latter are `last_ack` here: they differ only in which ACK is still owed, and
// `closeCheck` settles that from the sequence numbers.
self.tcp.state = if (self.tcp.peer_fin) .last_ack else .fin_wait_1;
}
}
/// Half-close: everything we mean to send has been sent, so send FIN once the data is out.
fn tcpFinish(self: *Stack) void {
if (self.tcp.fin_queued) return;
self.tcp.fin_queued = true;
self.tcpSendData();
}
/// Both directions closed and our FIN acknowledged: nothing is left in flight, so the
/// connection block can be released after TIME_WAIT. Called once at the end of every segment,
/// which covers both orders of arrival - the peer's FIN then its ACK, or the reverse.
fn closeCheck(self: *Stack) void {
switch (self.tcp.state) {
.fin_wait_1, .fin_wait_2, .last_ack => {},
else => return,
}
if (!self.tcp.peer_fin) return;
if (self.tcp.snd_una != self.tcp.snd_nxt) return;
self.tcp.state = .time_wait;
self.tcp.rto_deadline_ms = 0;
self.tcp.close_deadline_ms = self.now_ms + tcp_time_wait_ms;
}
fn tcpInput(self: *Stack, src: Ip4, dst: Ip4, seg: []const u8) void {
if (seg.len < tcp.hlen) {
self.counters.rx_dropped += 1;
return;
}
const hf = rd16(seg, tcp.hdrlen_flags);
const hlen = @as(usize, hf >> 12) * 4;
if (hlen < tcp.hlen or hlen > seg.len) {
self.counters.rx_dropped += 1;
return;
}
if (!transportChecksumOk(src, dst, .tcp, seg, rd16(seg, tcp.chksum))) {
self.counters.checksum_bad += 1;
return;
}
const flags: u8 = @truncate(hf & 0x3f);
const sport = rd16(seg, tcp.src_port);
const dport = rd16(seg, tcp.dst_port);
if (self.tcp.state == .closed or
dport != self.tcp.local_port or
sport != self.tcp.peer_port or
!std.mem.eql(u8, &src, &self.tcp.peer_ip))
{
// Not for our one connection. A real stack would RST; a client with no listening port
// gains nothing by telling a scanner it is there.
self.counters.rx_dropped += 1;
return;
}
self.counters.tcp_rx += 1;
const seq = rd32(seg, tcp.seq);
const ackno = rd32(seg, tcp.ack);
const data = seg[hlen..];
if (flags & tcp.rst != 0) {
self.counters.tcp_rst_rx += 1;
// RFC 5961 3: only a RST whose sequence number is the next one expected may tear the
// connection down. Anything else gets a challenge ACK, which is also what stops a
// blind off-path reset.
if (self.tcp.state == .syn_sent) {
// In SYN-SENT the RST is validated by its ACK instead: there is no rcv_nxt yet.
if (flags & tcp.ack_f != 0 and ackno == self.tcp.snd_nxt) self.tcpAbort(error.ConnectionReset, false);
return;
}
if (seq == self.tcp.rcv_nxt) {
self.tcpAbort(error.ConnectionReset, false);
} else {
_ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false);
}
return;
}
if (self.tcp.state == .syn_sent) {
if (flags & tcp.syn == 0) {
self.counters.rx_dropped += 1;
return;
}
if (flags & tcp.ack_f == 0) {
// A simultaneous open. Nothing on the other end of an HTTP GET does this.
self.counters.rx_dropped += 1;
return;
}
if (ackno != self.tcp.iss +% 1) {
// Not acknowledging our SYN: an old duplicate. RFC 793 says reset it.
_ = self.emitTcp(tcp.rst, ackno, &.{}, false);
return;
}
self.tcp.rcv_nxt = seq +% 1;
self.tcp.snd_una = ackno;
self.tcp.snd_wnd = rd16(seg, tcp.window);
self.tcp.snd_mss = parseMss(seg[tcp.hlen..hlen]) orelse tcp_default_mss;
self.tcp.state = .established;
self.tcp.rto_ms = tcp_rto_initial_ms;
self.tcp.retries = 0;
// The ACK completing the handshake carries the first data segment, which is one frame
// saved and what every other stack does.
self.tcp.rto_deadline_ms = 0;
self.tcpSendData();
if (self.tcp.snd_nxt == self.tcp.snd_una) {
// Nothing to send yet; the handshake still needs acknowledging.
_ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false);
}
return;
}
// A duplicate SYN in an established connection is either a retransmitted SYN whose ACK was
// lost - answer with an ACK - or an attack. Never a reason to re-open.
if (flags & tcp.syn != 0 and seqLt(seq, self.tcp.rcv_nxt)) {
_ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false);
return;
}
if (flags & tcp.ack_f != 0) self.tcpAck(ackno, rd16(seg, tcp.window));
// ---- receive side
var payload = data;
var accept = false;
if (payload.len != 0) {
if (seqLe(seq, self.tcp.rcv_nxt) and seqGt(seq +% @as(u32, @intCast(payload.len)), self.tcp.rcv_nxt)) {
// Overlaps what we already have: trim the duplicate prefix. A retransmission after
// a lost ACK arrives exactly like this, and rejecting it would deadlock.
const skip: usize = self.tcp.rcv_nxt -% seq;
payload = payload[skip..];
accept = true;
} else if (seqLe(seq +% @as(u32, @intCast(payload.len)), self.tcp.rcv_nxt)) {
// Wholly old. Re-acknowledge so the peer stops.
_ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false);
return;
} else {
// Out of order, and there is nowhere to keep it. The duplicate ACK below is the
// signal that makes the peer resend.
_ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false);
return;
}
}
if (accept) {
// Never accept more than the window we advertised.
const room = self.rcvWindow();
if (payload.len > room) payload = payload[0..room];
self.tcp.rcv_nxt +%= @intCast(payload.len);
// Consuming the data may itself put a segment on the wire - completing the body sends
// our FIN - and every segment carries `rcv_nxt`, so a separate ACK would be a wasted
// frame. Counting is the honest way to know: anything emitted has already acknowledged
// this data, and nothing emitted means we still owe an ACK.
const tx_before = self.counters.tcp_tx;
self.httpOnData(payload);
if (self.tcp.state == .closed) return; // httpOnData failed and aborted
if (self.counters.tcp_tx == tx_before) {
_ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false);
}
}
// ---- FIN, in order only. An out-of-order FIN names a sequence number beyond data we have
// not seen, and honouring it would close the connection over a hole.
if (flags & tcp.fin != 0) {
const fin_seq = seq +% @as(u32, @intCast(data.len));
const in_order = fin_seq == self.tcp.rcv_nxt;
// A FIN we have already consumed, arriving again because our ACK was lost. It must be
// re-acknowledged or the peer retransmits until it gives up and resets.
const duplicate = self.tcp.peer_fin and fin_seq +% 1 == self.tcp.rcv_nxt;
if (in_order and !self.tcp.peer_fin) {
self.tcp.rcv_nxt +%= 1;
self.tcp.peer_fin = true;
self.httpOnEof();
}
if (in_order or duplicate) {
// Our own FIN, if it has not gone yet, acknowledges the peer's on the way out.
const tx_before = self.counters.tcp_tx;
self.tcpFinish();
if (self.counters.tcp_tx == tx_before) {
_ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false);
}
}
}
self.closeCheck();
}
fn tcpAck(self: *Stack, ackno: u32, window: u16) void {
// An ACK ahead of what we sent is invalid; an old one is a duplicate.
if (seqGt(ackno, self.tcp.snd_nxt)) return;
self.tcp.snd_wnd = window;
if (seqLe(ackno, self.tcp.snd_una)) {
// Duplicate ACK. With one segment in flight there is nothing to fast-retransmit.
return;
}
self.tcp.snd_una = ackno;
self.tcp.retries = 0;
self.tcp.rto_ms = tcp_rto_initial_ms;
if (self.tcp.snd_una == self.tcp.snd_nxt) {
self.tcp.rto_deadline_ms = 0; // nothing outstanding
} else {
self.armRto();
}
if (self.tcp.state == .fin_wait_1 and self.tcp.snd_una == self.tcp.snd_nxt) {
self.tcp.state = .fin_wait_2;
self.tcp.close_deadline_ms = self.now_ms + tcp_fin_wait2_ms;
}
// Window opened or data acknowledged: there may be more to send.
self.tcpSendData();
}
/// RFC 793 3.1 option format: kind, then for kinds above 1 a length byte covering both.
fn parseMss(opts: []const u8) ?u16 {
var i: usize = 0;
while (i < opts.len) {
const kind = opts[i];
if (kind == tcp.opt_end) return null;
if (kind == tcp.opt_nop) {
i += 1;
continue;
}
if (i + 2 > opts.len) return null;
const len = opts[i + 1];
if (len < 2 or i + len > opts.len) return null;
if (kind == tcp.opt_mss and len == 4) {
const v = rd16(opts, i + 2);
// Below RFC 1122's floor a peer's MSS is not believable; above our MTU it cannot
// be honoured anyway.
return @min(@max(v, 64), tcp_mss);
}
i += len;
}
return null;
}
fn tcpTick(self: *Stack) void {
switch (self.tcp.state) {
.closed => {},
.arp_wait => {
if (self.arpLookup(self.tcpNextHop())) |_| {
self.tcpSendSyn();
return;
}
if (self.arp_retry_ms != 0 and self.now_ms < self.arp_retry_ms) return;
if (self.arp_tries >= arp_max_tries) {
self.tcpAbort(error.HostUnreachable, false);
return;
}
self.arpRequest(self.tcpNextHop());
self.arp_tries += 1;
self.arp_retry_ms = self.now_ms + arp_retry_ms;
},
.fin_wait_2 => {
// Our FIN is acknowledged and nothing is outstanding, so there is no RTO to run:
// the only thing left is the peer's FIN, and this is how long we wait for it.
if (self.now_ms >= self.tcp.close_deadline_ms) self.tcp.state = .closed;
},
.time_wait => {
if (self.now_ms >= self.tcp.close_deadline_ms) self.tcp.state = .closed;
},
else => {
if (self.tcp.rto_deadline_ms == 0) return;
if (self.now_ms < self.tcp.rto_deadline_ms) return;
if (self.tcp.retries >= tcp_max_retries) {
self.tcpAbort(error.TimedOut, true);
return;
}
self.tcp.retries += 1;
self.counters.tcp_retx += 1;
// Exponential backoff, RFC 6298 5.5.
self.tcp.rto_ms = @min(self.tcp.rto_ms * 2, tcp_rto_max_ms);
self.tcpRetransmit();
},
}
}
fn tcpNextHop(self: *Stack) Ip4 {
return if (self.onLink(self.tcp.peer_ip)) self.tcp.peer_ip else self.gw;
}
/// Go back to `snd_una` and send again. With one segment in flight this is the whole of
/// retransmission: there is no queue to walk and no partial-ACK case to handle.
fn tcpRetransmit(self: *Stack) void {
const una = self.tcp.snd_una;
if (una == self.tcp.iss) {
// The SYN. Its MSS option must be repeated: a peer that only ever sees the
// retransmission would otherwise assume 536.
self.tcp.snd_nxt = self.tcp.iss;
self.tcpSendSyn();
return;
}
const end = self.tcp.dataEnd();
if (seqLt(una, end)) {
self.tcp.snd_nxt = una;
self.tcpSendData();
return;
}
if (self.tcp.fin_queued and una == end) {
self.tcp.snd_nxt = una;
// `tcpSendData` re-sends the FIN and re-arms, but it refuses to run in FIN_WAIT_1
// (which is where a lost FIN leaves us), so the segment is emitted directly.
_ = self.emitTcp(tcp.fin | tcp.ack_f, una, &.{}, false);
self.tcp.snd_nxt = una +% 1;
self.armRto();
return;
}
// Nothing identifiable outstanding: a bare ACK, which costs one frame and cannot hurt.
_ = self.emitTcp(tcp.ack_f, self.tcp.snd_nxt, &.{}, false);
self.armRto();
}
// ================================================================================ HTTP
/// Fetch `path` from `host:port` over HTTP/1.1 and write the body to `out`, sending the
/// address literal as the `Host:` header. Exactly `httpGetHost(host, null, ...)`; see there
/// for the protocol, which is the whole of how this is used.
pub fn httpGet(self: *Stack, host: [4]u8, port: u16, path: []const u8, out: []u8) HttpError!usize {
return self.httpGetHost(host, null, port, path, out);
}
/// Fetch `path` from `host:port` over HTTP/1.1 and write the body to `out`.
///
/// `name` is the `Host:` header. `null` sends the address literal - `Host: 192.168.1.90` -
/// which is right for a bare address and is what `httpGet` does. A name is what a
/// name-based virtual host requires: one address behind a CDN serves thousands of sites and
/// picks between them on this header alone, so `Host: 104.21.46.8` gets the CDN's own error
/// page and never the site. The address is still where the connection goes; the name only
/// ever appears in the header, and nothing here resolves it - `resolve` does that, and the
/// two are separate because a caller may have the address already.
///
/// The port is appended as `:port` only when it is not 80, name or no name. RFC 7230 5.4.
///
/// **This does not block, and it is not a one-shot call.** There is no `std.Io` here and no
/// clock, so there is nothing for a blocking call to block on: the frames that carry the
/// response arrive through `onFrame` and time advances through `tick`, both of which are the
/// caller's to drive. So the first call starts the request and returns `error.WouldBlock`, and
/// the caller keeps driving and keeps calling with the same arguments until it returns a length:
///
/// while (true) {
/// stack.tick(hal.systimer.millis());
/// while (transport.next()) |frame| stack.onFrame(frame);
/// if (stack.httpGetHost(addr, "0x4200.cafe", 80, "/", &buf)) |n| break :done buf[0..n]
/// else |e| if (e != error.WouldBlock) return e;
/// }
///
/// `out` is borrowed until the request completes: it is written to from inside `onFrame` as the
/// body arrives, so it must not move or be reused meanwhile. Calling with different arguments
/// while a request is in flight returns `error.Busy` rather than quietly abandoning the first,
/// and `name` is one of those arguments: two requests to one address for one path but
/// different virtual hosts are different requests.
///
/// `Content-Length` is honoured, and so is `Transfer-Encoding: chunked` - the body handed back
/// is decoded, with no framing bytes in it. A response with neither ends at the peer's FIN,
/// which is why the request says `Connection: close`.
pub fn httpGetHost(
self: *Stack,
host: [4]u8,
name: ?[]const u8,
port: u16,
path: []const u8,
out: []u8,
) HttpError!usize {
// The name is folded into the path hash rather than given a field of its own: `Stack` has
// a 4 KiB budget, and what this has to distinguish is "the same call again" from "a
// different call", which a hash does exactly. Seeding with the name's hash rather than
// concatenating keeps `null` (seed 0) distinct from any name, including the empty one.
const req_hash = std.hash.Wyhash.hash(
if (name) |nm| std.hash.Wyhash.hash(0x486f_7374, nm) else 0,
path,
);
switch (self.http.phase) {
.idle => {},
.head, .body => {
if (!std.mem.eql(u8, &self.http.req_host, &host) or
self.http.req_port != port or
self.http.req_hash != req_hash or
self.http.out.ptr != out.ptr or
self.http.out.len != out.len) return error.Busy;
return error.WouldBlock;
},
.complete => {
const n = self.http.out_len;
self.http.phase = .idle;
return n;
},
.failed => {
const e = self.http.err;
self.http.phase = .idle;
return e;
},
}
if (self.addr == null) return error.NoAddress;
// The request, built once into the TCP send buffer where it stays until acknowledged.
var w: RequestWriter = .{ .buf = &self.tcp.tx };
w.str("GET ");
w.str(if (path.len == 0) "/" else path);
w.str(" HTTP/1.1\r\nHost: ");
if (name) |nm| w.str(nm) else w.ipv4(host);
if (port != 80) {
w.str(":");
w.dec(port);
}
// Connection: close is not politeness, it is the framing: it is what makes a response with
// no Content-Length terminable, and it is what makes the peer's FIN the end of the body.
w.str("\r\nUser-Agent: zig-p4/0.1\r\nAccept: */*\r\nConnection: close\r\n\r\n");
if (w.overflow) return error.RequestTooLong;
self.http = .{
.phase = .head,
.out = out,
.req_host = host,
.req_port = port,
.req_hash = req_hash,
};
self.tcpConnect(host, port);
self.tcp.tx_len = w.i;
self.tcp.fin_queued = false;
// A MAC address may already be known, in which case the SYN goes out now rather than one
// `tick` later.
if (self.arpLookup(self.tcpNextHop()) != null) {
self.tcpSendSyn();
} else {
self.arpRequest(self.tcpNextHop());
self.arp_tries = 1;
self.arp_retry_ms = self.now_ms + arp_retry_ms;
}
return error.WouldBlock;
}
/// A bounds-checked append into a fixed buffer. Overflow is recorded, not asserted: a caller's
/// long path is a request error, not a bug in this file.
const RequestWriter = struct {
buf: []u8,
i: usize = 0,
overflow: bool = false,
fn str(self: *RequestWriter, s: []const u8) void {
if (self.overflow or self.i + s.len > self.buf.len) {
self.overflow = true;
return;
}
@memcpy(self.buf[self.i..][0..s.len], s);
self.i += s.len;
}
fn dec(self: *RequestWriter, v: u32) void {
var tmp: [10]u8 = undefined;
var n: usize = 0;
var x = v;
while (true) {
tmp[n] = '0' + @as(u8, @intCast(x % 10));
n += 1;
x /= 10;
if (x == 0) break;
}
while (n > 0) {
n -= 1;
self.str(tmp[n .. n + 1]);
}
}
fn ipv4(self: *RequestWriter, a: Ip4) void {
for (a, 0..) |b, k| {
if (k != 0) self.str(".");
self.dec(b);
}
}
};
/// Fail the request and reset the connection. The phase is set before `tcpAbort`, which would
/// otherwise overwrite `err` with its own argument on the way past.
fn httpFail(self: *Stack, e: HttpError) void {
self.http.phase = .failed;
self.http.err = e;
self.tcpAbort(e, true);
}
/// In-order TCP payload. Called from inside `onFrame`.
fn httpOnData(self: *Stack, bytes: []const u8) void {
var rest = bytes;
if (self.http.phase == .head) {
const room = http_head_max - self.http.head_len;
const n = @min(room, rest.len);
@memcpy(self.http.head[self.http.head_len..][0..n], rest[0..n]);
const scan_from = self.http.head_len -| 3;
self.http.head_len += n;
rest = rest[n..];
const blank = std.mem.indexOfPos(u8, self.http.head[0..self.http.head_len], scan_from, "\r\n\r\n") orelse {
if (self.http.head_len == http_head_max) self.httpFail(error.HttpHeadersTooLong);
return;
};
const head_end = blank + 4;
// Anything the head buffer swallowed past the blank line is body. This is the case a
// test has to cover deliberately, because it only happens when a segment boundary does
// not coincide with the end of the headers - which on a real server is most of the time.
const spill = self.http.head[head_end..self.http.head_len];
self.parseHead(self.http.head[0..blank]) catch |e| {
self.httpFail(e);
return;
};
self.http.phase = .body;
// `spill` aliases `self.http.head`, and `httpBody` only ever writes to `self.http.out`,
// so passing it through is safe. Copy first if that ever stops being true.
//
// It is called unconditionally, even when `spill` is empty: that is what completes a
// `Content-Length: 0` response, whose body is over the moment its headers are.
self.httpBody(spill);
if (self.http.phase != .body) return;
}
if (rest.len != 0) self.httpBody(rest);
}
/// Status line and headers, without the terminating blank line.
fn parseHead(self: *Stack, head: []const u8) HttpError!void {
var lines = std.mem.splitSequence(u8, head, "\r\n");
const status_line = lines.next() orelse return error.HttpMalformed;
// "HTTP/1.1 200 OK": version, space, three digits.
if (status_line.len < 12) return error.HttpMalformed;
if (!std.mem.startsWith(u8, status_line, "HTTP/1.")) return error.HttpMalformed;
if (status_line[8] != ' ') return error.HttpMalformed;
var code: u16 = 0;
for (status_line[9..12]) |c| {
if (c < '0' or c > '9') return error.HttpMalformed;
code = code * 10 + (c - '0');
}
self.http.status = code;
self.http.content_length = null;
self.http.chunked = false;
while (lines.next()) |line| {
if (line.len == 0) continue;
const colon = std.mem.indexOfScalar(u8, line, ':') orelse continue;
const name = line[0..colon];
const value = std.mem.trim(u8, line[colon + 1 ..], " \t");
// RFC 7230 3.2: field names are case-insensitive. Servers vary, and a stack that
// compares them exactly works against nginx and fails against something else.
if (std.ascii.eqlIgnoreCase(name, "content-length")) {
self.http.content_length = std.fmt.parseInt(usize, value, 10) catch
return error.HttpMalformed;
} else if (std.ascii.eqlIgnoreCase(name, "transfer-encoding")) {
// RFC 7230 3.3.1: the final coding decides the framing. Exactly two are
// understood - `chunked`, which frames the body, and `identity`, which does not -
// and a list, or a coding that transforms the bytes, is refused. Guessing at
// `gzip` would hand the caller compressed data and call it a body.
if (std.ascii.eqlIgnoreCase(value, "chunked")) {
self.http.chunked = true;
} else if (!std.ascii.eqlIgnoreCase(value, "identity")) {
return error.UnsupportedTransferEncoding;
}
}
}
if (self.http.chunked) {
// RFC 7230 3.3.3 case 3: when both are present the chunked framing wins and
// `Content-Length` must be ignored - it is the classic request-smuggling
// disagreement, and a response that carries both is not to be believed twice.
self.http.content_length = null;
self.http.chunk = .size;
self.http.chunk_left = 0;
self.http.chunk_digit = false;
self.http.chunk_skip = 0;
}
// A response whose body cannot possibly fit is refused now rather than after copying most
// of it: the caller gets a clean error instead of a truncated buffer. A chunked response
// announces no total, so its equivalent check is per chunk, in `httpChunkedBody`.
if (self.http.content_length) |len| {
if (len > self.http.out.len) return error.StreamTooLong;
}
}
fn httpBody(self: *Stack, bytes: []const u8) void {
if (self.http.chunked) return self.httpChunkedBody(bytes);
var b = bytes;
if (self.http.content_length) |len| {
const want = len - self.http.out_len;
if (b.len > want) b = b[0..want];
}
if (self.http.out_len + b.len > self.http.out.len) {
self.httpFail(error.StreamTooLong);
return;
}
@memcpy(self.http.out[self.http.out_len..][0..b.len], b);
self.http.out_len += b.len;
if (self.http.content_length) |len| {
if (self.http.out_len >= len) self.httpComplete();
}
}
/// Charge `n` bytes against the framing budget. False means the request has been failed and
/// the decoder must stop.
fn chunkSkip(self: *Stack, n: usize) bool {
const total = @as(usize, self.http.chunk_skip) + n;
if (total > http_framing_max) {
self.httpFail(error.HttpHeadersTooLong);
return false;
}
self.http.chunk_skip = @intCast(total);
return true;
}
/// RFC 7230 4.1 chunked decoding, resumable between any two bytes.
///
/// The decoder's whole position lives in `http.chunk`, `chunk_left`, `chunk_digit` and
/// `chunk_skip`, and `bytes` is whatever the last segment happened to carry. Nothing is
/// buffered and nothing is looked ahead at: a size split across two segments accumulates a
/// digit at a time, a CRLF split across two segments is two states, and a chunk's data is
/// copied out as it arrives however it is cut up. That is not a hypothetical - a 1,460-byte
/// segment ends where the server's writes ended, which is nowhere in particular.
///
/// `out` receives decoded data only. No size, no extension, no CRLF and no trailer byte is
/// ever copied into it, and every failure is a named error rather than a short body.
fn httpChunkedBody(self: *Stack, bytes: []const u8) void {
var b = bytes;
while (b.len != 0) {
switch (self.http.chunk) {
.size => {
const c = b[0];
const digit: ?u8 = switch (c) {
'0'...'9' => c - '0',
'a'...'f' => c - 'a' + 10,
'A'...'F' => c - 'A' + 10,
else => null,
};
b = b[1..];
if (digit) |d| {
// Checked, not truncated: a size that does not fit `usize` is a malformed
// message, and wrapping it would turn a hostile header into a short read
// that looks like a complete body.
if (self.http.chunk_left > (std.math.maxInt(usize) - @as(usize, d)) / 16) {
self.httpFail(error.HttpChunkMalformed);
return;
}
self.http.chunk_left = self.http.chunk_left * 16 + d;
self.http.chunk_digit = true;
continue;
}
// RFC 7230 4.1 is `1*HEXDIG`. Without this an empty line reads as a chunk of
// size zero, which is the terminator, which ends the body early.
if (!self.http.chunk_digit) {
self.httpFail(error.HttpChunkMalformed);
return;
}
self.http.chunk_skip = 0;
switch (c) {
';' => self.http.chunk = .ext,
'\r' => self.http.chunk = .size_lf,
else => {
self.httpFail(error.HttpChunkMalformed);
return;
},
}
},
.ext => {
// chunk-ext is skipped whole: nothing here depends on one, so the only thing
// that matters is finding the CR that ends the header - possibly not in this
// segment at all.
const cr = std.mem.indexOfScalar(u8, b, '\r');
const n = cr orelse b.len;
if (!self.chunkSkip(n)) return;
b = b[n..];
if (cr != null) {
b = b[1..];
self.http.chunk = .size_lf;
}
},
.size_lf => {
if (b[0] != '\n') {
self.httpFail(error.HttpChunkMalformed);
return;
}
b = b[1..];
if (self.http.chunk_left == 0) {
// The zero-length chunk. What follows is the trailer section, and the
// body is not complete until its final CRLF.
self.http.chunk_skip = 0;
self.http.chunk = .trailer;
} else {
// Refused on the header rather than part-way through the copy, which is
// what `Content-Length` gets: the caller sees the error before the buffer
// has been half filled with a body it will never be given.
if (self.http.chunk_left > self.http.out.len - self.http.out_len) {
self.httpFail(error.StreamTooLong);
return;
}
self.http.chunk = .data;
}
},
.data => {
// In bounds by construction: `.size_lf` refused any chunk larger than the room
// left, and this only ever takes `chunk_left` of it.
const n = @min(self.http.chunk_left, b.len);
@memcpy(self.http.out[self.http.out_len..][0..n], b[0..n]);
self.http.out_len += n;
self.http.chunk_left -= n;
b = b[n..];
if (self.http.chunk_left == 0) self.http.chunk = .data_cr;
},
.data_cr => {
if (b[0] != '\r') {
self.httpFail(error.HttpChunkMalformed);
return;
}
b = b[1..];
self.http.chunk = .data_lf;
},
.data_lf => {
if (b[0] != '\n') {
self.httpFail(error.HttpChunkMalformed);
return;
}
b = b[1..];
// `.data` is only ever left with the chunk exhausted, so the accumulator the
// next size builds in already reads zero and is not re-zeroed here. Asserted
// rather than assumed: re-zeroing would be dead code that hides the day the
// invariant stops holding, and a stale count would be silent.
assert(self.http.chunk_left == 0);
self.http.chunk_digit = false;
self.http.chunk = .size;
},
.trailer => {
if (!self.chunkSkip(1)) return;
const cr = b[0] == '\r';
b = b[1..];
self.http.chunk = if (cr) .end_lf else .trailer_line;
},
.trailer_line => {
const cr = std.mem.indexOfScalar(u8, b, '\r');
const n = cr orelse b.len;
if (!self.chunkSkip(n)) return;
b = b[n..];
if (cr != null) {
b = b[1..];
self.http.chunk = .trailer_lf;
}
},
.trailer_lf => {
if (b[0] != '\n') {
self.httpFail(error.HttpChunkMalformed);
return;
}
b = b[1..];
self.http.chunk = .trailer;
},
.end_lf => {
if (b[0] != '\n') {
self.httpFail(error.HttpChunkMalformed);
return;
}
self.httpComplete();
// Anything after the final CRLF belongs to a response this connection will
// never ask for: `Connection: close` was sent, and the FIN follows.
return;
},
}
}
}
fn httpComplete(self: *Stack) void {
self.http.phase = .complete;
// The body is in hand; close our half. Reading further would only cost frames.
self.tcpFinish();
}
/// The peer closed. Whether that completes the response depends on the framing.
fn httpOnEof(self: *Stack) void {
switch (self.http.phase) {
.body => {
if (self.http.chunked) {
// The zero-length chunk and its trailer never arrived. RFC 7230 4.1 makes
// them the framing, so a close before them is a truncated body, however many
// whole chunks came first - reporting what did arrive would be reporting a
// prefix as the whole.
self.http.phase = .failed;
self.http.err = error.ConnectionClosed;
} else if (self.http.content_length) |len| {
if (self.http.out_len >= len) {
self.http.phase = .complete;
} else {
// Fewer body bytes than Content-Length promised.
self.http.phase = .failed;
self.http.err = error.ConnectionClosed;
}
} else {
// No Content-Length: the FIN *is* the framing (RFC 7230 3.3.3 case 7).
self.http.phase = .complete;
}
},
.head => {
self.http.phase = .failed;
self.http.err = error.ConnectionClosed;
},
else => {},
}
}
// ================================================================================ tick
/// Advance time. Drives DHCP retransmission and renewal, ARP resolution and TCP
/// retransmission. `now_ms` must be monotonic; it need not start at zero and it need not be
/// called at any particular rate, but nothing times out between calls, so a 47-second TCP
/// deadline needs ticks more often than every 47 seconds to be observed on time.
pub fn tick(self: *Stack, now_ms: u64) void {
self.now_ms = now_ms;
// Stir. The MAC alone would make every boot draw the same transaction ids, initial sequence
// numbers and ephemeral ports, which is how two runs of the same firmware end up accepting
// each other's stale DHCP replies. `now_ms` is the only outside input this file has, and a
// caller that ticks a real timer before starting DHCP therefore gets a different sequence
// every boot. Still not a source of security-relevant randomness - see `entropy`.
self.entropy ^= now_ms *% 0x9e37_79b9_7f4a_7c15;
self.dhcpTick();
self.dnsTick();
self.tcpTick();
}
};
// The footprint claim, enforced at compile time, so a buffer that grows fails the build rather than
// the board.
//
// 4 KiB is the budget and it is measured, not guessed: the image has ~128 KB of L2MEM, nothing
// initialises the 32 MB of PSRAM, and ESP-Hosted's queues and its seven task stacks are competing
// for the same space. `Stack` is 3,576 bytes today. The failure this prevents is a stack overflow
// on a part with no debugger, which is indistinguishable from the SDIO bus not coming up.
comptime {
// 6 KiB, raised from 4 KiB when `http_head_max` went from 1024 to 2048 to fit a real CDN
// response head (1043 bytes measured). This is a regression alarm, not a budget: it exists so a
// buffer cannot grow unnoticed, and moving it is a decision to be justified at the buffer that
// caused it - which the comment on `http_head_max` does. The image's real constraint is the
// ~128 KB of L2MEM, and the heap in examples/http.zig was reduced by the same amount to pay for
// this.
if (Stack.footprint > 6 * 1024) @compileError(std.fmt.comptimePrint(
"ip.Stack is {d} bytes, over the 4 KiB budget",
.{Stack.footprint},
));
}
// The host tests live in `ip_test.zig` - 117 cases, and they are the correctness argument for this
// slice, since it is the one part of the P4 bring-up that can be proven without the board. They are
// in their own file because they are longer than the stack, and because the tests deliberately
// re-derive every header offset from the RFCs rather than importing the tables above: a test that
// shares the constant it is checking passes on a consistent misreading.
//
// This reference is what makes `zig build test` find them: build.zig runs `src/net/ip.zig` as a
// test root, and Zig only collects tests from files the root actually references.
test {
_ = @import("ip_test.zig");
}
|