Repository navigation
Expand file tree
/
Copy patheth_litex.c
More file actions
1175 lines (987 loc) · 36.1 KB
/
Copy patheth_litex.c
File metadata and controls
1175 lines (987 loc) · 36.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
/*
* LiteX SoC Ethernet driver for AES67 FPGA
*
* Accesses the FPGA via:
* - AES67_CSR registers (control/status)
* - ETH_BUF CSR registers (EventManager + rx_len/ack/tx_len)
* - ETH_BUF memory (Wishbone-mapped RX/TX packet buffers)
* - IRQ (eth_buf EventManager rx_ready)
*
* All addresses are imported from LiteX-generated headers (csr.h, mem.h, soc.h).
*/
#include <zephyr/kernel.h>
#include <zephyr/device.h>
#include <zephyr/net/ethernet.h>
#include <zephyr/net/net_if.h>
#include <zephyr/net/net_pkt.h>
#include <zephyr/logging/log.h>
#include <zephyr/irq.h>
#include <zephyr/sys/atomic.h>
#include <zephyr/spinlock.h>
#include <string.h>
#include "eth_litex.h"
LOG_MODULE_REGISTER(eth_litex, LOG_LEVEL_INF);
/* ---- Driver data structures ---- */
struct eth_litex_config {
uint32_t irq_num;
};
struct eth_litex_data {
struct net_if *iface;
const struct device *dev;
uint8_t mac_addr[6];
struct k_thread rx_thread;
struct k_thread tx_thread;
struct k_work_delayable link_work;
struct k_spinlock ctrl_lock; /* Protects AES67_CSR_CTRL read-modify-write */
struct k_sem rx_sem; /* Signaled by ISR on RX packet */
bool link_up; /* Cached link state for edge detection */
/* Word-aligned: the packed eth_buf copy loops move 32-bit words. */
uint8_t rx_buf[ETH_LITEX_MAX_PKT_SIZE + ETH_LITEX_RX_TRAILER_MAX] __aligned(4);
uint8_t tx_buf[ETH_LITEX_MAX_PKT_SIZE] __aligned(4);
K_KERNEL_STACK_MEMBER(rx_stack, CONFIG_ETH_LITEX_RX_STACK_SIZE);
K_KERNEL_STACK_MEMBER(tx_stack, CONFIG_ETH_LITEX_RX_STACK_SIZE);
};
K_MSGQ_DEFINE(litex_tx_queue, sizeof(struct net_pkt *),
CONFIG_ETH_LITEX_TX_QUEUE_DEPTH, 4);
/* Forward-declare the device so DEVICE_GET(eth_litex0) works in init */
DEVICE_DECLARE(eth_litex0);
/* ---- CSR convenience (thread-safe ctrl register) ---- */
int eth_litex_ctrl_set_bits(const struct device *dev, uint32_t bits)
{
struct eth_litex_data *data = dev->data;
k_spinlock_key_t key = k_spin_lock(&data->ctrl_lock);
uint32_t val = litex_csr_read(CSR_AES67_CSR_CTRL_ADDR);
val |= bits;
litex_csr_write(CSR_AES67_CSR_CTRL_ADDR, val);
k_spin_unlock(&data->ctrl_lock, key);
return 0;
}
int eth_litex_ctrl_clear_bits(const struct device *dev, uint32_t bits)
{
struct eth_litex_data *data = dev->data;
k_spinlock_key_t key = k_spin_lock(&data->ctrl_lock);
uint32_t val = litex_csr_read(CSR_AES67_CSR_CTRL_ADDR);
val &= ~bits;
litex_csr_write(CSR_AES67_CSR_CTRL_ADDR, val);
k_spin_unlock(&data->ctrl_lock, key);
return 0;
}
/* ---- Public helpers ---- */
int eth_litex_write_mac(const struct device *dev, const uint8_t mac[6])
{
ARG_UNUSED(dev);
/* FPGA expects mac_addr(47 downto 0) in big-endian bit order:
* mac[0] → mac_addr(47..40), mac[1] → mac_addr(39..32), ...
* mac_addr_lo holds bits 31..0, mac_addr_hi holds bits 47..32.
* So mac[0..1] go into hi, mac[2..5] go into lo. */
uint32_t hi = ((uint32_t)mac[0] << 8) |
(uint32_t)mac[1];
uint32_t lo = ((uint32_t)mac[2] << 24) |
((uint32_t)mac[3] << 16) |
((uint32_t)mac[4] << 8) |
(uint32_t)mac[5];
litex_csr_write(CSR_AES67_CSR_MAC_ADDR_LO_ADDR, lo);
litex_csr_write(CSR_AES67_CSR_MAC_ADDR_HI_ADDR, hi);
return 0;
}
int eth_litex_write_ip(const struct device *dev, const struct in_addr *ip)
{
ARG_UNUSED(dev);
/* ip->s_addr is in network byte order (big-endian). On a
* little-endian CPU a direct 32-bit store would byte-swap it,
* putting the octets in the wrong CSR bit positions.
* The FPGA expects ip_addr(31 downto 24) = first octet, so we
* need the value in big-endian order in the CSR register. */
const uint8_t *b = (const uint8_t *)&ip->s_addr;
uint32_t val = ((uint32_t)b[0] << 24) |
((uint32_t)b[1] << 16) |
((uint32_t)b[2] << 8) |
(uint32_t)b[3];
litex_csr_write(CSR_AES67_CSR_IP_ADDR_ADDR, val);
return 0;
}
int eth_litex_write_ptp_config(const struct device *dev,
uint8_t time_source,
int8_t log_msg_interval,
int8_t log_announce_interval)
{
ARG_UNUSED(dev);
litex_csr_write(CSR_AES67_CSR_PTP_TIME_SOURCE_ADDR, time_source);
litex_csr_write(CSR_AES67_CSR_PTP_LOG_MSG_INTERVAL_ADDR, (uint8_t)log_msg_interval);
litex_csr_write(CSR_AES67_CSR_PTP_ANNOUNCE_MSG_INTERVAL_ADDR, (uint8_t)log_announce_interval);
return 0;
}
bool eth_litex_read_ptp_leader_id(const struct device *dev, uint8_t leader_clock_id[8])
{
ARG_UNUSED(dev);
uint32_t id_lo = litex_csr_read(CSR_AES67_CSR_PTP_LEADER_ID_LO_ADDR);
uint32_t id_hi = litex_csr_read(CSR_AES67_CSR_PTP_LEADER_ID_HI_ADDR);
leader_clock_id[0] = (uint8_t)(id_hi >> 24);
leader_clock_id[1] = (uint8_t)(id_hi >> 16);
leader_clock_id[2] = (uint8_t)(id_hi >> 8);
leader_clock_id[3] = (uint8_t)(id_hi);
leader_clock_id[4] = (uint8_t)(id_lo >> 24);
leader_clock_id[5] = (uint8_t)(id_lo >> 16);
leader_clock_id[6] = (uint8_t)(id_lo >> 8);
leader_clock_id[7] = (uint8_t)(id_lo);
return (id_lo | id_hi) != 0;
}
int eth_litex_write_ptp_gm_quality(const struct device *dev,
uint8_t priority1, uint8_t priority2,
uint8_t clock_class, uint8_t clock_accuracy)
{
ARG_UNUSED(dev);
litex_csr_write(CSR_AES67_CSR_PTP_GM_PRIORITY1_ADDR, priority1);
litex_csr_write(CSR_AES67_CSR_PTP_GM_PRIORITY2_ADDR, priority2);
litex_csr_write(CSR_AES67_CSR_PTP_GM_CLOCK_CLASS_ADDR, clock_class);
litex_csr_write(CSR_AES67_CSR_PTP_GM_CLOCK_ACCURACY_ADDR, clock_accuracy);
return 0;
}
bool eth_litex_read_ppb_counts(const struct device *dev,
uint32_t *wc_count, uint32_t *pll_count)
{
ARG_UNUSED(dev);
uint32_t status = litex_csr_read(CSR_AES67_CSR_PLL_PPB_STATUS_ADDR);
*wc_count = litex_csr_read(CSR_AES67_CSR_PLL_PPB_WC_COUNT_ADDR) & 0x3FFFFF;
*pll_count = litex_csr_read(CSR_AES67_CSR_PLL_PPB_PLL_COUNT_ADDR) & 0x3FFFFF;
return (status & AES67_PPB_STATUS_VALID) != 0;
}
uint32_t eth_litex_read_path_delay(const struct device *dev)
{
ARG_UNUSED(dev);
/* Always in the bridge CSR map; a software-PTP FPGA reads 0 and the
* caller sources the value from the Zephyr PTP stack instead. */
return litex_csr_read(CSR_AES67_CSR_PTP_PATH_DELAY_ADDR);
}
uint32_t eth_litex_read_ptp_offset(const struct device *dev)
{
ARG_UNUSED(dev);
return litex_csr_read(CSR_AES67_CSR_PTP_OFFSET_ADDR);
}
uint32_t eth_litex_read_status(const struct device *dev)
{
ARG_UNUSED(dev);
return litex_csr_read(CSR_AES67_CSR_STATUS_ADDR);
}
/* ---- Stream config RAM access ----
*
* The StreamConfigRAM modules are Wishbone-mapped with 1 byte per 32-bit word.
* TX_STREAM_CFG_BASE = 0x90004000, RX_STREAM_CFG_BASE = 0x90005000.
* Each stream occupies 32 bytes (addresses stream_id*32 .. stream_id*32+31).
* Byte layout documented in config_ram_address_map.md.
*/
static inline void stream_cfg_write_byte(uintptr_t base, uint8_t addr, uint8_t val)
{
volatile uint32_t *p = (volatile uint32_t *)(base + ((uint32_t)addr << 2));
*p = val;
}
int eth_litex_write_tx_stream_config(const struct device *dev,
uint8_t stream_id,
const struct in_addr *dst_ip,
uint8_t channel_count,
uint8_t samples_per_pkt,
const uint8_t *ch_ids,
uint8_t num_ch_ids,
uint32_t ssrc)
{
ARG_UNUSED(dev);
if (stream_id > 7) {
return -EINVAL;
}
uint8_t buf[20];
const uint8_t *ip = (const uint8_t *)&dst_ip->s_addr;
memset(buf, 0, sizeof(buf));
buf[0] = stream_id & 0x07;
buf[1] = ip[0];
buf[2] = ip[1];
buf[3] = ip[2];
buf[4] = ip[3];
buf[5] = channel_count;
buf[6] = samples_per_pkt;
for (uint8_t i = 0; i < num_ch_ids && i < 8; i++) {
buf[7 + i] = ch_ids[i];
}
/* Byte 15 reserved, bytes 16-19: SSRC (big-endian) */
buf[16] = (ssrc >> 24) & 0xFF;
buf[17] = (ssrc >> 16) & 0xFF;
buf[18] = (ssrc >> 8) & 0xFF;
buf[19] = ssrc & 0xFF;
/* Write all 20 bytes to the TX stream config RAM.
* The FPGA-side StreamConfigRAM computes the actual RAM address
* from stream_id (byte 0), so we write to Wishbone offset 0..19. */
uint8_t base_addr = stream_id * 32;
for (uint8_t i = 0; i < 20; i++) {
stream_cfg_write_byte(TX_STREAM_CFG_BASE, base_addr + i, buf[i]);
}
LOG_DBG("TX stream %u configured: ip=%u.%u.%u.%u ch=%u spp=%u ssrc=0x%08x",
stream_id, ip[0], ip[1], ip[2], ip[3],
channel_count, samples_per_pkt, ssrc);
return 0;
}
int eth_litex_write_rx_stream_config(const struct device *dev,
uint8_t stream_id,
const struct in_addr *dst_ip,
uint16_t dst_port,
const uint8_t *ch_map,
uint8_t channel_count,
uint8_t output_delay,
uint8_t samples_per_channel)
{
ARG_UNUSED(dev);
/* The Wishbone StreamConfigRAM writes directly into the rx_ringbuffer's
* stream_ram — no base-address indirection.
* Layout in stream_ram (17 bytes at stream_id * 32):
* 0..3: dest IP (big-endian)
* 4..5: dest UDP port (big-endian)
* 6..13: channel output map
* 14: channel count
* 15: output delay
* 16: samples per channel per packet
*/
uint8_t buf[17];
const uint8_t *ip = (const uint8_t *)&dst_ip->s_addr;
memset(buf, 0, sizeof(buf));
/* Bytes 0..3: destination IP address (network byte order) */
buf[0] = ip[0];
buf[1] = ip[1];
buf[2] = ip[2];
buf[3] = ip[3];
/* Bytes 4..5: destination UDP port (big-endian) */
buf[4] = (dst_port >> 8) & 0xFF;
buf[5] = dst_port & 0xFF;
/* Bytes 6..13: output channel map */
for (uint8_t i = 0; i < 8 && i < channel_count; i++) {
buf[6 + i] = ch_map[i];
}
/* Byte 14: channel count */
buf[14] = channel_count;
/* Byte 15: output delay in samples */
buf[15] = output_delay;
/* Byte 16: samples per channel per packet */
buf[16] = samples_per_channel;
/* Write 17 config bytes directly to the stream config RAM */
uint8_t base_addr = stream_id * 32;
for (uint8_t i = 0; i < 17; i++) {
stream_cfg_write_byte(RX_STREAM_CFG_BASE, base_addr + i, buf[i]);
}
LOG_DBG("RX stream %u configured: ip=%u.%u.%u.%u port=%u ch=%u delay=%u spc=%u",
stream_id, ip[0], ip[1], ip[2], ip[3],
dst_port, channel_count, output_delay, samples_per_channel);
return 0;
}
/* ---- PTP servo / parser tuning + monitoring ---- */
int eth_litex_write_ptp_tuning(const struct device *dev,
const struct eth_litex_ptp_tuning *t)
{
ARG_UNUSED(dev);
if (!t) {
return -EINVAL;
}
/* Servo tuning CSRs exist only when the aes67_bridge is built with the
* servo (default for software PTP; dropped via --no-servo-csr when the FPGA
* configures the servo statically). The parser CSRs below are always present. */
#ifdef CSR_AES67_CSR_SERVO_KP_GAIN_ADDR
litex_csr_write(CSR_AES67_CSR_SERVO_KP_GAIN_ADDR, (uint8_t)t->kp_gain);
litex_csr_write(CSR_AES67_CSR_SERVO_KI_GAIN_ADDR, (uint8_t)t->ki_gain);
litex_csr_write(CSR_AES67_CSR_SERVO_GAIN_SHIFT_ADDR, t->gain_shift & 0x1F);
litex_csr_write(CSR_AES67_CSR_SERVO_GAIN_SHIFT_LOCKED_ADDR, t->gain_shift_locked & 0x1F);
litex_csr_write(CSR_AES67_CSR_SERVO_KI_EXTRA_SHIFT_ADDR, t->ki_extra_shift & 0x1F);
litex_csr_write(CSR_AES67_CSR_SERVO_FILTER_SHIFT_ADDR, t->filter_shift & 0x1F);
litex_csr_write(CSR_AES67_CSR_SERVO_WARMUP_SAMPLES_ADDR, t->warmup_samples);
litex_csr_write(CSR_AES67_CSR_SERVO_LOCK_THRESHOLD_NS_ADDR, t->lock_threshold_ns);
litex_csr_write(CSR_AES67_CSR_SERVO_UNLOCK_THRESHOLD_NS_ADDR, t->unlock_threshold_ns);
litex_csr_write(CSR_AES67_CSR_SERVO_LOCK_COUNT_THRESHOLD_ADDR, t->lock_count_threshold);
#endif
uint32_t mf = (t->min_filter_enable ? 1u : 0u) |
((uint32_t)t->min_filter_active_depth << 8);
litex_csr_write(CSR_AES67_CSR_PARSER_MIN_FILTER_ADDR, mf);
litex_csr_write(CSR_AES67_CSR_PARSER_DELAY_ASYMMETRY_NS_ADDR,
(uint32_t)t->delay_asymmetry_ns);
return 0;
}
void eth_litex_read_ptp_tuning(const struct device *dev,
struct eth_litex_ptp_tuning *t)
{
ARG_UNUSED(dev);
if (!t) {
return;
}
/* Servo CSRs absent in a --no-servo-csr (static-servo) bridge build: report
* zeros for the servo tuning so callers see a defined value. */
#ifdef CSR_AES67_CSR_SERVO_KP_GAIN_ADDR
t->kp_gain = (int8_t)litex_csr_read(CSR_AES67_CSR_SERVO_KP_GAIN_ADDR);
t->ki_gain = (int8_t)litex_csr_read(CSR_AES67_CSR_SERVO_KI_GAIN_ADDR);
t->gain_shift = (uint8_t)(litex_csr_read(CSR_AES67_CSR_SERVO_GAIN_SHIFT_ADDR) & 0x1F);
t->gain_shift_locked = (uint8_t)(litex_csr_read(CSR_AES67_CSR_SERVO_GAIN_SHIFT_LOCKED_ADDR) & 0x1F);
t->ki_extra_shift = (uint8_t)(litex_csr_read(CSR_AES67_CSR_SERVO_KI_EXTRA_SHIFT_ADDR) & 0x1F);
t->filter_shift = (uint8_t)(litex_csr_read(CSR_AES67_CSR_SERVO_FILTER_SHIFT_ADDR) & 0x1F);
t->warmup_samples = (uint8_t)litex_csr_read(CSR_AES67_CSR_SERVO_WARMUP_SAMPLES_ADDR);
t->lock_threshold_ns = litex_csr_read(CSR_AES67_CSR_SERVO_LOCK_THRESHOLD_NS_ADDR);
t->unlock_threshold_ns = litex_csr_read(CSR_AES67_CSR_SERVO_UNLOCK_THRESHOLD_NS_ADDR);
t->lock_count_threshold = (uint8_t)litex_csr_read(CSR_AES67_CSR_SERVO_LOCK_COUNT_THRESHOLD_ADDR);
#else
t->kp_gain = 0;
t->ki_gain = 0;
t->gain_shift = 0;
t->gain_shift_locked = 0;
t->ki_extra_shift = 0;
t->filter_shift = 0;
t->warmup_samples = 0;
t->lock_threshold_ns = 0;
t->unlock_threshold_ns = 0;
t->lock_count_threshold = 0;
#endif
uint32_t mf = litex_csr_read(CSR_AES67_CSR_PARSER_MIN_FILTER_ADDR);
t->min_filter_enable = (mf & 0x1) != 0;
t->min_filter_active_depth = (uint8_t)((mf >> 8) & 0xFF);
t->delay_asymmetry_ns = (int32_t)litex_csr_read(
CSR_AES67_CSR_PARSER_DELAY_ASYMMETRY_NS_ADDR);
}
void eth_litex_read_ptp_monitor(const struct device *dev,
struct eth_litex_ptp_monitor *m)
{
ARG_UNUSED(dev);
if (!m) {
return;
}
/* Servo monitoring CSRs share the servo gate; absent in a --no-servo-csr
* bridge build. Report a zeroed snapshot so callers see defined values. */
#ifdef CSR_AES67_CSR_SERVO_MON_STATUS_ADDR
m->filtered_offset = (int32_t)litex_csr_read(CSR_AES67_CSR_SERVO_MON_FILTERED_OFFSET_ADDR);
m->integral_sum = (int32_t)litex_csr_read(CSR_AES67_CSR_SERVO_MON_INTEGRAL_SUM_ADDR);
m->pi_proportional = (int32_t)litex_csr_read(CSR_AES67_CSR_SERVO_MON_PI_PROPORTIONAL_ADDR);
m->pi_sum_raw = (int32_t)litex_csr_read(CSR_AES67_CSR_SERVO_MON_PI_SUM_RAW_ADDR);
uint32_t status = litex_csr_read(CSR_AES67_CSR_SERVO_MON_STATUS_ADDR);
m->effective_gain_shift = (uint8_t)(status & 0xFF);
m->lock_counter = (uint16_t)((status >> 8) & 0xFFFF);
m->first_lock_achieved = ((status >> 24) & 0x1) != 0;
m->sample_count = (uint16_t)litex_csr_read(CSR_AES67_CSR_SERVO_MON_SAMPLE_COUNT_ADDR);
#else
m->filtered_offset = 0;
m->integral_sum = 0;
m->pi_proportional = 0;
m->pi_sum_raw = 0;
m->effective_gain_shift = 0;
m->lock_counter = 0;
m->first_lock_achieved = false;
m->sample_count = 0;
#endif
}
int eth_litex_set_ptp_reset(const struct device *dev, bool held_in_reset)
{
ARG_UNUSED(dev);
/* The PTP reset is bit 0 of the unified "reset" CSR (bits: ptp, tx, rx,
* eth). Read-modify-write so we don't disturb the other reset domains. */
uint32_t reg = litex_csr_read(CSR_AES67_CSR_RESET_ADDR);
if (held_in_reset) {
reg |= (1u << CSR_AES67_CSR_RESET_PTP_OFFSET);
} else {
reg &= ~(1u << CSR_AES67_CSR_RESET_PTP_OFFSET);
}
litex_csr_write(CSR_AES67_CSR_RESET_ADDR, reg);
return 0;
}
/* ---- Packet buffer access ----
*
* The EthPacketBuffer word packing depends on the loaded bitstream
* (eth_litex_buf_packed(), from the eth_buf bytes_per_word CSR):
*
* packed: 4 payload bytes per 32-bit word, little-endian lanes.
* RX words 0x000..0x17B, TX words 0x800..0x97B.
* legacy: 1 payload byte per word (lower 8 bits valid).
* RX words 0x000..0x5ED, TX words 0x800..0xDED.
*/
static inline void eth_buf_write_byte(uint16_t offset, uint8_t val)
{
if (eth_litex_buf_packed()) {
/* Byte store: the Wishbone sel lane hits exactly this byte. */
*(volatile uint8_t *)(ETH_BUF_TX_MEM + offset) = val;
} else {
volatile uint32_t *p = (volatile uint32_t *)
(ETH_BUF_TX_MEM + ((uint32_t)offset << 2));
*p = val;
}
}
static inline uint8_t eth_buf_read_byte(uint16_t offset)
{
if (eth_litex_buf_packed()) {
return *(volatile uint8_t *)(ETH_BUF_RX_MEM + offset);
}
volatile uint32_t *p = (volatile uint32_t *)
(ETH_BUF_RX_MEM + ((uint32_t)offset << 2));
return (uint8_t)(*p & 0xFF);
}
/*
* Hand-written tight loops for packet buffer access.
*
* The CPU fetches instructions from HyperRAM which shares the Wishbone bus.
* A C for-loop generates ~6 instructions per byte, each requiring a HyperRAM
* fetch (~40+ cycles). By using a tiny asm loop (3 instructions, fits in one
* ICache line), the instruction fetches happen only once and the CPU spends
* almost all its time on the actual Wishbone data transfers.
*/
static void eth_buf_read_packet(uint8_t *dst, uint16_t len)
{
__asm__ volatile("fence" ::: "memory");
if (len == 0) return;
uintptr_t src_base = ETH_BUF_RX_MEM;
if (eth_litex_buf_packed()) {
/* Packed layout: one 32-bit word moves 4 frame bytes (both
* sides little-endian, dst is 4-byte aligned). */
uint16_t words = len >> 2;
uint16_t tail = len & 3;
uint32_t *dst32 = (uint32_t *)dst;
if (words) {
__asm__ volatile(
"1:\n"
" lw t0, 0(%[src])\n"
" sw t0, 0(%[dst])\n"
" addi %[dst], %[dst], 4\n"
" addi %[src], %[src], 4\n"
" addi %[rem], %[rem], -1\n"
" bnez %[rem], 1b\n"
: [dst] "+r"(dst32), [src] "+r"(src_base),
[rem] "+r"(words)
:
: "t0", "memory"
);
}
if (tail) {
uint32_t w = *(volatile uint32_t *)src_base;
uint8_t *d = (uint8_t *)dst32;
for (uint16_t k = 0; k < tail; k++) {
d[k] = (uint8_t)(w >> (8 * k));
}
}
return;
}
/* Legacy layout: one word per byte (low byte valid). */
__asm__ volatile(
"1:\n"
" lw t0, 0(%[src])\n" /* read 32-bit word from RX buffer */
" sb t0, 0(%[dst])\n" /* store low byte to dst */
" addi %[dst], %[dst], 1\n"
" addi %[src], %[src], 4\n" /* next Wishbone word (4 bytes apart) */
" addi %[rem], %[rem], -1\n"
" bnez %[rem], 1b\n"
: [dst] "+r"(dst), [src] "+r"(src_base), [rem] "+r"(len)
:
: "t0", "memory"
);
}
static void eth_buf_write_packet(const uint8_t *src, uint16_t len)
{
if (len == 0) return;
uintptr_t dst_base = ETH_BUF_TX_MEM;
if (eth_litex_buf_packed()) {
uint16_t words = len >> 2;
uint16_t tail = len & 3;
const uint32_t *src32 = (const uint32_t *)src;
if (words) {
__asm__ volatile(
"1:\n"
" lw t0, 0(%[src])\n"
" sw t0, 0(%[dst])\n"
" addi %[src], %[src], 4\n"
" addi %[dst], %[dst], 4\n"
" addi %[rem], %[rem], -1\n"
" bnez %[rem], 1b\n"
: [src] "+r"(src32), [dst] "+r"(dst_base),
[rem] "+r"(words)
:
: "t0", "memory"
);
}
if (tail) {
/* Zero-padded final word; tx_len bounds the frame. */
const uint8_t *s = (const uint8_t *)src32;
uint32_t w = 0;
for (uint16_t k = 0; k < tail; k++) {
w |= (uint32_t)s[k] << (8 * k);
}
*(volatile uint32_t *)dst_base = w;
}
__asm__ volatile("fence" ::: "memory");
return;
}
/* Legacy layout: one word per byte. */
__asm__ volatile(
"1:\n"
" lbu t0, 0(%[src])\n" /* load byte from src */
" sw t0, 0(%[dst])\n" /* write 32-bit word to TX buffer */
" addi %[src], %[src], 1\n"
" addi %[dst], %[dst], 4\n" /* next Wishbone word */
" addi %[rem], %[rem], -1\n"
" bnez %[rem], 1b\n"
: [src] "+r"(src), [dst] "+r"(dst_base), [rem] "+r"(len)
:
: "t0", "memory"
);
__asm__ volatile("fence" ::: "memory");
}
/* Temporary RX-path diagnostics (remove after the cyc1000 regression hunt):
* one WRN line every 2 s distinguishes ack-loss (redeliver>0), dead IRQ
* (irq==0 while poll>0 drains frames) and FIFO pressure (ready_after_ack). */
static struct {
uint32_t irq; /* ISR invocations */
uint32_t wake_sem; /* rx thread woken by semaphore */
uint32_t wake_poll; /* rx thread woken by poll timeout */
uint32_t frames; /* frames delivered to the stack */
uint32_t redeliver; /* identical frame seen twice in a row */
uint32_t ready_after_ack; /* rx_ready still/again set right after ack */
uint16_t last_len;
uint8_t last_sig[16];
int64_t last_report;
} rx_dbg;
/* ---- ISR ---- */
static void eth_litex_isr(const struct device *dev)
{
struct eth_litex_data *data = dev->data;
/* Unconditionally disable EventManager + clear pending.
* Skip the ev_pending READ — if the Wishbone bus hangs on
* CSR reads during IRQ context, this avoids it entirely.
* RX thread will re-enable after processing. */
litex_csr_write(CSR_ETH_BUF_EV_ENABLE_ADDR, 0);
litex_csr_write(CSR_ETH_BUF_EV_PENDING_ADDR, ETH_BUF_EV_RX_READY);
rx_dbg.irq++;
k_sem_give(&data->rx_sem);
}
/* ---- TX path ---- */
static void eth_litex_tx_thread(void *p1, void *p2, void *p3)
{
const struct device *dev = p1;
struct eth_litex_data *data = dev->data;
struct net_pkt *pkt = NULL;
ARG_UNUSED(p2);
ARG_UNUSED(p3);
while (1) {
k_msgq_get(&litex_tx_queue, &pkt, K_FOREVER);
size_t len = net_pkt_get_len(pkt);
LOG_DBG("TX: dequeued pkt len=%zu", len);
if (len > ETH_LITEX_MAX_PKT_SIZE) {
LOG_ERR("TX too large: %zu", len);
net_pkt_unref(pkt);
continue;
}
if (net_pkt_read(pkt, data->tx_buf, len) < 0) {
LOG_ERR("TX pkt read failed");
net_pkt_unref(pkt);
continue;
}
/* Wait for previous TX to complete (eth_tx_done status bit).
* On the very first call tx_done starts de-asserted (FPGA reset
* default), so we only wait if a transmit is actually in-flight
* (i.e. we previously pulsed eth_tx_request).
*
* A 1518-byte frame at 100 Mbps takes ~122µs on the wire.
* We use k_busy_wait() because the system tick (10ms at 100Hz)
* is far too coarse for this — k_usleep/k_sleep would round up
* to 10ms per iteration, making TX unbearably slow.
* We yield once before spinning so other threads get a chance. */
static bool first_tx = true;
if (!first_tx) {
k_yield();
int wait = 0;
while (!(litex_csr_read(CSR_AES67_CSR_STATUS_ADDR) & AES67_STATUS_ETH_TX_DONE)) {
k_busy_wait(10);
if (++wait > 20000) { /* ~200ms timeout */
LOG_WRN("TX done timeout (status=0x%08x)",
litex_csr_read(CSR_AES67_CSR_STATUS_ADDR));
break;
}
}
}
first_tx = false;
#ifdef CONFIG_AES67_PTP_SOFTWARE
/* Egress timestamps only exist in PTP_IN_SOFTWARE gateware (and
* only the software PTP stack requests them). */
bool want_ts = fpga_hal_ptp_in_software() &&
net_pkt_is_tx_timestamping(pkt);
uint32_t pre_sec = 0, pre_nsec = 0;
/* Snapshot the egress-timestamp CSR while the previous frame is
* complete (wait above): the CSR follows the live TSU capture
* register, so whatever it shows now belongs to an earlier frame
* (non-timestamped transmits update it too). Our frame's capture
* is the next CHANGE of this value — eth_tx_done alone cannot
* attribute the capture to our frame, since for short frames the
* done latch can precede the on-wire SOF. */
if (want_ts) {
pre_sec = litex_csr_read(
CSR_AES67_CSR_TX_TIMESTAMP_SEC_IN_ADDR) & 0xF;
pre_nsec = litex_csr_read(
CSR_AES67_CSR_TX_TIMESTAMP_NSEC_IN_ADDR) & 0x3FFFFFFF;
}
#endif
LOG_DBG("TX: writing %zu bytes to buffer", len);
/* Write packet data to TX buffer */
eth_buf_write_packet(data->tx_buf, len);
/* Set TX length */
litex_csr_write(CSR_ETH_BUF_TX_LEN_ADDR, len);
LOG_DBG("TX: pulsing tx_request");
/* Assert eth_tx_request and hold long enough for CDC capture.
* The FPGA synchronises this signal through a 2-FF chain clocked
* by mac_tx_clock (≥2.5 MHz → ≤400ns period). We need the pulse
* to span at least 3 mac_tx_clock edges, so 2µs is safe. */
eth_litex_ctrl_set_bits(dev, AES67_CTRL_ETH_TX_REQUEST);
k_busy_wait(2);
eth_litex_ctrl_clear_bits(dev, AES67_CTRL_ETH_TX_REQUEST);
#ifdef CONFIG_AES67_PTP_SOFTWARE
/* PTP event frames need their egress timestamp reported back to
* the stack. Order: pulse the request, wait for eth_tx_done, read
* the timestamp CSR — but accept it only once it differs from the
* pre-transmit snapshot (that change IS our frame's SOF capture
* arriving; see snapshot comment above). */
if (want_ts) {
struct net_ptp_time tx_ts;
int ts_wait = 0;
bool fresh = false;
uint32_t sec = 0, nsec = 0;
k_busy_wait(5); /* let the request edge propagate/clear tx_done */
while (ts_wait++ < ETH_LITEX_TX_TS_POLL_LIMIT) {
if (litex_csr_read(CSR_AES67_CSR_STATUS_ADDR) &
AES67_STATUS_ETH_TX_DONE) {
sec = litex_csr_read(
CSR_AES67_CSR_TX_TIMESTAMP_SEC_IN_ADDR) & 0xF;
nsec = litex_csr_read(
CSR_AES67_CSR_TX_TIMESTAMP_NSEC_IN_ADDR) & 0x3FFFFFFF;
if (sec != pre_sec || nsec != pre_nsec) {
fresh = true;
break;
}
}
k_busy_wait(50);
}
if (!fresh) {
/* Report a null timestamp: the PTP stack
* discards the exchange and keeps the last
* good path delay. */
LOG_WRN("TX timestamp stale after %d polls - invalidating exchange",
ts_wait);
tx_ts.second = 0;
tx_ts.nanosecond = 0;
} else {
aes67_ptp_reconstruct((uint8_t)sec, nsec, &tx_ts);
}
net_pkt_set_timestamp(pkt, &tx_ts);
net_if_add_tx_timestamp(pkt);
}
#endif
LOG_DBG("TX: done, unref pkt");
net_pkt_unref(pkt);
}
}
/* ---- RX path ---- */
/* Acknowledge the presented RX frame. The ack level crosses into the MAC RX
* clock domain through a 2-FF synchronizer + edge detector (same CDC as
* eth_tx_request): back-to-back CSR writes make a ~50ns pulse the 50MHz
* (RMII) — or 2.5MHz (10M MII) — sampling misses, which re-delivers the same
* frame on the next poll and stalls the RX FIFO drain. Hold it wide. */
static void eth_litex_rx_ack(void)
{
litex_csr_write(CSR_ETH_BUF_RX_ACK_ADDR, 1);
k_busy_wait(2);
litex_csr_write(CSR_ETH_BUF_RX_ACK_ADDR, 0);
}
static void eth_litex_rx_thread(void *p1, void *p2, void *p3)
{
const struct device *dev = p1;
struct eth_litex_data *data = dev->data;
ARG_UNUSED(p2);
ARG_UNUSED(p3);
while (1) {
if (k_sem_take(&data->rx_sem,
K_MSEC(CONFIG_ETH_LITEX_POLL_INTERVAL_MS)) == 0) {
rx_dbg.wake_sem++;
} else {
rx_dbg.wake_poll++;
}
int64_t now = k_uptime_get();
if (now - rx_dbg.last_report >= 2000) {
rx_dbg.last_report = now;
LOG_WRN("RXDBG irq=%u sem=%u poll=%u frames=%u redeliv=%u rdy_after_ack=%u",
rx_dbg.irq, rx_dbg.wake_sem, rx_dbg.wake_poll,
rx_dbg.frames, rx_dbg.redeliver, rx_dbg.ready_after_ack);
}
/* Check if a packet is ready */
uint32_t ready = litex_csr_read(CSR_ETH_BUF_RX_READY_ADDR);
if (!(ready & 0x01)) {
/* Re-enable EventManager in case ISR disabled it but
* no packet was actually ready (race with ack). */
litex_csr_write(CSR_ETH_BUF_EV_ENABLE_ADDR, ETH_BUF_EV_RX_READY);
continue;
}
uint16_t raw_len = litex_csr_read(CSR_ETH_BUF_RX_LEN_ADDR) & 0xFFFF;
uint16_t pkt_len;
/* PTP_IN_SOFTWARE gateware appends a 5-byte timestamp trailer
* after the FCS; whether it is present comes from the loaded
* bitstream's system_cfg (0 on hardware-PTP gateware). */
uint16_t trailer = eth_litex_rx_trailer_len();
#ifdef CONFIG_AES67_PTP_SOFTWARE
struct net_ptp_time rx_ts;
bool rx_have_ts = false;
#endif
/* Read the whole raw buffer (frame + FCS + optional timestamp
* trailer), peel the trailer off the end and reconstruct the RX
* timestamp, then strip the FCS from what remains. */
raw_len = MIN(raw_len, (uint16_t)sizeof(data->rx_buf));
eth_buf_read_packet(data->rx_buf, raw_len);
if (raw_len >= trailer + 4) {
uint16_t t = raw_len - trailer;
#ifdef CONFIG_AES67_PTP_SOFTWARE
if (trailer != 0) {
uint32_t ts_nsec =
(uint32_t)data->rx_buf[t + 1] |
((uint32_t)data->rx_buf[t + 2] << 8) |
((uint32_t)data->rx_buf[t + 3] << 16) |
((uint32_t)data->rx_buf[t + 4] << 24);
/* Integrity gate: seconds byte is 0x0S and
* ns bits [31:30] are zero by construction —
* anything else is a corrupted trailer read. */
if ((data->rx_buf[t] & 0xF0) != 0 ||
(data->rx_buf[t + 4] & 0xC0) != 0) {
LOG_WRN("RX trailer corrupt (%02x .. %02x) - dropping ts",
data->rx_buf[t],
data->rx_buf[t + 4]);
} else {
aes67_ptp_reconstruct(data->rx_buf[t],
ts_nsec, &rx_ts);
rx_have_ts = true;
}
}
#endif
/* Drop the trailer (if any), then the 4-byte FCS. */
pkt_len = t - 4;
} else {
pkt_len = 0;
}
if (pkt_len < 14 || pkt_len > ETH_LITEX_MAX_PKT_SIZE) {
LOG_WRN("RX invalid len %u (raw %u)", pkt_len, raw_len);
eth_litex_rx_ack();
litex_csr_write(CSR_ETH_BUF_EV_ENABLE_ADDR, ETH_BUF_EV_RX_READY);
continue;
}
LOG_HEXDUMP_DBG(data->rx_buf, pkt_len, "RX frame");
/* Debug: log ethertype and dst port for TCP packets */
{
uint16_t ethertype = ((uint16_t)data->rx_buf[12] << 8) |
data->rx_buf[13];
if (ethertype == 0x0800 && pkt_len >= 34) {
uint8_t proto = data->rx_buf[23];
if (proto == 6) { /* TCP */
uint16_t dst_port =
((uint16_t)data->rx_buf[36] << 8) |
data->rx_buf[37];
if (dst_port == 80) {
uint16_t ip_total_len =
((uint16_t)data->rx_buf[16] << 8) |
data->rx_buf[17];
uint8_t ip_ihl =
(data->rx_buf[14] & 0x0F) * 4;
uint8_t tcp_doff =
((data->rx_buf[14 + ip_ihl + 12] >> 4) & 0xF) * 4;
uint16_t tcp_payload =
ip_total_len - ip_ihl - tcp_doff;
uint8_t tcp_flags =
data->rx_buf[14 + ip_ihl + 13];
LOG_DBG("RX TCP:80 eth=%u ip=%u ihl=%u "
"doff=%u payload=%u flags=0x%02x "
"seq=%02x%02x%02x%02x",
pkt_len, ip_total_len,
ip_ihl, tcp_doff,
tcp_payload, tcp_flags,
data->rx_buf[38],
data->rx_buf[39],
data->rx_buf[40],
data->rx_buf[41]);
}
}
}
}
/* Redelivery detector: an identical frame (len + first 16 bytes,
* which include the IP ID/checksum) means the previous ack did
* not advance the FPGA-side FIFO. */
if (pkt_len == rx_dbg.last_len &&
memcmp(data->rx_buf, rx_dbg.last_sig,
MIN(pkt_len, sizeof(rx_dbg.last_sig))) == 0) {
rx_dbg.redeliver++;
}
rx_dbg.last_len = pkt_len;
memcpy(rx_dbg.last_sig, data->rx_buf,
MIN(pkt_len, sizeof(rx_dbg.last_sig)));
rx_dbg.frames++;
/* ACK: release the RX buffer for the next packet */
eth_litex_rx_ack();
if (litex_csr_read(CSR_ETH_BUF_RX_READY_ADDR) & 0x01) {
rx_dbg.ready_after_ack++;
}
/* Re-enable EventManager IRQ (ISR disables it to prevent
* re-entry while the pending clear propagates). */
litex_csr_write(CSR_ETH_BUF_EV_ENABLE_ADDR, ETH_BUF_EV_RX_READY);
/* Allocate net_pkt and deliver to stack */
struct net_pkt *pkt = net_pkt_rx_alloc_with_buffer(
data->iface, pkt_len, AF_UNSPEC, 0, K_NO_WAIT);
if (!pkt) {
LOG_ERR("RX alloc failed (len=%u)", pkt_len);
continue;
}
if (net_pkt_write(pkt, data->rx_buf, pkt_len)) {
LOG_ERR("RX write failed");
net_pkt_unref(pkt);
continue;
}
#ifdef CONFIG_AES67_PTP_SOFTWARE
if (rx_have_ts) {
net_pkt_set_timestamp(pkt, &rx_ts);
}
#endif
int ret = net_recv_data(data->iface, pkt);
if (ret < 0) {
LOG_ERR("RX deliver err %d", ret);
net_pkt_unref(pkt);
}
}
}
/* ---- Zephyr ethernet API ---- */
static int eth_litex_send(const struct device *dev, struct net_pkt *pkt)
{
ARG_UNUSED(dev);
/* Take a reference BEFORE posting to the TX queue. The TX thread
* runs at K_PRIO_COOP(2) and will preempt any preemptive-priority
* caller as soon as k_msgq_put wakes it. If we ref after the put,