v1 datasheet latex source
This commit is contained in:
@@ -0,0 +1,34 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand \babel@aux [2]{\global \let \babel@toc \@gobbletwo }
|
||||||
|
\@nameuse{bbl@beforestart}
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\providecommand\HyField@AuxAddToFields[1]{}
|
||||||
|
\providecommand\HyField@AuxAddToCoFields[2]{}
|
||||||
|
\babel@aux{english}{}
|
||||||
|
\pgfsyspdfmark {pgfid1}{5216401}{50184874}
|
||||||
|
\gdef \LT@i {\LT@entry
|
||||||
|
{1}{103.04872pt}\LT@entry
|
||||||
|
{1}{114.43008pt}\LT@entry
|
||||||
|
{1}{254.83714pt}}
|
||||||
|
\gdef \LT@ii {\LT@entry
|
||||||
|
{1}{88.82234pt}\LT@entry
|
||||||
|
{1}{40.45274pt}\LT@entry
|
||||||
|
{1}{40.45274pt}\LT@entry
|
||||||
|
{1}{37.6073pt}\LT@entry
|
||||||
|
{1}{264.98082pt}}
|
||||||
|
\@input{chapters/01-overview.aux}
|
||||||
|
\@input{chapters/02-architettura.aux}
|
||||||
|
\@input{chapters/03-datapath.aux}
|
||||||
|
\@input{chapters/04-parametri.aux}
|
||||||
|
\@input{chapters/05-memoria.aux}
|
||||||
|
\@input{chapters/06-sequencer.aux}
|
||||||
|
\@input{chapters/06b-grafo.aux}
|
||||||
|
\@input{chapters/07-spi.aux}
|
||||||
|
\@input{chapters/07b-programmazione.aux}
|
||||||
|
\@input{chapters/08-toplevel.aux}
|
||||||
|
\@input{chapters/09-implementazione.aux}
|
||||||
|
\@input{chapters/10-hardware.aux}
|
||||||
|
\@input{chapters/11-registri.aux}
|
||||||
|
\@input{chapters/12-roadmap.aux}
|
||||||
|
\@input{chapters/A-moduli.aux}
|
||||||
|
\gdef \@abspage@last{56}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,121 @@
|
|||||||
|
\BOOKMARK [0][-]{chapter.1}{\376\377\000S\000y\000s\000t\000e\000m\000\040\000o\000v\000e\000r\000v\000i\000e\000w}{}% 1
|
||||||
|
\BOOKMARK [1][-]{section.1.1}{\376\377\000P\000r\000o\000j\000e\000c\000t\000\040\000g\000o\000a\000l}{chapter.1}% 2
|
||||||
|
\BOOKMARK [1][-]{section.1.2}{\376\377\000H\000a\000r\000d\000w\000a\000r\000e\000\040\000c\000o\000n\000f\000i\000g\000u\000r\000a\000t\000i\000o\000n\000\040\000v\000e\000r\000s\000u\000s\000\040\000n\000e\000t\000w\000o\000r\000k\000\040\000c\000o\000n\000f\000i\000g\000u\000r\000a\000t\000i\000o\000n}{chapter.1}% 3
|
||||||
|
\BOOKMARK [1][-]{section.1.3}{\376\377\000B\000o\000o\000t\000\040\000a\000n\000d\000\040\000i\000n\000i\000t\000i\000a\000l\000i\000z\000a\000t\000i\000o\000n}{chapter.1}% 4
|
||||||
|
\BOOKMARK [1][-]{section.1.4}{\376\377\000T\000r\000a\000i\000n\000i\000n\000g\000\040\000a\000n\000d\000\040\000i\000n\000f\000e\000r\000e\000n\000c\000e}{chapter.1}% 5
|
||||||
|
\BOOKMARK [1][-]{section.1.5}{\376\377\000D\000e\000s\000i\000g\000n\000\040\000p\000h\000i\000l\000o\000s\000o\000p\000h\000y\000\040\000a\000n\000d\000\040\000r\000e\000u\000s\000e}{chapter.1}% 6
|
||||||
|
\BOOKMARK [0][-]{chapter.2}{\376\377\000R\000T\000L\000\040\000a\000r\000c\000h\000i\000t\000e\000c\000t\000u\000r\000e}{}% 7
|
||||||
|
\BOOKMARK [1][-]{section.2.1}{\376\377\000H\000i\000e\000r\000a\000r\000c\000h\000i\000c\000a\000l\000\040\000o\000r\000g\000a\000n\000i\000z\000a\000t\000i\000o\000n}{chapter.2}% 8
|
||||||
|
\BOOKMARK [1][-]{section.2.2}{\376\377\000R\000o\000l\000e\000\040\000o\000f\000\040\000e\000a\000c\000h\000\040\000m\000o\000d\000u\000l\000e}{chapter.2}% 9
|
||||||
|
\BOOKMARK [1][-]{section.2.3}{\376\377\000T\000w\000o\000\040\000e\000x\000e\000c\000u\000t\000i\000o\000n\000\040\000p\000a\000t\000h\000s}{chapter.2}% 10
|
||||||
|
\BOOKMARK [0][-]{chapter.3}{\376\377\000C\000o\000m\000p\000u\000t\000e\000\040\000d\000a\000t\000a\000p\000a\000t\000h}{}% 11
|
||||||
|
\BOOKMARK [1][-]{section.3.1}{\376\377\000I\000N\000T\0008\000/\000I\000N\000T\0003\0002\000\040\000a\000r\000i\000t\000h\000m\000e\000t\000i\000c\000\040\000c\000h\000a\000i\000n}{chapter.3}% 12
|
||||||
|
\BOOKMARK [1][-]{section.3.2}{\376\377\000m\000a\000c\000\137\000u\000n\000i\000t\000\040\040\024\000\040\000m\000u\000l\000t\000i\000p\000l\000y\000-\000a\000c\000c\000u\000m\000u\000l\000a\000t\000o\000r}{chapter.3}% 13
|
||||||
|
\BOOKMARK [1][-]{section.3.3}{\376\377\000m\000a\000c\0008\000\040\040\024\000\040\000p\000a\000r\000a\000l\000l\000e\000l\000\040\000M\000A\000C\000\040\000a\000n\000d\000\040\000b\000a\000l\000a\000n\000c\000e\000d\000\040\000a\000d\000d\000e\000r\000\040\000t\000r\000e\000e}{chapter.3}% 14
|
||||||
|
\BOOKMARK [1][-]{section.3.4}{\376\377\000n\000e\000u\000r\000o\000n\000\137\000p\000a\000r\000a\000l\000l\000e\000l\000\040\040\024\000\040\000n\000e\000u\000r\000o\000n\000\040\000F\000S\000M}{chapter.3}% 15
|
||||||
|
\BOOKMARK [2][-]{subsection.3.4.1}{\376\377\000P\000a\000r\000a\000m\000e\000t\000e\000r\000\040\000g\000u\000a\000r\000d\000\040\000\050\000e\000l\000a\000b\000o\000r\000a\000t\000i\000o\000n\000-\000t\000i\000m\000e\000\051}{section.3.4}% 16
|
||||||
|
\BOOKMARK [1][-]{section.3.5}{\376\377\000A\000c\000t\000i\000v\000a\000t\000i\000o\000n\000\040\000f\000u\000n\000c\000t\000i\000o\000n\000s}{chapter.3}% 17
|
||||||
|
\BOOKMARK [1][-]{section.3.6}{\376\377\000I\000N\000T\0008\000\040\000s\000a\000t\000u\000r\000a\000t\000i\000o\000n}{chapter.3}% 18
|
||||||
|
\BOOKMARK [1][-]{section.3.7}{\376\377\000l\000a\000y\000e\000r\000\040\040\024\000\040\000n\000e\000u\000r\000o\000n\000s\000\040\000i\000n\000\040\000p\000a\000r\000a\000l\000l\000e\000l}{chapter.3}% 19
|
||||||
|
\BOOKMARK [0][-]{chapter.4}{\376\377\000P\000a\000r\000a\000m\000e\000t\000e\000r\000s\000\040\000a\000n\000d\000\040\000c\000o\000n\000f\000i\000g\000u\000r\000a\000b\000i\000l\000i\000t\000y}{}% 20
|
||||||
|
\BOOKMARK [1][-]{section.4.1}{\376\377\000B\000u\000i\000l\000d\000\040\000p\000a\000r\000a\000m\000e\000t\000e\000r\000s\000\040\000\050\000s\000y\000n\000t\000h\000e\000s\000i\000s\000-\000t\000i\000m\000e\000\051}{chapter.4}% 21
|
||||||
|
\BOOKMARK [1][-]{section.4.2}{\376\377\000R\000u\000n\000t\000i\000m\000e\000\040\000n\000e\000t\000w\000o\000r\000k\000\040\000w\000i\000d\000t\000h}{chapter.4}% 22
|
||||||
|
\BOOKMARK [2][-]{subsection.4.2.1}{\376\377\000M\000e\000a\000s\000u\000r\000e\000d\000\040\000s\000a\000v\000i\000n\000g\000s}{section.4.2}% 23
|
||||||
|
\BOOKMARK [1][-]{section.4.3}{\376\377\000C\000h\000a\000r\000a\000c\000t\000e\000r\000i\000z\000e\000d\000\040\000c\000o\000n\000f\000i\000g\000u\000r\000a\000t\000i\000o\000n\000s}{chapter.4}% 24
|
||||||
|
\BOOKMARK [1][-]{section.4.4}{\376\377\000B\000u\000i\000l\000d\000\040\000v\000e\000r\000s\000u\000s\000\040\000r\000u\000n\000t\000i\000m\000e\000\040\000s\000u\000m\000m\000a\000r\000y}{chapter.4}% 25
|
||||||
|
\BOOKMARK [0][-]{chapter.5}{\376\377\000M\000e\000m\000o\000r\000y\000\040\000s\000u\000b\000s\000y\000s\000t\000e\000m}{}% 26
|
||||||
|
\BOOKMARK [1][-]{section.5.1}{\376\377\000M\000e\000m\000o\000r\000y\000\040\000c\000h\000a\000i\000n}{chapter.5}% 27
|
||||||
|
\BOOKMARK [1][-]{section.5.2}{\376\377\000i\000n\000t\0008\000\137\000m\000e\000m\000o\000r\000y\000\137\000a\000c\000c\000e\000s\000s\000\040\040\024\000\040\000b\000y\000t\000e\000/\000w\000o\000r\000d\000\040\000c\000o\000n\000v\000e\000r\000s\000i\000o\000n}{chapter.5}% 28
|
||||||
|
\BOOKMARK [1][-]{section.5.3}{\376\377\000m\000e\000m\000o\000r\000y\000\137\000i\000n\000t\000e\000r\000f\000a\000c\000e\000\040\040\024\000\040\000h\000a\000n\000d\000s\000h\000a\000k\000e}{chapter.5}% 29
|
||||||
|
\BOOKMARK [1][-]{section.5.4}{\376\377\000p\000s\000r\000a\000m\000\137\000c\000o\000n\000t\000r\000o\000l\000l\000e\000r\000\040\040\024\000\040\000p\000h\000y\000s\000i\000c\000a\000l\000\040\000b\000u\000s}{chapter.5}% 30
|
||||||
|
\BOOKMARK [2][-]{subsection.5.4.1}{\376\377\000T\000i\000m\000i\000n\000g}{section.5.4}% 31
|
||||||
|
\BOOKMARK [1][-]{section.5.5}{\376\377\000R\000e\000a\000d\000\040\000p\000a\000g\000e\000\040\000m\000o\000d\000e}{chapter.5}% 32
|
||||||
|
\BOOKMARK [1][-]{section.5.6}{\376\377\000A\000d\000d\000r\000e\000s\000s\000\040\000m\000a\000p\000\040\000a\000n\000d\000\040\000c\000o\000n\000v\000e\000n\000t\000i\000o\000n\000s}{chapter.5}% 33
|
||||||
|
\BOOKMARK [2][-]{subsection.5.6.1}{\376\377\000P\000S\000R\000A\000M\000\040\000p\000h\000y\000s\000i\000c\000a\000l\000\040\000a\000d\000d\000r\000e\000s\000s\000i\000n\000g}{section.5.6}% 34
|
||||||
|
\BOOKMARK [1][-]{section.5.7}{\376\377\000B\000a\000n\000d\000w\000i\000d\000t\000h}{chapter.5}% 35
|
||||||
|
\BOOKMARK [0][-]{chapter.6}{\376\377\000M\000e\000m\000o\000r\000y\000,\000\040\000m\000u\000l\000t\000i\000-\000n\000e\000u\000r\000o\000n\000\040\000a\000n\000d\000\040\000m\000u\000l\000t\000i\000-\000l\000a\000y\000e\000r}{}% 36
|
||||||
|
\BOOKMARK [1][-]{section.6.1}{\376\377\000n\000e\000u\000r\000o\000n\000\137\000m\000e\000m\000o\000r\000y\000\040\040\024\000\040\000m\000e\000m\000o\000r\000y\000/\000n\000e\000u\000r\000o\000n\000\040\000b\000r\000i\000d\000g\000e}{chapter.6}% 37
|
||||||
|
\BOOKMARK [1][-]{section.6.2}{\376\377\000l\000a\000y\000e\000r\000\137\000s\000e\000q\000u\000e\000n\000c\000e\000r\000\040\040\024\000\040\000m\000u\000l\000t\000i\000-\000l\000a\000y\000e\000r\000\040\000n\000e\000t\000w\000o\000r\000k}{chapter.6}% 38
|
||||||
|
\BOOKMARK [2][-]{subsection.6.2.1}{\376\377\000P\000i\000n\000g\000-\000p\000o\000n\000g\000\040\000b\000u\000f\000f\000e\000r\000s}{section.6.2}% 39
|
||||||
|
\BOOKMARK [2][-]{subsection.6.2.2}{\376\377\000D\000e\000s\000c\000r\000i\000p\000t\000o\000r\000\040\000t\000a\000b\000l\000e}{section.6.2}% 40
|
||||||
|
\BOOKMARK [1][-]{section.6.3}{\376\377\000H\000i\000e\000r\000a\000r\000c\000h\000y\000\040\000o\000f\000\040\000t\000h\000e\000\040\000b\000u\000s\000y\000/\000d\000o\000n\000e\000\040\000s\000i\000g\000n\000a\000l\000s}{chapter.6}% 41
|
||||||
|
\BOOKMARK [0][-]{chapter.7}{\376\377\000G\000r\000a\000p\000h\000\040\000n\000e\000t\000w\000o\000r\000k\000\040\000\050\000T\000y\000p\000e\000\040\000\043\0002\000\051}{}% 42
|
||||||
|
\BOOKMARK [1][-]{section.7.1}{\376\377\000T\000w\000o\000\040\000n\000e\000t\000w\000o\000r\000k\000\040\000t\000y\000p\000e\000s}{chapter.7}% 43
|
||||||
|
\BOOKMARK [1][-]{section.7.2}{\376\377\000G\000l\000o\000b\000a\000l\000\040\000a\000c\000t\000i\000v\000a\000t\000i\000o\000n\000\040\000b\000u\000f\000f\000e\000r}{chapter.7}% 44
|
||||||
|
\BOOKMARK [1][-]{section.7.3}{\376\377\000F\000e\000e\000d\000-\000f\000o\000r\000w\000a\000r\000d\000\040\000D\000A\000G\000\040\000a\000n\000d\000\040\000t\000h\000e\000\040\000s\000r\000c\000\137\000i\000d\000\040\000<\000\040\000o\000u\000t\000\137\000i\000d\000\040\000r\000u\000l\000e}{chapter.7}% 45
|
||||||
|
\BOOKMARK [1][-]{section.7.4}{\376\377\000D\000a\000t\000a\000\040\000f\000o\000r\000m\000a\000t\000s}{chapter.7}% 46
|
||||||
|
\BOOKMARK [2][-]{subsection.7.4.1}{\376\377\000T\000y\000p\000e\000\040\000\043\0002\000\040\000d\000e\000s\000c\000r\000i\000p\000t\000o\000r\000\040\000\050\000g\000r\000a\000p\000h\000\051}{section.7.4}% 47
|
||||||
|
\BOOKMARK [2][-]{subsection.7.4.2}{\376\377\000G\000r\000a\000p\000h\000\040\000e\000d\000g\000e\000\040\000\050\0004\000\040\000b\000y\000t\000e\000s\000,\000\040\000a\000l\000i\000g\000n\000e\000d\000\051}{section.7.4}% 48
|
||||||
|
\BOOKMARK [1][-]{section.7.5}{\376\377\000g\000r\000a\000p\000h\000\137\000e\000n\000g\000i\000n\000e\000\040\040\024\000\040\000g\000r\000a\000p\000h\000\040\000e\000n\000g\000i\000n\000e}{chapter.7}% 49
|
||||||
|
\BOOKMARK [1][-]{section.7.6}{\376\377\000T\000y\000p\000e\000\040\000\043\0002\000\040\000o\000p\000c\000o\000d\000e\000s\000\040\000a\000n\000d\000\040\000r\000e\000g\000i\000s\000t\000e\000r\000s}{chapter.7}% 50
|
||||||
|
\BOOKMARK [1][-]{section.7.7}{\376\377\000O\000c\000c\000u\000p\000a\000n\000c\000y\000\040\000\050\000T\000y\000p\000e\000\040\000\043\0002\000\040\000e\000n\000a\000b\000l\000e\000d\000\051}{chapter.7}% 51
|
||||||
|
\BOOKMARK [1][-]{section.7.8}{\376\377\000G\000a\000t\000h\000e\000r\000\040\000b\000a\000n\000d\000w\000i\000d\000t\000h\000\040\000\050\000m\000e\000a\000s\000u\000r\000e\000d\000\051}{chapter.7}% 52
|
||||||
|
\BOOKMARK [1][-]{section.7.9}{\376\377\000n\000e\000t\000a\000s\000m\000\040\000h\000o\000s\000t\000\040\000a\000s\000s\000e\000m\000b\000l\000e\000r}{chapter.7}% 53
|
||||||
|
\BOOKMARK [0][-]{chapter.8}{\376\377\000S\000P\000I\000\040\000h\000o\000s\000t\000\040\000i\000n\000t\000e\000r\000f\000a\000c\000e}{}% 54
|
||||||
|
\BOOKMARK [1][-]{section.8.1}{\376\377\000P\000h\000y\000s\000i\000c\000a\000l\000\040\000l\000a\000y\000e\000r}{chapter.8}% 55
|
||||||
|
\BOOKMARK [1][-]{section.8.2}{\376\377\000F\000r\000a\000m\000i\000n\000g\000\040\000a\000n\000d\000\040\000e\000x\000p\000l\000i\000c\000i\000t\000\040\000l\000e\000n\000g\000t\000h}{chapter.8}% 56
|
||||||
|
\BOOKMARK [1][-]{section.8.3}{\376\377\000O\000p\000c\000o\000d\000e\000\040\000t\000a\000b\000l\000e}{chapter.8}% 57
|
||||||
|
\BOOKMARK [1][-]{section.8.4}{\376\377\000S\000E\000T\000\137\000B\000A\000S\000E\000\040\000s\000e\000l\000e\000c\000t\000o\000r\000s}{chapter.8}% 58
|
||||||
|
\BOOKMARK [1][-]{section.8.5}{\376\377\000S\000T\000A\000T\000U\000S\000.\000d\000o\000n\000e\000\040\000s\000t\000i\000c\000k\000y\000\040\000/\000\040\000c\000l\000e\000a\000r\000-\000o\000n\000-\000r\000e\000a\000d}{chapter.8}% 59
|
||||||
|
\BOOKMARK [1][-]{section.8.6}{\376\377\000H\000o\000s\000t\000\040\000a\000t\000t\000e\000n\000t\000i\000o\000n\000\040\000p\000i\000n\000s\000\040\000\050\000d\000a\000t\000a\000\137\000r\000e\000a\000d\000y\000\137\000n\000,\000\040\000i\000r\000q\000\137\000n\000\051}{chapter.8}% 60
|
||||||
|
\BOOKMARK [1][-]{section.8.7}{\376\377\000R\000E\000A\000D\000\137\000C\000O\000N\000F\000I\000G}{chapter.8}% 61
|
||||||
|
\BOOKMARK [1][-]{section.8.8}{\376\377\000F\000l\000a\000s\000h\000\040\000s\000u\000b\000s\000y\000s\000t\000e\000m\000\040\000\050\000o\000p\000c\000o\000d\000e\000s\000\040\0000\000x\0004\0000\040\023\0000\000x\0004\0007\000,\000\040\000c\000o\000m\000p\000l\000e\000t\000e\000d\000\040\0002\0000\0002\0006\000-\0000\0009\000-\0000\0004\000\051}{chapter.8}% 62
|
||||||
|
\BOOKMARK [1][-]{section.8.9}{\376\377\000S\000e\000s\000s\000i\000o\000n\000\040\000s\000e\000q\000u\000e\000n\000c\000e\000s}{chapter.8}% 63
|
||||||
|
\BOOKMARK [2][-]{subsection.8.9.1}{\376\377\000S\000i\000n\000g\000l\000e\000-\000l\000a\000y\000e\000r\000\040\000p\000a\000t\000h}{section.8.9}% 64
|
||||||
|
\BOOKMARK [2][-]{subsection.8.9.2}{\376\377\000M\000u\000l\000t\000i\000-\000l\000a\000y\000e\000r\000\040\000p\000a\000t\000h\000\040\000\050\000R\000U\000N\000\137\000N\000E\000T\000W\000O\000R\000K\000\051}{section.8.9}% 65
|
||||||
|
\BOOKMARK [0][-]{chapter.9}{\376\377\000N\000e\000t\000w\000o\000r\000k\000\040\000p\000r\000o\000g\000r\000a\000m\000m\000i\000n\000g}{}% 66
|
||||||
|
\BOOKMARK [1][-]{section.9.1}{\376\377\000G\000e\000n\000e\000r\000a\000l\000\040\000f\000l\000o\000w}{chapter.9}% 67
|
||||||
|
\BOOKMARK [1][-]{section.9.2}{\376\377\000R\000e\000g\000i\000s\000t\000e\000r\000s\000\040\000a\000n\000d\000\040\000o\000p\000c\000o\000d\000e\000s\000\040\000i\000n\000v\000o\000l\000v\000e\000d}{chapter.9}% 68
|
||||||
|
\BOOKMARK [1][-]{section.9.3}{\376\377\000T\000y\000p\000e\000\040\000\043\0001\000\040\040\024\000\040\000d\000e\000n\000s\000e\000\040\000n\000e\000t\000w\000o\000r\000k}{chapter.9}% 69
|
||||||
|
\BOOKMARK [2][-]{subsection.9.3.1}{\376\377\000M\000e\000m\000o\000r\000y\000\040\000l\000a\000y\000o\000u\000t}{section.9.3}% 70
|
||||||
|
\BOOKMARK [2][-]{subsection.9.3.2}{\376\377\000W\000o\000r\000k\000e\000d\000\040\000e\000x\000a\000m\000p\000l\000e\000:\000\040\000a\000\040\0004\0004\0002\000\040\000n\000e\000t\000w\000o\000r\000k}{section.9.3}% 71
|
||||||
|
\BOOKMARK [2][-]{subsection.9.3.3}{\376\377\000H\000o\000s\000t\000\040\000p\000s\000e\000u\000d\000o\000c\000o\000d\000e\000\040\000\050\000d\000e\000n\000s\000e\000\051}{section.9.3}% 72
|
||||||
|
\BOOKMARK [1][-]{section.9.4}{\376\377\000T\000y\000p\000e\000\040\000\043\0002\000\040\040\024\000\040\000g\000r\000a\000p\000h\000\040\000n\000e\000t\000w\000o\000r\000k}{chapter.9}% 73
|
||||||
|
\BOOKMARK [2][-]{subsection.9.4.1}{\376\377\000M\000e\000m\000o\000r\000y\000\040\000l\000a\000y\000o\000u\000t}{section.9.4}% 74
|
||||||
|
\BOOKMARK [2][-]{subsection.9.4.2}{\376\377\000W\000o\000r\000k\000e\000d\000\040\000e\000x\000a\000m\000p\000l\000e}{section.9.4}% 75
|
||||||
|
\BOOKMARK [2][-]{subsection.9.4.3}{\376\377\000H\000o\000s\000t\000\040\000p\000s\000e\000u\000d\000o\000c\000o\000d\000e\000\040\000\050\000g\000r\000a\000p\000h\000\051}{section.9.4}% 76
|
||||||
|
\BOOKMARK [2][-]{subsection.9.4.4}{\376\377\000n\000e\000t\000a\000s\000m\000\040\000p\000s\000e\000u\000d\000o\000-\000a\000s\000s\000e\000m\000b\000l\000y}{section.9.4}% 77
|
||||||
|
\BOOKMARK [0][-]{chapter.10}{\376\377\000A\000r\000b\000i\000t\000r\000a\000t\000i\000o\000n\000\040\000a\000n\000d\000\040\000t\000o\000p\000-\000l\000e\000v\000e\000l}{}% 78
|
||||||
|
\BOOKMARK [1][-]{section.10.1}{\376\377\000m\000e\000m\000\137\000a\000r\000b\000i\000t\000e\000r\000\040\040\024\000\040\000t\000h\000r\000e\000e\000-\000p\000o\000r\000t\000\040\000a\000r\000b\000i\000t\000e\000r}{chapter.10}% 79
|
||||||
|
\BOOKMARK [1][-]{section.10.2}{\376\377\000s\000p\000i\000\137\000n\000e\000u\000r\000o\000n\000\137\000t\000o\000p\000\040\040\024\000\040\000f\000u\000l\000l\000\040\000i\000n\000t\000e\000g\000r\000a\000t\000i\000o\000n}{chapter.10}% 80
|
||||||
|
\BOOKMARK [0][-]{chapter.11}{\376\377\000E\000C\000P\0005\000\040\000i\000m\000p\000l\000e\000m\000e\000n\000t\000a\000t\000i\000o\000n}{}% 81
|
||||||
|
\BOOKMARK [1][-]{section.11.1}{\376\377\000F\000l\000o\000w\000\040\000a\000n\000d\000\040\000v\000e\000r\000i\000f\000i\000c\000a\000t\000i\000o\000n}{chapter.11}% 82
|
||||||
|
\BOOKMARK [1][-]{section.11.2}{\376\377\000D\000a\000t\000a\000p\000a\000t\000h\000\040\000b\000e\000n\000c\000h\000m\000a\000r\000k\000\040\000\050\0002\0005\0006\0004\000\051}{chapter.11}% 83
|
||||||
|
\BOOKMARK [2][-]{subsection.11.2.1}{\376\377\000F\000m\000a\000x\000\040\000a\000n\000d\000\040\000t\000h\000r\000o\000u\000g\000h\000p\000u\000t\000\040\000v\000e\000r\000s\000u\000s\000\040\000p\000a\000r\000a\000l\000l\000e\000l\000i\000s\000m}{section.11.2}% 84
|
||||||
|
\BOOKMARK [2][-]{subsection.11.2.2}{\376\377\000I\000n\000t\000e\000r\000p\000r\000e\000t\000a\000t\000i\000o\000n}{section.11.2}% 85
|
||||||
|
\BOOKMARK [2][-]{subsection.11.2.3}{\376\377\000C\000r\000i\000t\000i\000c\000a\000l\000\040\000p\000a\000t\000h\000\040\000a\000n\000d\000\040\000t\000h\000e\000\040\0001\0000\0000\000\040\000M\000H\000z\000\040\000l\000i\000m\000i\000t}{section.11.2}% 86
|
||||||
|
\BOOKMARK [1][-]{section.11.3}{\376\377\000F\000u\000l\000l\000\040\000i\000n\000t\000e\000g\000r\000a\000t\000e\000d\000\040\000s\000y\000s\000t\000e\000m}{chapter.11}% 87
|
||||||
|
\BOOKMARK [2][-]{subsection.11.3.1}{\376\377\000C\000a\000u\000s\000e\000:\000\040\000t\000h\000e\000\040\000s\000a\000t\000u\000r\000a\000t\000i\000o\000n\000/\000R\000e\000L\000U\000\040\000c\000a\000r\000r\000y\000\040\000c\000h\000a\000i\000n}{section.11.3}% 88
|
||||||
|
\BOOKMARK [2][-]{subsection.11.3.2}{\376\377\000T\000i\000m\000i\000n\000g\000\040\000c\000l\000o\000s\000u\000r\000e\000\040\000\050\0002\0000\0002\0006\000-\0000\0009\000-\0000\0003\000\051}{section.11.3}% 89
|
||||||
|
\BOOKMARK [0][-]{chapter.12}{\376\377\000H\000a\000r\000d\000w\000a\000r\000e\000\040\000d\000e\000s\000i\000g\000n\000\040\000a\000n\000d\000\040\000p\000i\000n\000o\000u\000t}{}% 90
|
||||||
|
\BOOKMARK [1][-]{section.12.1}{\376\377\000T\000a\000r\000g\000e\000t\000\040\000d\000e\000v\000i\000c\000e}{chapter.12}% 91
|
||||||
|
\BOOKMARK [1][-]{section.12.2}{\376\377\000P\000i\000n\000\040\000b\000u\000d\000g\000e\000t}{chapter.12}% 92
|
||||||
|
\BOOKMARK [1][-]{section.12.3}{\376\377\000S\000i\000g\000n\000a\000l\000\040\000m\000a\000p\000\040\000\050\000t\000o\000p\000-\000l\000e\000v\000e\000l\000\040\000s\000p\000i\000\137\000n\000e\000u\000r\000o\000n\000\137\000t\000o\000p\000\051\000\040\040\024\000\040\000r\000e\000a\000l\000\040\000b\000a\000l\000l\000s}{chapter.12}% 93
|
||||||
|
\BOOKMARK [1][-]{section.12.4}{\376\377\000P\000e\000r\000-\000b\000a\000n\000k\000\040\000a\000l\000l\000o\000c\000a\000t\000i\000o\000n\000\040\000\050\000r\000e\000a\000l\000\040\000d\000i\000e\000\040\000g\000e\000o\000m\000e\000t\000r\000y\000\051}{chapter.12}% 94
|
||||||
|
\BOOKMARK [1][-]{section.12.5}{\376\377\000P\000S\000R\000A\000M\000\040\000s\000u\000b\000s\000y\000s\000t\000e\000m}{chapter.12}% 95
|
||||||
|
\BOOKMARK [2][-]{subsection.12.5.1}{\376\377\000P\000S\000R\000A\000M\000\040\000c\000o\000n\000n\000e\000c\000t\000i\000o\000n\000\040\000\050\000F\000P\000G\000A\000-\000e\000x\000c\000l\000u\000s\000i\000v\000e\000\051}{section.12.5}% 96
|
||||||
|
\BOOKMARK [1][-]{section.12.6}{\376\377\000C\000l\000o\000c\000k}{chapter.12}% 97
|
||||||
|
\BOOKMARK [1][-]{section.12.7}{\376\377\000P\000o\000w\000e\000r}{chapter.12}% 98
|
||||||
|
\BOOKMARK [1][-]{section.12.8}{\376\377\000C\000o\000n\000f\000i\000g\000u\000r\000a\000t\000i\000o\000n\000\040\000a\000n\000d\000\040\000p\000r\000o\000g\000r\000a\000m\000m\000i\000n\000g}{chapter.12}% 99
|
||||||
|
\BOOKMARK [2][-]{subsection.12.8.1}{\376\377\000J\000T\000A\000G\000\040\000\050\000d\000e\000v\000e\000l\000o\000p\000m\000e\000n\000t\000\040\000/\000\040\000d\000e\000b\000u\000g\000\051}{section.12.8}% 100
|
||||||
|
\BOOKMARK [2][-]{subsection.12.8.2}{\376\377\000C\000o\000n\000f\000i\000g\000-\000S\000P\000I\000\040\000t\000o\000\040\000b\000o\000o\000t\000\040\000f\000l\000a\000s\000h}{section.12.8}% 101
|
||||||
|
\BOOKMARK [2][-]{subsection.12.8.3}{\376\377\000C\000o\000n\000f\000i\000g\000u\000r\000a\000t\000i\000o\000n\000\040\000m\000o\000d\000e\000s\000\040\000\050\000C\000F\000G\000M\000D\000N\000\051}{section.12.8}% 102
|
||||||
|
\BOOKMARK [1][-]{section.12.9}{\376\377\000O\000p\000e\000n\000\040\000t\000a\000s\000k\000s\000\040\000b\000e\000f\000o\000r\000e\000\040\000s\000c\000h\000e\000m\000a\000t\000i\000c\000\040\000c\000a\000p\000t\000u\000r\000e}{chapter.12}% 103
|
||||||
|
\BOOKMARK [0][-]{chapter.13}{\376\377\000Q\000u\000i\000c\000k\000\040\000r\000e\000f\000e\000r\000e\000n\000c\000e}{}% 104
|
||||||
|
\BOOKMARK [1][-]{section.13.1}{\376\377\000S\000P\000I\000\040\000o\000p\000c\000o\000d\000e\000s}{chapter.13}% 105
|
||||||
|
\BOOKMARK [1][-]{section.13.2}{\376\377\000S\000T\000A\000T\000U\000S\000\040\000b\000y\000t\000e}{chapter.13}% 106
|
||||||
|
\BOOKMARK [1][-]{section.13.3}{\376\377\000S\000E\000T\000\137\000B\000A\000S\000E\000\040\000s\000e\000l\000e\000c\000t\000o\000r\000s}{chapter.13}% 107
|
||||||
|
\BOOKMARK [1][-]{section.13.4}{\376\377\000D\000e\000s\000c\000r\000i\000p\000t\000o\000r\000\040\000t\000a\000b\000l\000e\000\040\000\050\0001\0001\000\040\000b\000y\000t\000e\000s\000/\000l\000a\000y\000e\000r\000,\000\040\000M\000S\000B\000-\000f\000i\000r\000s\000t\000\051}{chapter.13}% 108
|
||||||
|
\BOOKMARK [1][-]{section.13.5}{\376\377\000B\000u\000i\000l\000d\000\040\000p\000a\000r\000a\000m\000e\000t\000e\000r\000s}{chapter.13}% 109
|
||||||
|
\BOOKMARK [0][-]{chapter.14}{\376\377\000R\000o\000a\000d\000m\000a\000p\000\040\000a\000n\000d\000\040\000d\000e\000v\000e\000l\000o\000p\000m\000e\000n\000t\000\040\000s\000t\000a\000t\000u\000s}{}% 110
|
||||||
|
\BOOKMARK [1][-]{section.14.1}{\376\377\000D\000e\000v\000e\000l\000o\000p\000m\000e\000n\000t\000\040\000p\000h\000a\000s\000e\000s}{chapter.14}% 111
|
||||||
|
\BOOKMARK [1][-]{section.14.2}{\376\377\000C\000o\000m\000p\000o\000n\000e\000n\000t\000\040\000s\000t\000a\000t\000u\000s}{chapter.14}% 112
|
||||||
|
\BOOKMARK [1][-]{section.14.3}{\376\377\000A\000r\000c\000h\000i\000t\000e\000c\000t\000u\000r\000a\000l\000\040\000p\000r\000i\000n\000c\000i\000p\000l\000e\000\040\000\050\000s\000u\000m\000m\000a\000r\000y\000\051}{chapter.14}% 113
|
||||||
|
\BOOKMARK [1][-]{section.14.4}{\376\377\000L\000o\000n\000g\000-\000t\000e\000r\000m\000\040\000v\000i\000s\000i\000o\000n}{chapter.14}% 114
|
||||||
|
\BOOKMARK [0][-]{appendix.A}{\376\377\000M\000o\000d\000u\000l\000e\000s\000\040\000a\000n\000d\000\040\000t\000o\000o\000l\000c\000h\000a\000i\000n}{}% 115
|
||||||
|
\BOOKMARK [1][-]{section.A.1}{\376\377\000L\000i\000s\000t\000\040\000o\000f\000\040\000R\000T\000L\000\040\000m\000o\000d\000u\000l\000e\000s}{appendix.A}% 116
|
||||||
|
\BOOKMARK [1][-]{section.A.2}{\376\377\000P\000o\000r\000t\000s\000\040\000o\000f\000\040\000t\000h\000e\000\040\000t\000o\000p\000-\000l\000e\000v\000e\000l\000\040\000s\000p\000i\000\137\000n\000e\000u\000r\000o\000n\000\137\000t\000o\000p}{appendix.A}% 117
|
||||||
|
\BOOKMARK [1][-]{section.A.3}{\376\377\000T\000o\000o\000l\000c\000h\000a\000i\000n}{appendix.A}% 118
|
||||||
|
\BOOKMARK [2][-]{subsection.A.3.1}{\376\377\000M\000a\000i\000n\000\040\000n\000e\000x\000t\000p\000n\000r\000\040\000p\000a\000r\000a\000m\000e\000t\000e\000r\000s}{section.A.3}% 119
|
||||||
|
\BOOKMARK [2][-]{subsection.A.3.2}{\376\377\000S\000i\000m\000u\000l\000a\000t\000i\000o\000n\000\040\000e\000x\000a\000m\000p\000l\000e}{section.A.3}% 120
|
||||||
|
\BOOKMARK [1][-]{section.A.4}{\376\377\000M\000a\000i\000n\000\040\000t\000e\000s\000t\000b\000e\000n\000c\000h\000e\000s}{appendix.A}% 121
|
||||||
Binary file not shown.
@@ -0,0 +1,119 @@
|
|||||||
|
% ======================================================================
|
||||||
|
% FPGA-Neural -- INT8 Neural Network Engine
|
||||||
|
% Datasheet / Technical reference manual
|
||||||
|
% Repository: github.com/manvalan/FPGA-Neural
|
||||||
|
% ======================================================================
|
||||||
|
\documentclass[11pt,a4paper,openany]{report}
|
||||||
|
|
||||||
|
\newcommand{\datasheetrev}{A1}
|
||||||
|
\newcommand{\datasheetdate}{September 2026}
|
||||||
|
|
||||||
|
\input{preamble}
|
||||||
|
|
||||||
|
\begin{document}
|
||||||
|
\sloppy
|
||||||
|
|
||||||
|
% ======================================================================
|
||||||
|
% TITLE PAGE
|
||||||
|
% ======================================================================
|
||||||
|
\begin{titlepage}
|
||||||
|
\thispagestyle{empty}
|
||||||
|
\begin{tikzpicture}[remember picture,overlay]
|
||||||
|
\fill[fnDark] (current page.north west) rectangle
|
||||||
|
([yshift=-4.3cm]current page.north east);
|
||||||
|
\fill[fnTeal] ([yshift=-4.3cm]current page.north west) rectangle
|
||||||
|
([yshift=-4.55cm]current page.north east);
|
||||||
|
\node[anchor=north west,text=white,font=\Huge\bfseries]
|
||||||
|
at ([xshift=2.2cm,yshift=-1.15cm]current page.north west)
|
||||||
|
{FPGA\,--\,Neural};
|
||||||
|
\node[anchor=north west,text=fnLight,font=\large]
|
||||||
|
at ([xshift=2.25cm,yshift=-2.15cm]current page.north west)
|
||||||
|
{INT8 Neural Network Engine for FPGA};
|
||||||
|
\node[anchor=north west,text=fnLight2,font=\normalsize]
|
||||||
|
at ([xshift=2.25cm,yshift=-2.85cm]current page.north west)
|
||||||
|
{Parametric hardware accelerator -- Datasheet and reference manual};
|
||||||
|
\node[anchor=north east,text=white,font=\ttfamily\small]
|
||||||
|
at ([xshift=-2.2cm,yshift=-3.55cm]current page.north east)
|
||||||
|
{Rev.~\datasheetrev~~\textbullet~~\datasheetdate};
|
||||||
|
\end{tikzpicture}
|
||||||
|
|
||||||
|
\vspace*{5.0cm}
|
||||||
|
|
||||||
|
% --- compact block diagram on the title page ---
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[node distance=7mm and 12mm]
|
||||||
|
\node[fnblockD,minimum width=30mm] (host) {HOST\\{\scriptsize Linux / ESP32 / MCU / PC}};
|
||||||
|
\node[fnblockT,right=18mm of host,minimum width=34mm] (fpga)
|
||||||
|
{FPGA\\{\scriptsize Neural Network Engine}};
|
||||||
|
\node[fnblock,right=18mm of fpga,minimum width=26mm] (ram)
|
||||||
|
{PSRAM\\{\scriptsize 8\,MB dedicated}};
|
||||||
|
\draw[fnbus] (host) -- node[fnlbl,above]{SPI Mode 0} (fpga);
|
||||||
|
\draw[fnbus] (fpga) -- node[fnlbl,above]{async 16-bit} (ram);
|
||||||
|
\node[below=1mm of fpga,font=\scriptsize\itshape,text=fnGrey]
|
||||||
|
{computation entirely on-chip};
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
\vfill
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}
|
||||||
|
\node[draw=fnRule,rounded corners=3pt,inner sep=10pt,fill=fnLight,text width=15.5cm]{
|
||||||
|
\footnotesize
|
||||||
|
\textbf{\color{fnDark}Reference target device:} Lattice ECP5 \code{LFE5U-45F-8BG381C}
|
||||||
|
(speed grade $-8$, CABGA381, 72$\times$MULT18X18D, $\approx$44k LUT).\\[2pt]
|
||||||
|
\textbf{\color{fnDark}Baseline configuration:} INT8/INT32, \code{N\_INPUTS}=256, \code{N\_NEURONS}=4,
|
||||||
|
parametric \code{PARALLEL}, PSRAM working memory ISSI \code{IS66WVE4M16EBLL-70BLI}.\\[2pt]
|
||||||
|
\textbf{\color{fnDark}Status:} RTL verified in simulation (Icarus) and real synthesis
|
||||||
|
(Yosys + nextpnr-ecp5). Document describing the project as of \datasheetdate.
|
||||||
|
};
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
\vspace{0.6cm}
|
||||||
|
{\footnotesize\color{fnGrey}\raggedright
|
||||||
|
Project author: Michele Bigi \textbullet{} MIKILAB / manvalan.\\
|
||||||
|
This datasheet documents the RTL code, documentation and benchmarks
|
||||||
|
present in the repository \texttt{github.com/manvalan/FPGA-Neural}.\par}
|
||||||
|
\end{titlepage}
|
||||||
|
|
||||||
|
% ======================================================================
|
||||||
|
% "FEATURES" PAGE (datasheet style)
|
||||||
|
% ======================================================================
|
||||||
|
\input{chapters/00-features}
|
||||||
|
|
||||||
|
% ======================================================================
|
||||||
|
% PINOUT SUMMARY (pages 2-3, pin-by-pin -- not bus ranges)
|
||||||
|
% ======================================================================
|
||||||
|
\newpage
|
||||||
|
\input{chapters/00b-pinout}
|
||||||
|
|
||||||
|
% ======================================================================
|
||||||
|
% TABLE OF CONTENTS
|
||||||
|
% ======================================================================
|
||||||
|
\newpage
|
||||||
|
\pagenumbering{roman}
|
||||||
|
{\color{fnDark}\tableofcontents}
|
||||||
|
\newpage
|
||||||
|
\pagenumbering{arabic}
|
||||||
|
|
||||||
|
% ======================================================================
|
||||||
|
% CHAPTERS
|
||||||
|
% ======================================================================
|
||||||
|
\include{chapters/01-overview}
|
||||||
|
\include{chapters/02-architettura}
|
||||||
|
\include{chapters/03-datapath}
|
||||||
|
\include{chapters/04-parametri}
|
||||||
|
\include{chapters/05-memoria}
|
||||||
|
\include{chapters/06-sequencer}
|
||||||
|
\include{chapters/06b-grafo}
|
||||||
|
\include{chapters/07-spi}
|
||||||
|
\include{chapters/07b-programmazione}
|
||||||
|
\include{chapters/08-toplevel}
|
||||||
|
\include{chapters/09-implementazione}
|
||||||
|
\include{chapters/10-hardware}
|
||||||
|
\include{chapters/11-registri}
|
||||||
|
\include{chapters/12-roadmap}
|
||||||
|
|
||||||
|
\appendix
|
||||||
|
\include{chapters/A-moduli}
|
||||||
|
|
||||||
|
\end{document}
|
||||||
@@ -0,0 +1,122 @@
|
|||||||
|
\babel@toc {english}{}\relax
|
||||||
|
\contentsline {chapter}{\numberline {1}System overview}{1}{chapter.1}%
|
||||||
|
\contentsline {section}{\numberline {1.1}Project goal}{1}{section.1.1}%
|
||||||
|
\contentsline {section}{\numberline {1.2}Hardware configuration versus network configuration}{1}{section.1.2}%
|
||||||
|
\contentsline {section}{\numberline {1.3}Boot and initialization}{2}{section.1.3}%
|
||||||
|
\contentsline {section}{\numberline {1.4}Training and inference}{2}{section.1.4}%
|
||||||
|
\contentsline {section}{\numberline {1.5}Design philosophy and reuse}{2}{section.1.5}%
|
||||||
|
\contentsline {chapter}{\numberline {2}RTL architecture}{3}{chapter.2}%
|
||||||
|
\contentsline {section}{\numberline {2.1}Hierarchical organization}{3}{section.2.1}%
|
||||||
|
\contentsline {section}{\numberline {2.2}Role of each module}{3}{section.2.2}%
|
||||||
|
\contentsline {section}{\numberline {2.3}Two execution paths}{4}{section.2.3}%
|
||||||
|
\contentsline {chapter}{\numberline {3}Compute datapath}{6}{chapter.3}%
|
||||||
|
\contentsline {section}{\numberline {3.1}INT8/INT32 arithmetic chain}{6}{section.3.1}%
|
||||||
|
\contentsline {section}{\numberline {3.2}\texttt {mac\_unit} --- multiply-accumulator}{6}{section.3.2}%
|
||||||
|
\contentsline {section}{\numberline {3.3}\texttt {mac8} --- parallel MAC and balanced adder tree}{6}{section.3.3}%
|
||||||
|
\contentsline {section}{\numberline {3.4}\texttt {neuron\_parallel} --- neuron FSM}{7}{section.3.4}%
|
||||||
|
\contentsline {subsection}{\numberline {3.4.1}Parameter guard (elaboration-time)}{7}{subsection.3.4.1}%
|
||||||
|
\contentsline {section}{\numberline {3.5}Activation functions}{8}{section.3.5}%
|
||||||
|
\contentsline {section}{\numberline {3.6}INT8 saturation}{8}{section.3.6}%
|
||||||
|
\contentsline {section}{\numberline {3.7}\texttt {layer} --- neurons in parallel}{8}{section.3.7}%
|
||||||
|
\contentsline {chapter}{\numberline {4}Parameters and configurability}{9}{chapter.4}%
|
||||||
|
\contentsline {section}{\numberline {4.1}Build parameters (synthesis-time)}{9}{section.4.1}%
|
||||||
|
\contentsline {section}{\numberline {4.2}Runtime network width}{9}{section.4.2}%
|
||||||
|
\contentsline {subsection}{\numberline {4.2.1}Measured savings}{10}{subsection.4.2.1}%
|
||||||
|
\contentsline {section}{\numberline {4.3}Characterized configurations}{10}{section.4.3}%
|
||||||
|
\contentsline {section}{\numberline {4.4}Build versus runtime summary}{10}{section.4.4}%
|
||||||
|
\contentsline {chapter}{\numberline {5}Memory subsystem}{11}{chapter.5}%
|
||||||
|
\contentsline {section}{\numberline {5.1}Memory chain}{11}{section.5.1}%
|
||||||
|
\contentsline {section}{\numberline {5.2}\texttt {int8\_memory\_access} --- byte/word conversion}{11}{section.5.2}%
|
||||||
|
\contentsline {section}{\numberline {5.3}\texttt {memory\_interface} --- handshake}{11}{section.5.3}%
|
||||||
|
\contentsline {section}{\numberline {5.4}\texttt {psram\_controller} --- physical bus}{11}{section.5.4}%
|
||||||
|
\contentsline {subsection}{\numberline {5.4.1}Timing}{12}{subsection.5.4.1}%
|
||||||
|
\contentsline {section}{\numberline {5.5}Read page mode}{12}{section.5.5}%
|
||||||
|
\contentsline {section}{\numberline {5.6}Address map and conventions}{13}{section.5.6}%
|
||||||
|
\contentsline {subsection}{\numberline {5.6.1}PSRAM physical addressing}{13}{subsection.5.6.1}%
|
||||||
|
\contentsline {section}{\numberline {5.7}Bandwidth}{13}{section.5.7}%
|
||||||
|
\contentsline {chapter}{\numberline {6}Memory, multi-neuron and multi-layer}{15}{chapter.6}%
|
||||||
|
\contentsline {section}{\numberline {6.1}\texttt {neuron\_memory} --- memory/neuron bridge}{15}{section.6.1}%
|
||||||
|
\contentsline {section}{\numberline {6.2}\texttt {layer\_sequencer} --- multi-layer network}{15}{section.6.2}%
|
||||||
|
\contentsline {subsection}{\numberline {6.2.1}Ping-pong buffers}{16}{subsection.6.2.1}%
|
||||||
|
\contentsline {subsection}{\numberline {6.2.2}Descriptor table}{16}{subsection.6.2.2}%
|
||||||
|
\contentsline {section}{\numberline {6.3}Hierarchy of the \texttt {busy}/\texttt {done} signals}{16}{section.6.3}%
|
||||||
|
\contentsline {chapter}{\numberline {7}Graph network (Type \#2)}{17}{chapter.7}%
|
||||||
|
\contentsline {section}{\numberline {7.1}Two network types}{17}{section.7.1}%
|
||||||
|
\contentsline {section}{\numberline {7.2}Global activation buffer}{17}{section.7.2}%
|
||||||
|
\contentsline {section}{\numberline {7.3}Feed-forward DAG and the \texttt {src\_id < out\_id} rule}{17}{section.7.3}%
|
||||||
|
\contentsline {section}{\numberline {7.4}Data formats}{17}{section.7.4}%
|
||||||
|
\contentsline {subsection}{\numberline {7.4.1}Type \#2 descriptor (graph)}{18}{subsection.7.4.1}%
|
||||||
|
\contentsline {subsection}{\numberline {7.4.2}Graph edge (4~bytes, aligned)}{18}{subsection.7.4.2}%
|
||||||
|
\contentsline {section}{\numberline {7.5}\texttt {graph\_engine} --- graph engine}{18}{section.7.5}%
|
||||||
|
\contentsline {section}{\numberline {7.6}Type \#2 opcodes and registers}{19}{section.7.6}%
|
||||||
|
\contentsline {section}{\numberline {7.7}Occupancy (Type \#2 enabled)}{19}{section.7.7}%
|
||||||
|
\contentsline {section}{\numberline {7.8}Gather bandwidth (measured)}{20}{section.7.8}%
|
||||||
|
\contentsline {section}{\numberline {7.9}\texttt {netasm} host assembler}{20}{section.7.9}%
|
||||||
|
\contentsline {chapter}{\numberline {8}SPI host interface}{21}{chapter.8}%
|
||||||
|
\contentsline {section}{\numberline {8.1}Physical layer}{21}{section.8.1}%
|
||||||
|
\contentsline {section}{\numberline {8.2}Framing and explicit length}{21}{section.8.2}%
|
||||||
|
\contentsline {section}{\numberline {8.3}Opcode table}{21}{section.8.3}%
|
||||||
|
\contentsline {section}{\numberline {8.4}\texttt {SET\_BASE} selectors}{23}{section.8.4}%
|
||||||
|
\contentsline {section}{\numberline {8.5}\texttt {STATUS.done} sticky / clear-on-read}{24}{section.8.5}%
|
||||||
|
\contentsline {section}{\numberline {8.6}Host attention pins (\texttt {data\_ready\_n}, \texttt {irq\_n})}{24}{section.8.6}%
|
||||||
|
\contentsline {section}{\numberline {8.7}\texttt {READ\_CONFIG}}{25}{section.8.7}%
|
||||||
|
\contentsline {section}{\numberline {8.8}Flash subsystem (opcodes 0x40--0x47, completed 2026-09-04)}{25}{section.8.8}%
|
||||||
|
\contentsline {section}{\numberline {8.9}Session sequences}{26}{section.8.9}%
|
||||||
|
\contentsline {subsection}{\numberline {8.9.1}Single-layer path}{26}{subsection.8.9.1}%
|
||||||
|
\contentsline {subsection}{\numberline {8.9.2}Multi-layer path (RUN\_NETWORK)}{26}{subsection.8.9.2}%
|
||||||
|
\contentsline {chapter}{\numberline {9}Network programming}{28}{chapter.9}%
|
||||||
|
\contentsline {section}{\numberline {9.1}General flow}{28}{section.9.1}%
|
||||||
|
\contentsline {section}{\numberline {9.2}Registers and opcodes involved}{28}{section.9.2}%
|
||||||
|
\contentsline {section}{\numberline {9.3}Type \#1 --- dense network}{29}{section.9.3}%
|
||||||
|
\contentsline {subsection}{\numberline {9.3.1}Memory layout}{29}{subsection.9.3.1}%
|
||||||
|
\contentsline {subsection}{\numberline {9.3.2}Worked example: a $4\to 4\to 2$ network}{29}{subsection.9.3.2}%
|
||||||
|
\contentsline {subsection}{\numberline {9.3.3}Host pseudocode (dense)}{29}{subsection.9.3.3}%
|
||||||
|
\contentsline {section}{\numberline {9.4}Type \#2 --- graph network}{30}{section.9.4}%
|
||||||
|
\contentsline {subsection}{\numberline {9.4.1}Memory layout}{30}{subsection.9.4.1}%
|
||||||
|
\contentsline {subsection}{\numberline {9.4.2}Worked example}{30}{subsection.9.4.2}%
|
||||||
|
\contentsline {subsection}{\numberline {9.4.3}Host pseudocode (graph)}{30}{subsection.9.4.3}%
|
||||||
|
\contentsline {subsection}{\numberline {9.4.4}\texttt {netasm} pseudo-assembly}{31}{subsection.9.4.4}%
|
||||||
|
\contentsline {chapter}{\numberline {10}Arbitration and top-level}{32}{chapter.10}%
|
||||||
|
\contentsline {section}{\numberline {10.1}\texttt {mem\_arbiter} --- three-port arbiter}{32}{section.10.1}%
|
||||||
|
\contentsline {section}{\numberline {10.2}\texttt {spi\_neuron\_top} --- full integration}{32}{section.10.2}%
|
||||||
|
\contentsline {chapter}{\numberline {11}ECP5 implementation}{34}{chapter.11}%
|
||||||
|
\contentsline {section}{\numberline {11.1}Flow and verification}{34}{section.11.1}%
|
||||||
|
\contentsline {section}{\numberline {11.2}Datapath benchmark (256$\times $4)}{34}{section.11.2}%
|
||||||
|
\contentsline {subsection}{\numberline {11.2.1}Fmax and throughput versus parallelism}{35}{subsection.11.2.1}%
|
||||||
|
\contentsline {subsection}{\numberline {11.2.2}Interpretation}{35}{subsection.11.2.2}%
|
||||||
|
\contentsline {subsection}{\numberline {11.2.3}Critical path and the 100~MHz limit}{35}{subsection.11.2.3}%
|
||||||
|
\contentsline {section}{\numberline {11.3}Full integrated system}{35}{section.11.3}%
|
||||||
|
\contentsline {subsection}{\numberline {11.3.1}Cause: the saturation/ReLU carry chain}{35}{subsection.11.3.1}%
|
||||||
|
\contentsline {subsection}{\numberline {11.3.2}Timing closure (2026-09-03)}{36}{subsection.11.3.2}%
|
||||||
|
\contentsline {chapter}{\numberline {12}Hardware design and pinout}{37}{chapter.12}%
|
||||||
|
\contentsline {section}{\numberline {12.1}Target device}{37}{section.12.1}%
|
||||||
|
\contentsline {section}{\numberline {12.2}Pin budget}{37}{section.12.2}%
|
||||||
|
\contentsline {section}{\numberline {12.3}Signal map (top-level \texttt {spi\_neuron\_top}) --- real balls}{38}{section.12.3}%
|
||||||
|
\contentsline {section}{\numberline {12.4}Per-bank allocation (real die geometry)}{40}{section.12.4}%
|
||||||
|
\contentsline {section}{\numberline {12.5}PSRAM subsystem}{40}{section.12.5}%
|
||||||
|
\contentsline {subsection}{\numberline {12.5.1}PSRAM connection (FPGA-exclusive)}{40}{subsection.12.5.1}%
|
||||||
|
\contentsline {section}{\numberline {12.6}Clock}{41}{section.12.6}%
|
||||||
|
\contentsline {section}{\numberline {12.7}Power}{41}{section.12.7}%
|
||||||
|
\contentsline {section}{\numberline {12.8}Configuration and programming}{41}{section.12.8}%
|
||||||
|
\contentsline {subsection}{\numberline {12.8.1}JTAG (development / debug)}{41}{subsection.12.8.1}%
|
||||||
|
\contentsline {subsection}{\numberline {12.8.2}Config-SPI to boot flash}{41}{subsection.12.8.2}%
|
||||||
|
\contentsline {subsection}{\numberline {12.8.3}Configuration modes (\texttt {CFGMDN})}{42}{subsection.12.8.3}%
|
||||||
|
\contentsline {section}{\numberline {12.9}Open tasks before schematic capture}{42}{section.12.9}%
|
||||||
|
\contentsline {chapter}{\numberline {13}Quick reference}{44}{chapter.13}%
|
||||||
|
\contentsline {section}{\numberline {13.1}SPI opcodes}{44}{section.13.1}%
|
||||||
|
\contentsline {section}{\numberline {13.2}STATUS byte}{44}{section.13.2}%
|
||||||
|
\contentsline {section}{\numberline {13.3}SET\_BASE selectors}{44}{section.13.3}%
|
||||||
|
\contentsline {section}{\numberline {13.4}Descriptor table (11 bytes/layer, MSB-first)}{44}{section.13.4}%
|
||||||
|
\contentsline {section}{\numberline {13.5}Build parameters}{44}{section.13.5}%
|
||||||
|
\contentsline {chapter}{\numberline {14}Roadmap and development status}{45}{chapter.14}%
|
||||||
|
\contentsline {section}{\numberline {14.1}Development phases}{45}{section.14.1}%
|
||||||
|
\contentsline {section}{\numberline {14.2}Component status}{45}{section.14.2}%
|
||||||
|
\contentsline {section}{\numberline {14.3}Architectural principle (summary)}{46}{section.14.3}%
|
||||||
|
\contentsline {section}{\numberline {14.4}Long-term vision}{46}{section.14.4}%
|
||||||
|
\contentsline {chapter}{\numberline {A}Modules and toolchain}{47}{appendix.A}%
|
||||||
|
\contentsline {section}{\numberline {A.1}List of RTL modules}{47}{section.A.1}%
|
||||||
|
\contentsline {section}{\numberline {A.2}Ports of the top-level \texttt {spi\_neuron\_top}}{47}{section.A.2}%
|
||||||
|
\contentsline {section}{\numberline {A.3}Toolchain}{47}{section.A.3}%
|
||||||
|
\contentsline {subsection}{\numberline {A.3.1}Main nextpnr parameters}{47}{subsection.A.3.1}%
|
||||||
|
\contentsline {subsection}{\numberline {A.3.2}Simulation example}{48}{subsection.A.3.2}%
|
||||||
|
\contentsline {section}{\numberline {A.4}Main testbenches}{48}{section.A.4}%
|
||||||
@@ -0,0 +1,120 @@
|
|||||||
|
\thispagestyle{plain}
|
||||||
|
\noindent
|
||||||
|
\begin{tikzpicture}
|
||||||
|
\node[fill=fnDark,text=white,rounded corners=2pt,inner sep=6pt,
|
||||||
|
minimum width=\textwidth,anchor=west]
|
||||||
|
{\large\bfseries FPGA-Neural --- General description and features};
|
||||||
|
\end{tikzpicture}
|
||||||
|
|
||||||
|
\vspace{6pt}
|
||||||
|
\noindent
|
||||||
|
{\small FPGA-Neural is a \textbf{parametric hardware accelerator for feed-forward
|
||||||
|
neural networks} contained entirely within the FPGA. Computation (multiplication,
|
||||||
|
accumulation, bias, activation, saturation) takes place entirely on-chip in INT8/INT32
|
||||||
|
integer arithmetic; the host system only provides configuration, weights, input data
|
||||||
|
and control through a simple SPI interface, without ever being part of the
|
||||||
|
computational datapath. A single bitstream serves any topology up to the build
|
||||||
|
maximum.}
|
||||||
|
|
||||||
|
\vspace{8pt}
|
||||||
|
\begin{multicols}{2}
|
||||||
|
{\color{fnDark}\large\bfseries Features}\\[2pt]
|
||||||
|
{\footnotesize
|
||||||
|
\begin{itemize}[leftmargin=1.1em]
|
||||||
|
\item \textbf{INT8 $\times$ INT8 $\to$ INT16 $\to$ INT32} datapath, 32-bit accumulation
|
||||||
|
with sign extension.
|
||||||
|
\item \textbf{Balanced binary adder tree} ($O(\log_2 \text{PARALLEL})$) instead of
|
||||||
|
linear reduction.
|
||||||
|
\item Configurable parallel MAC: \code{PARALLEL} simultaneous hardware MACs per neuron,
|
||||||
|
mapped onto \code{MULT18X18D} DSPs.
|
||||||
|
\item Fully \textbf{parametric} architecture: \code{N\_INPUTS}, \code{N\_NEURONS},
|
||||||
|
\code{PARALLEL}, \code{DATA\_WIDTH}, \code{ACC\_WIDTH}, \code{N\_LAYERS}.
|
||||||
|
\item \textbf{Runtime network width}: per-layer \code{n\_inputs\_real}/\code{n\_neurons\_real},
|
||||||
|
a single bitstream for every topology up to the maximum.
|
||||||
|
\item Configurable activations: \code{ACT\_RELU} (default) and \code{ACT\_NONE} (linear
|
||||||
|
with bilateral saturation), with INT8 saturation.
|
||||||
|
\item \textbf{Two network types}: classic multi-layer dense (\code{layer\_sequencer},
|
||||||
|
ping-pong buffers) and \textbf{arbitrary sparse graph} (\code{graph\_engine} +
|
||||||
|
activation buffer in \code{DP16KD} block RAM), selectable at runtime.
|
||||||
|
\item \textbf{Dedicated memory} subsystem: byte$\leftrightarrow$word interface,
|
||||||
|
asynchronous parallel PSRAM controller with \textbf{page mode} (70~ns random
|
||||||
|
access, 20~ns page burst), 8~MB addressable (23~bit).
|
||||||
|
\item \textbf{SPI Mode 0} MSB-first host interface, \code{SET\_NET\_TYPE}+dispatch, \code{STATUS.done}
|
||||||
|
sticky/clear-on-read, runtime \code{READ\_CONFIG}.
|
||||||
|
\item \textbf{Flash subsystem} for boot/persistence: FPGA-exclusive access to a
|
||||||
|
\code{W25Q128JV} SPI NOR (16~MB) via a dedicated SPI master, a
|
||||||
|
flash$\leftrightarrow$PSRAM copy engine, and a 16-slot catalog with CRC32,
|
||||||
|
8 host opcodes.
|
||||||
|
\item Verified in \textbf{simulation} (Icarus Verilog) and \textbf{real synthesis}
|
||||||
|
(Yosys + nextpnr-ecp5 + ecppack).
|
||||||
|
\end{itemize}}
|
||||||
|
|
||||||
|
\columnbreak
|
||||||
|
|
||||||
|
{\color{fnDark}\large\bfseries Applications}\\[2pt]
|
||||||
|
{\footnotesize
|
||||||
|
\begin{itemize}[leftmargin=1.1em]
|
||||||
|
\item Deterministic low-latency inference as a peripheral of a
|
||||||
|
Linux SoC, Raspberry-Pi-like board, ESP32, microcontrollers.
|
||||||
|
\item Reusable hardware block integrable into heterogeneous projects
|
||||||
|
(a platform, not a single network).
|
||||||
|
\item Edge AI on compact dense INT8-quantized networks.
|
||||||
|
\item Off-loading the neural workload from the host CPU to dedicated
|
||||||
|
hardware with predictable throughput.
|
||||||
|
\end{itemize}}
|
||||||
|
|
||||||
|
\vspace{4pt}
|
||||||
|
{\color{fnDark}\large\bfseries Target \& toolchain}\\[2pt]
|
||||||
|
{\footnotesize
|
||||||
|
\begin{itemize}[leftmargin=1.1em]
|
||||||
|
\item FPGA: Lattice ECP5 \code{LFE5U-45F-8BG381C} ($-8$, CABGA381).
|
||||||
|
\item Synthesis: Yosys; place\&route: nextpnr-ecp5; bitstream: Project~Trellis
|
||||||
|
(\code{ecppack}).
|
||||||
|
\item Simulation: Icarus Verilog (\code{-g2012}).
|
||||||
|
\item PSRAM: ISSI \code{IS66WVE4M16EBLL-70BLI} (64\,Mb, 4M$\times$16).
|
||||||
|
\end{itemize}}
|
||||||
|
\end{multicols}
|
||||||
|
|
||||||
|
\vspace{2pt}
|
||||||
|
% --- key parameter table ---
|
||||||
|
\noindent
|
||||||
|
{\small\color{fnDark}\bfseries Key parameters (characterized baseline configuration)}
|
||||||
|
\vspace{2pt}
|
||||||
|
|
||||||
|
\noindent
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.2cm}L{3.6cm}Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Quantity} & \thd{Value} & \thd{Notes} \\
|
||||||
|
\midrule
|
||||||
|
Data precision & INT8 (signed) & \code{DATA\_WIDTH}=8 \\
|
||||||
|
\rowa Accumulator & INT32 (signed) & \code{ACC\_WIDTH}=32 \\
|
||||||
|
Inputs / neurons & 256 / 4 & datapath benchmark baseline \\
|
||||||
|
\rowa Simultaneous MACs & $2\ldots64$ & $=$\code{PARALLEL}$\times$\code{N\_NEURONS} \\
|
||||||
|
Activations & ReLU, linear & \code{ACT\_RELU} / \code{ACT\_NONE} \\
|
||||||
|
\rowa Fmax (P=2, datapath) & 87.88~MHz & isolated datapath benchmark \\
|
||||||
|
Fmax (P=2, integrated system) & 67.91~MHz & full system incl. flash subsystem, real place\&route \\
|
||||||
|
MAC throughput (P=16) & $\approx$3.34~G\,MAC/s & theoretical, datapath only \\
|
||||||
|
\rowa Working memory & 8~MB PSRAM & 16-bit parallel bus, 70~ns / 20~ns page mode \\
|
||||||
|
Address space & 23~bit (byte) & \code{ADDR\_WIDTH}=23 \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\vspace{8pt}
|
||||||
|
\noindent
|
||||||
|
{\small\color{fnDark}\bfseries System block diagram}
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[node distance=6mm and 10mm,font=\footnotesize]
|
||||||
|
\node[fnblockD,minimum width=26mm,minimum height=13mm] (host){HOST\\{\scriptsize configures / trains / controls}};
|
||||||
|
\node[fnblockT,right=16mm of host,minimum width=52mm,minimum height=22mm] (eng){};
|
||||||
|
\node[anchor=north,font=\footnotesize\bfseries,text=fnDark] at (eng.north){FPGA -- Neural Network Engine};
|
||||||
|
\node[fnreg,fill=white] (spi) at ([yshift=-2mm]eng.center){\code{spi\_slave} + \code{spi\_engine}};
|
||||||
|
\node[fnreg,fill=white,below=2.5mm of spi] (arb){\code{mem\_arbiter} + \code{layer\_sequencer}};
|
||||||
|
\node[fnreg,fill=white,above=2.5mm of spi] (core){\code{neuron\_memory} $\to$ \code{neuron\_parallel} $\to$ \code{mac8}};
|
||||||
|
\node[fnblock,right=16mm of eng,minimum width=24mm,minimum height=13mm] (ram){PSRAM 8\,MB\\{\scriptsize \code{psram\_controller}}};
|
||||||
|
\draw[fnbus] (host) -- node[fnlbl,above]{SPI} (eng.west|-host);
|
||||||
|
\draw[fnbus] (eng.east|-ram) -- node[fnlbl,above]{16-bit async} (ram);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
\begin{center}\footnotesize\itshape\color{fnGrey}
|
||||||
|
The neural datapath is entirely inside the FPGA; the host does not take part in the
|
||||||
|
individual MAC operations.\end{center}
|
||||||
@@ -0,0 +1,103 @@
|
|||||||
|
\thispagestyle{plain}
|
||||||
|
\noindent
|
||||||
|
\begin{tikzpicture}
|
||||||
|
\node[fill=fnDark,text=white,rounded corners=2pt,inner sep=6pt,
|
||||||
|
minimum width=\textwidth,anchor=west]
|
||||||
|
{\large\bfseries Pinout summary --- pin-by-pin connection};
|
||||||
|
\end{tikzpicture}
|
||||||
|
|
||||||
|
\vspace{6pt}
|
||||||
|
\noindent
|
||||||
|
{\footnotesize
|
||||||
|
Quick-reference table: the \textbf{57 real signals} of the top-level
|
||||||
|
\code{spi\_neuron\_top}, each with its own individual \code{CABGA381} ball
|
||||||
|
(\textbf{not} a bus range) --- real data from Project~Trellis's device
|
||||||
|
database (\code{iodb.json}), \textbf{verified by a complete
|
||||||
|
\code{nextpnr-ecp5} place\&route run at 0 errors} (not a planned pinout).
|
||||||
|
Full description, per-bank placement rationale and the pin-by-pin
|
||||||
|
connection to the ISSI PSRAM: ch.~\ref{ch:hw}.
|
||||||
|
}
|
||||||
|
|
||||||
|
\vspace{4pt}
|
||||||
|
\noindent
|
||||||
|
\renewcommand{\arraystretch}{1.08}
|
||||||
|
\begin{tabularx}{\textwidth}{L{2.7cm} C{1.0cm} C{1.0cm} C{0.9cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Signal} & \thd{Ball} & \thd{Bank} & \thd{Dir} & \thd{Corresponding pin / function} \\
|
||||||
|
\midrule
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}Clock and reset}}\\
|
||||||
|
\code{clk} & H5 & 7 & IN & System clock, pad \code{GR\_PCLK7\_0} (dedicated global clock). \\
|
||||||
|
\rowa \code{rst} & B4 & 7 & IN & Global synchronous reset, active high. \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}Application SPI (host $\leftrightarrow$ FPGA, Mode~0)}}\\
|
||||||
|
\code{sclk} & B5 & 7 & IN & SPI clock (CPOL=0, CPHA=0). \\
|
||||||
|
\rowa \code{mosi} & C5 & 7 & IN & Master-Out Slave-In. \\
|
||||||
|
\code{miso} & A3 & 7 & OUT & Master-In Slave-Out. \\
|
||||||
|
\rowa \code{cs\_n} & B3 & 7 & IN & Chip-select, active low. \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}Host attention (active-low, level)}}\\
|
||||||
|
\code{data\_ready\_n} & C3 & 7 & OUT & Low while a result is waiting to be read. \\
|
||||||
|
\rowa \code{irq\_n} & C4 & 7 & OUT & Low while the graph engine's load-time guard has tripped. \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}Flash subsystem --- SPI toward W25Q128JV (boot/persistence)}}\\
|
||||||
|
\code{flash\_sclk} & E3 & 7 & OUT & SPI clock toward the flash --- ordinary GPIO, independent (Phase F7, ch.~\ref{ch:hw}). \\
|
||||||
|
\rowa \code{flash\_mosi} & D3 & 7 & OUT & Master-Out Slave-In toward the onboard flash. \\
|
||||||
|
\code{flash\_miso} & D5 & 7 & IN & Master-In Slave-Out from the flash. \\
|
||||||
|
\rowa \code{flash\_cs\_n} & E4 & 7 & OUT & Flash chip-select, active low. \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}PSRAM address bus \code{psram\_a[21:0]} --- 22 individual balls (bank 2)}}\\
|
||||||
|
\code{psram\_a[0]} & E16 & 2 & OUT & PSRAM A0 \\
|
||||||
|
\rowa \code{psram\_a[1]} & F16 & 2 & OUT & PSRAM A1 \\
|
||||||
|
\code{psram\_a[2]} & D18 & 2 & OUT & PSRAM A2 \\
|
||||||
|
\rowa \code{psram\_a[3]} & E17 & 2 & OUT & PSRAM A3 \\
|
||||||
|
\code{psram\_a[4]} & E18 & 2 & OUT & PSRAM A4 \\
|
||||||
|
\rowa \code{psram\_a[5]} & F18 & 2 & OUT & PSRAM A5 \\
|
||||||
|
\code{psram\_a[6]} & F17 & 2 & OUT & PSRAM A6 \\
|
||||||
|
\rowa \code{psram\_a[7]} & G16 & 2 & OUT & PSRAM A7 \\
|
||||||
|
\code{psram\_a[8]} & G18 & 2 & OUT & PSRAM A8 \\
|
||||||
|
\rowa \code{psram\_a[9]} & H16 & 2 & OUT & PSRAM A9 \\
|
||||||
|
\code{psram\_a[10]} & H17 & 2 & OUT & PSRAM A10 \\
|
||||||
|
\rowa \code{psram\_a[11]} & H18 & 2 & OUT & PSRAM A11 \\
|
||||||
|
\code{psram\_a[12]} & J16 & 2 & OUT & PSRAM A12 \\
|
||||||
|
\rowa \code{psram\_a[13]} & J17 & 2 & OUT & PSRAM A13 \\
|
||||||
|
\code{psram\_a[14]} & C20 & 2 & OUT & PSRAM A14 \\
|
||||||
|
\rowa \code{psram\_a[15]} & D19 & 2 & OUT & PSRAM A15 \\
|
||||||
|
\code{psram\_a[16]} & E19 & 2 & OUT & PSRAM A16 \\
|
||||||
|
\rowa \code{psram\_a[17]} & E20 & 2 & OUT & PSRAM A17 \\
|
||||||
|
\code{psram\_a[18]} & F19 & 2 & OUT & PSRAM A18 \\
|
||||||
|
\rowa \code{psram\_a[19]} & F20 & 2 & OUT & PSRAM A19 \\
|
||||||
|
\code{psram\_a[20]} & G20 & 2 & OUT & PSRAM A20 \\
|
||||||
|
\rowa \code{psram\_a[21]} & H20 & 2 & OUT & PSRAM A21 \\
|
||||||
|
\code{psram\_a[22]} & P18 & 3 & OUT & Always 0 (byte$\to$word shift): NC on the board. \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}PSRAM data bus \code{psram\_dq[15:0]} --- 16 individual balls (banks 2 and 3)}}\\
|
||||||
|
\rowa \code{psram\_dq[0]} & K18 & 2 & IO & PSRAM DQ0 \\
|
||||||
|
\code{psram\_dq[1]} & C18 & 2 & IO & PSRAM DQ1 (dual-function ball, used as ordinary GPIO). \\
|
||||||
|
\rowa \code{psram\_dq[2]} & D17 & 2 & IO & PSRAM DQ2 \\
|
||||||
|
\code{psram\_dq[3]} & D20 & 2 & IO & PSRAM DQ3 \\
|
||||||
|
\rowa \code{psram\_dq[4]} & G19 & 2 & IO & PSRAM DQ4 \\
|
||||||
|
\code{psram\_dq[5]} & J18 & 2 & IO & PSRAM DQ5 \\
|
||||||
|
\rowa \code{psram\_dq[6]} & J19 & 2 & IO & PSRAM DQ6 \\
|
||||||
|
\code{psram\_dq[7]} & J20 & 2 & IO & PSRAM DQ7 \\
|
||||||
|
\rowa \code{psram\_dq[8]} & K19 & 2 & IO & PSRAM DQ8 \\
|
||||||
|
\code{psram\_dq[9]} & K20 & 2 & IO & PSRAM DQ9 \\
|
||||||
|
\rowa \code{psram\_dq[10]} & L17 & 3 & IO & PSRAM DQ10 \\
|
||||||
|
\code{psram\_dq[11]} & M18 & 3 & IO & PSRAM DQ11 \\
|
||||||
|
\rowa \code{psram\_dq[12]} & M17 & 3 & IO & PSRAM DQ12 \\
|
||||||
|
\code{psram\_dq[13]} & N16 & 3 & IO & PSRAM DQ13 \\
|
||||||
|
\rowa \code{psram\_dq[14]} & N18 & 3 & IO & PSRAM DQ14 \\
|
||||||
|
\code{psram\_dq[15]} & P17 & 3 & IO & PSRAM DQ15 \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}PSRAM control}}\\
|
||||||
|
\rowa \code{psram\_ce\_n} & N17 & 3 & OUT & PSRAM CE\# --- chip enable, active low. \\
|
||||||
|
\code{psram\_oe\_n} & R16 & 3 & OUT & PSRAM OE\# --- output enable (read). \\
|
||||||
|
\rowa \code{psram\_we\_n} & R17 & 3 & OUT & PSRAM WE\# --- write enable. \\
|
||||||
|
\code{psram\_lb\_n} & T16 & 3 & OUT & PSRAM LB\# --- lower-byte enable (DQ[7:0]). \\
|
||||||
|
\rowa \code{psram\_ub\_n} & N19 & 3 & OUT & PSRAM UB\# --- upper-byte enable (DQ[15:8]). \\
|
||||||
|
\code{psram\_zz\_n} & N20 & 3 & OUT & PSRAM ZZ\# --- sleep/snooze (high during normal operation). \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
\renewcommand{\arraystretch}{1.25}
|
||||||
|
|
||||||
|
\vspace{4pt}
|
||||||
|
\noindent
|
||||||
|
{\footnotesize\color{fnGrey}
|
||||||
|
Standard I/O: LVCMOS33 on all 57 signals. Boot config-SPI and JTAG balls (fixed-function
|
||||||
|
dedicated pins, no RTL port) do not appear in this table --- see ch.~\ref{ch:hw}
|
||||||
|
§``Configuration and programming''. Source: \code{synth/ecp5/spi\_neuron\_top.lpf},
|
||||||
|
generated by \code{tools/pinout/gen\_lpf.py} against Project~Trellis's
|
||||||
|
\code{iodb.json}.\par}
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {1}System overview}{1}{chapter.1}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:overview}{{1}{1}{System overview}{chapter.1}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {1.1}Project goal}{1}{section.1.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {1.2}Hardware configuration versus network configuration}{1}{section.1.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {1.3}Boot and initialization}{2}{section.1.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {1.4}Training and inference}{2}{section.1.4}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {1.5}Design philosophy and reuse}{2}{section.1.5}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/01-overview}{
|
||||||
|
\setcounter{page}{3}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{1}
|
||||||
|
\setcounter{section}{5}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{0}
|
||||||
|
\setcounter{LT@tables}{2}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{2}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{3}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{1}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{0}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{0}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{6}
|
||||||
|
\setcounter{lstlisting}{0}
|
||||||
|
}
|
||||||
@@ -0,0 +1,93 @@
|
|||||||
|
\chapter{System overview}
|
||||||
|
\label{ch:overview}
|
||||||
|
|
||||||
|
\section{Project goal}
|
||||||
|
FPGA-Neural implements a \textbf{reusable Neural Network Engine in FPGA hardware}.
|
||||||
|
The whole is made of three elements: the FPGA, which is the actual accelerator; a
|
||||||
|
dedicated RAM physically associated with the FPGA and not shared with the host; and a
|
||||||
|
host interface independent of the operating system, initially SPI (with possible
|
||||||
|
future extension to Dual~SPI).
|
||||||
|
|
||||||
|
The founding principle is the separation between who \emph{executes} the computation
|
||||||
|
and who \emph{uses} it: the neural network computation happens entirely inside the
|
||||||
|
FPGA, while the host system only provides configuration, network parameters, input
|
||||||
|
data, control and result readback. The host is not part of the computational datapath.
|
||||||
|
Possible host systems include Linux SoCs, Raspberry~Pi-like systems, ESP32,
|
||||||
|
microcontrollers and development PCs: the same engine architecture must be usable in
|
||||||
|
completely different systems.
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\footnotesize,node distance=8mm]
|
||||||
|
\node[fnblockD,minimum width=42mm,minimum height=20mm] (host){\textbf{HOST}\\[2pt]
|
||||||
|
{\scriptsize Configuration}\\{\scriptsize Training}\\{\scriptsize Control}};
|
||||||
|
\node[fnblockT,below=14mm of host,minimum width=42mm,minimum height=20mm] (fpga)
|
||||||
|
{\textbf{FPGA}\\[2pt]{\scriptsize Neural Network Engine}\\{\scriptsize Compute / Control}};
|
||||||
|
\node[fnblock,below=14mm of fpga,minimum width=42mm,minimum height=13mm] (ram)
|
||||||
|
{\textbf{Dedicated RAM}\\{\scriptsize weights / bias / buffers}};
|
||||||
|
\draw[fnbus] (host) -- node[fnlbl,right]{SPI / Dual SPI} (fpga);
|
||||||
|
\draw[fnbus] (fpga) -- node[fnlbl,right]{parallel bus} (ram);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
\section{Hardware configuration versus network configuration}
|
||||||
|
The project draws a precise distinction between the accelerator's \textbf{hardware
|
||||||
|
architecture} and the \textbf{neural network parameters}.
|
||||||
|
|
||||||
|
The physical architecture of the engine is defined at FPGA synthesis and
|
||||||
|
implementation time. Typical hardware parameters are \code{N\_INPUTS},
|
||||||
|
\code{N\_NEURONS}, \code{N\_LAYERS}, \code{PARALLEL}, \code{DATA\_WIDTH},
|
||||||
|
\code{ACC\_WIDTH}: they are Verilog parameters resolved at synthesis and they
|
||||||
|
determine the datapath contained in the bitstream. The network parameters --- weights,
|
||||||
|
bias, activation and quantization parameters, specific constants --- are instead loaded
|
||||||
|
at runtime through the host interface and stored in the RAM associated with the FPGA.
|
||||||
|
|
||||||
|
\begin{fnnote}[Central architectural principle]
|
||||||
|
A build fixes the \emph{ceiling} of the machine (maximum number of layers, maximum
|
||||||
|
width, \code{PARALLEL}); the host configures the \emph{actual} network --- number of
|
||||||
|
layers, per-layer input/output width, per-layer activation and trained parameters ---
|
||||||
|
entirely at runtime, over SPI, into the FPGA's local memory. A single bitstream serves
|
||||||
|
any topology up to that ceiling.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\section{Boot and initialization}
|
||||||
|
The FPGA is configured at power-on through the usual configuration mechanism (bitstream
|
||||||
|
loading from SPI flash). The bitstream defines the hardware architecture of the engine;
|
||||||
|
the host does not dynamically build the datapath during normal operation, but rather
|
||||||
|
configures the network data on which the already existing datapath operates.
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=4.5mm,start chain=going below,
|
||||||
|
every node/.style={on chain}]
|
||||||
|
\node[fnblockA,minimum width=60mm](p){Power-on};
|
||||||
|
\node[fnblock,minimum width=60mm]{FPGA configuration (bitstream from flash)};
|
||||||
|
\node[fnblockT,minimum width=60mm]{Neural Network Engine available};
|
||||||
|
\node[fnblock,minimum width=60mm]{Host initialization (SPI)};
|
||||||
|
\node[fnblock,minimum width=60mm]{Loading network parameters / weights / bias};
|
||||||
|
\node[fnblockD,minimum width=60mm]{Engine ready};
|
||||||
|
\begin{scope}[every path/.style={fnarrow}]
|
||||||
|
\foreach \a/\b in {1/2,2/3,3/4,4/5,5/6}{}
|
||||||
|
\end{scope}
|
||||||
|
\foreach \i [count=\j from 2] in {1,...,5}{
|
||||||
|
\draw[fnarrow] (chain-\i) -- (chain-\j);}
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
\section{Training and inference}
|
||||||
|
Training and inference are conceptually separate. The first implementation does not
|
||||||
|
require the FPGA to perform training: weights can be computed externally
|
||||||
|
(PC/Linux/other host) and transferred over SPI into the FPGA's RAM, which then performs
|
||||||
|
inference. This drastically reduces the complexity of the initial hardware, without
|
||||||
|
precluding a future implementation of assisted or fully hardware training (roadmap
|
||||||
|
Phase~8, ch.~\ref{ch:roadmap}). During inference the host only provides the input data
|
||||||
|
and retrieves the result, obtaining deterministic computation, reduced host load,
|
||||||
|
hardware parallelism, predictable latency and independence from the host CPU
|
||||||
|
architecture.
|
||||||
|
|
||||||
|
\section{Design philosophy and reuse}
|
||||||
|
The project should be understood as a \emph{reusable FPGA neural acceleration platform}
|
||||||
|
rather than a single network. The application determines input size, topology, number
|
||||||
|
of layers and neurons, parallelism, numeric precision, activation functions, memory and
|
||||||
|
performance requirements; the hardware generation process produces the corresponding
|
||||||
|
FPGA implementation. The same HDL architecture remains conceptually unchanged while the
|
||||||
|
synthesis parameters generate implementations appropriate to the different application
|
||||||
|
targets.
|
||||||
@@ -0,0 +1,54 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\gdef \LT@iii {\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{363.57677pt}}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {2}RTL architecture}{3}{chapter.2}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:arch}{{2}{3}{RTL architecture}{chapter.2}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {2.1}Hierarchical organization}{3}{section.2.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {2.2}Role of each module}{3}{section.2.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {2.3}Two execution paths}{4}{section.2.3}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/02-architettura}{
|
||||||
|
\setcounter{page}{6}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{2}
|
||||||
|
\setcounter{section}{3}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{1}
|
||||||
|
\setcounter{LT@tables}{3}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{5}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{1}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{0}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{0}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{10}
|
||||||
|
\setcounter{lstlisting}{0}
|
||||||
|
}
|
||||||
@@ -0,0 +1,80 @@
|
|||||||
|
\chapter[RTL architecture]{RTL architecture and module hierarchy}
|
||||||
|
\label{ch:arch}
|
||||||
|
|
||||||
|
\section{Hierarchical organization}
|
||||||
|
The design is organized in layers, from the elementary multiply-accumulator up to the
|
||||||
|
integrated top-level with SPI interface and PSRAM. Each layer encapsulates the previous
|
||||||
|
one and abstracts away its details: the validated datapath (\code{mac\_unit},
|
||||||
|
\code{mac8}, \code{neuron\_parallel}) is never modified by the higher orchestration
|
||||||
|
layers.
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\footnotesize,every node/.style={fnblock,minimum width=40mm},
|
||||||
|
level distance=13mm,sibling distance=0mm]
|
||||||
|
\node[fnblockD,minimum width=62mm](top){\code{spi\_neuron\_top} \\ {\scriptsize integrated top-level}};
|
||||||
|
\node[fnblockT,minimum width=62mm,below=8mm of top](arb){\code{mem\_arbiter} \;/\; \code{layer\_sequencer} \\ {\scriptsize 3-port arbitration + layer sequencing}};
|
||||||
|
\node[fnblock,minimum width=62mm,below=8mm of arb](nm){\code{neuron\_memory} \\ {\scriptsize memory $\leftrightarrow$ neuron bridge, neuron loop}};
|
||||||
|
\node[fnblock,minimum width=62mm,below=8mm of nm](np){\code{neuron\_parallel} \\ {\scriptsize neuron FSM: groups, bias, activation, saturation}};
|
||||||
|
\node[fnblockT,minimum width=62mm,below=8mm of np](m8){\code{mac8} \\ {\scriptsize \code{PARALLEL} MACs + balanced adder tree}};
|
||||||
|
\node[fnblock,minimum width=62mm,below=8mm of m8](mu){\code{mac\_unit} \\ {\scriptsize $x\cdot w$ + sign extension + accumulate}};
|
||||||
|
\foreach \a/\b in {top/arb,arb/nm,nm/np,np/m8,m8/mu}
|
||||||
|
\draw[fnarrow] (\a) -- (\b);
|
||||||
|
|
||||||
|
% memory branches on the right
|
||||||
|
\node[fnblockA,minimum width=34mm,right=14mm of nm](ma){\code{int8\_memory\_access}\\{\scriptsize byte $\leftrightarrow$ 16-bit word}};
|
||||||
|
\node[fnblockA,minimum width=34mm,below=6mm of ma](mi){\code{memory\_interface}\\{\scriptsize req/ready handshake}};
|
||||||
|
\node[fnblockA,minimum width=34mm,below=6mm of mi](pc){\code{psram\_controller}\\{\scriptsize physical PSRAM bus}};
|
||||||
|
\draw[fnarrowT] (ma)--(mi); \draw[fnarrowT] (mi)--(pc);
|
||||||
|
\draw[fnarrowT,dashed] (nm.east) -- (ma.west);
|
||||||
|
|
||||||
|
% SPI branches on the left
|
||||||
|
\node[fnblockA,minimum width=30mm,left=14mm of arb,yshift=6mm](ss){\code{spi\_slave}\\{\scriptsize Mode 0 physical layer}};
|
||||||
|
\node[fnblockA,minimum width=30mm,below=6mm of ss](se){\code{spi\_engine}\\{\scriptsize opcode FSM + registers}};
|
||||||
|
\draw[fnarrowT] (ss)--(se);
|
||||||
|
\draw[fnarrowT,dashed] (se.east) -- (arb.west);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
\section{Role of each module}
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.4cm}Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Module} & \thd{Function} \\
|
||||||
|
\midrule
|
||||||
|
\code{mac\_unit} & Single multiply-accumulate: $\mathrm{acc\_out}=\mathrm{acc\_in}+(x\cdot w)$, with sign extension of the product to \code{ACC\_WIDTH}. Parametric on \code{DATA\_WIDTH}/\code{ACC\_WIDTH}. \\
|
||||||
|
\rowa \code{mac8} & \code{PARALLEL} instances of \code{mac\_unit} whose products are summed by a \emph{balanced binary adder tree} of depth $\log_2(\text{PARALLEL})$; the result is added to the input accumulator. \\
|
||||||
|
\code{neuron\_parallel} & FSM of a single neuron: processes \code{N\_INPUTS} inputs in groups of \code{PARALLEL}, accumulates across groups, adds the bias, applies the activation and saturates to INT8. Includes the processing guard on \code{N\_INPUTS \% PARALLEL} and the runtime width \code{n\_inputs\_real}. \\
|
||||||
|
\rowa \code{layer} & Instantiates \code{N\_NEURONS} neurons \emph{in parallel} on the same input vector; \code{busy}=OR, \code{done}=AND of the neurons. A purely data-combinational path used in the datapath benchmarks. \\
|
||||||
|
\code{neuron\_memory} & Integrates computation with memory: reads $X$ (shared) once, then for each neuron re-reads $W$ and bias from RAM and reuses a single \code{neuron\_parallel} instance (memory-bound, one neuron at a time). Output \code{y\_bus} packed neuron-major. \\
|
||||||
|
\rowa \code{layer\_sequencer} & Chains up to \code{N\_LAYERS} executions of \code{neuron\_memory} by reading a descriptor table written by the host and alternating the ping-pong buffers in RAM (Phase~5). \\
|
||||||
|
\code{act\_buffer} & Global activation buffer in \code{DP16KD} block RAM, indexed by signal id (Type \#2). \\
|
||||||
|
\rowa \code{graph\_engine} & Graph-network engine (Type \#2): gather from \code{act\_buffer}, reuses \code{neuron\_parallel}, writes outputs by id (ch.~\ref{ch:grafo}). \\
|
||||||
|
\code{int8\_memory\_access} & Converts the byte/INT8 interface (byte address) into the 16-bit word interface, selecting the low/high byte via \code{lb\_n}/\code{ub\_n} and \code{addr>>1}. \\
|
||||||
|
\rowa \code{memory\_interface} & 2-state handshake FSM (IDLE/WAIT) that serializes the single transaction toward the controller. \\
|
||||||
|
\code{psram\_controller} & Asynchronous parallel PSRAM bus controller with read \textbf{page mode}: 70~ns random access (\code{tAA}), 20~ns same-page bursts (\code{tAPA}) with CE\#/OE\# held asserted; enables page mode on the chip at boot via the configuration register (ch.~\ref{ch:mem}, \S~5.5). Drives \code{ce\_n/oe\_n/we\_n/lb\_n/ub\_n/zz\_n} and the tri-state data bus. \\
|
||||||
|
\rowa \code{mem\_arbiter} & Fixed-priority arbiter (B$>$C$>$A) among three byte-level masters: \code{spi\_engine} (A), \code{neuron\_memory} (B), \code{layer\_sequencer} (C). \\
|
||||||
|
\code{spi\_slave} & SPI Mode 0 physical layer, MSB-first, 3-stage CDC synchronizer on SCLK/MOSI/CS\_N, shift register and CS framing. \\
|
||||||
|
\rowa \code{spi\_engine} & Protocol/opcode FSM and register bank (\code{x\_base}, \code{w\_base}, \code{bias\_addr}, ping-pong base, activation, runtime widths\ldots), with sticky/clear-on-read \code{STATUS.done}. \\
|
||||||
|
\code{spi\_neuron\_top} & Top-level: connects SPI, arbiter, sequencer, \code{neuron\_memory} and the PSRAM chain; multiplexes control of \code{neuron\_memory} between the sequencer and the direct single-layer path. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\vspace{6pt}
|
||||||
|
\begin{fnnote}[Simulation models]
|
||||||
|
\code{psram\_model.v} (in \code{sim/}) and \code{memory\_model.v} are behavioral memory
|
||||||
|
models used in the testbenches; they are not part of the synthesizable design but they
|
||||||
|
reproduce the real latency for end-to-end verification.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\section{Two execution paths}
|
||||||
|
The top-level exposes two mutually exclusive modes toward the same \code{neuron\_memory}
|
||||||
|
compute engine:
|
||||||
|
\begin{itemize}
|
||||||
|
\item \textbf{Single-layer / manual path}: the host sets the bases with
|
||||||
|
\op{SET\_BASE}, starts with \op{START} and reads with \op{READ\_OUTPUT}.
|
||||||
|
\code{spi\_engine} drives \code{neuron\_memory} directly.
|
||||||
|
\item \textbf{Multi-layer path}: the host writes the descriptor table and starts with
|
||||||
|
\op{RUN\_NETWORK}; \code{layer\_sequencer} takes over control of \code{neuron\_memory}
|
||||||
|
(while \code{seq\_busy} is high) and chains the layers.
|
||||||
|
\end{itemize}
|
||||||
|
The top-level multiplexer switches the control lines of \code{neuron\_memory} based on
|
||||||
|
\code{seq\_busy}, returning the engine to the direct path at the end of the sequence.
|
||||||
@@ -0,0 +1,63 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {3}Compute datapath}{6}{chapter.3}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:datapath}{{3}{6}{Compute datapath}{chapter.3}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {3.1}INT8/INT32 arithmetic chain}{6}{section.3.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {3.2}\texttt {mac\_unit} --- multiply-accumulator}{6}{section.3.2}\protected@file@percent }
|
||||||
|
\newlabel{lst:macunit}{{3.1}{6}{\texttt {rtl/mac\_unit.v} --- arithmetic core}{lstlisting.3.1}{}}
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {3.1}{\ignorespaces \texttt {rtl/mac\_unit.v} --- arithmetic core}}{6}{lstlisting.3.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {3.3}\texttt {mac8} --- parallel MAC and balanced adder tree}{6}{section.3.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {3.4}\texttt {neuron\_parallel} --- neuron FSM}{7}{section.3.4}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {3.4.1}Parameter guard (elaboration-time)}{7}{subsection.3.4.1}\protected@file@percent }
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {3.2}{\ignorespaces \texttt {rtl/neuron\_parallel.v} --- parameter guard}}{7}{lstlisting.3.2}\protected@file@percent }
|
||||||
|
\gdef \LT@iv {\LT@entry
|
||||||
|
{1}{85.97733pt}\LT@entry
|
||||||
|
{1}{51.83368pt}\LT@entry
|
||||||
|
{1}{334.50494pt}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {3.5}Activation functions}{8}{section.3.5}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {3.6}INT8 saturation}{8}{section.3.6}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {3.7}\texttt {layer} --- neurons in parallel}{8}{section.3.7}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/03-datapath}{
|
||||||
|
\setcounter{page}{9}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{3}
|
||||||
|
\setcounter{section}{7}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{1}
|
||||||
|
\setcounter{LT@tables}{4}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{9}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{7}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{0}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{0}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{19}
|
||||||
|
\setcounter{lstlisting}{2}
|
||||||
|
}
|
||||||
@@ -0,0 +1,166 @@
|
|||||||
|
\chapter{Compute datapath}
|
||||||
|
\label{ch:datapath}
|
||||||
|
|
||||||
|
\section{INT8/INT32 arithmetic chain}
|
||||||
|
The elementary datapath implements the typical sequence of a quantized neuron:
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=3mm,start chain=going right,
|
||||||
|
every node/.style={fnblock,minimum width=15mm,minimum height=8mm,on chain}]
|
||||||
|
\node[fnblockT]{INT8\\$\times$\,INT8};
|
||||||
|
\node{INT16\\product};
|
||||||
|
\node{sign-ext\\INT32};
|
||||||
|
\node[fnblockD]{accumulate\\INT32};
|
||||||
|
\node{$+$ bias};
|
||||||
|
\node[fnblockA]{activation};
|
||||||
|
\node[fnblockT]{sat. INT8};
|
||||||
|
\foreach \i [count=\j from 2] in {1,...,6}
|
||||||
|
\draw[fnarrow] (chain-\i) -- (chain-\j);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
Each INT8$\times$INT8 product fits in 16~bits; it is sign-extended to 32~bits before
|
||||||
|
accumulation, so the accumulator does not overflow on long vectors. Bias and activation
|
||||||
|
operate at 32~bits; only the final output is saturated to INT8.
|
||||||
|
|
||||||
|
\section{\texttt{mac\_unit} --- multiply-accumulator}
|
||||||
|
The \code{mac\_unit} module is purely combinational and parametric on \code{DATA\_WIDTH}
|
||||||
|
and \code{ACC\_WIDTH}. It computes:
|
||||||
|
\[
|
||||||
|
\mathrm{acc\_out} = \mathrm{acc\_in} + \mathrm{signext}_{ACC}(x \cdot w)
|
||||||
|
\]
|
||||||
|
The product has width $2\times$\code{DATA\_WIDTH} and is sign-extended by replicating
|
||||||
|
the most significant bit. On ECP5 the multiplication maps onto a \code{MULT18X18D} DSP
|
||||||
|
block.
|
||||||
|
|
||||||
|
\begin{lstlisting}[caption={\texttt{rtl/mac\_unit.v} --- arithmetic core},label={lst:macunit}]
|
||||||
|
localparam PROD_WIDTH = 2 * DATA_WIDTH;
|
||||||
|
wire signed [PROD_WIDTH-1:0] product = x * w;
|
||||||
|
wire signed [ACC_WIDTH-1:0] product_ext =
|
||||||
|
{{(ACC_WIDTH-PROD_WIDTH){product[PROD_WIDTH-1]}}, product};
|
||||||
|
assign acc_out = acc_in + product_ext;
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\section{\texttt{mac8} --- parallel MAC and balanced adder tree}
|
||||||
|
\code{mac8} instantiates \code{PARALLEL} \code{mac\_unit} units that generate
|
||||||
|
\code{PARALLEL} independent products, then sums them with a \emph{balanced binary
|
||||||
|
adder tree}. Compared to the linear reduction
|
||||||
|
$((((p_0{+}p_1){+}p_2){+}p_3){+}\dots)$, of depth $O(\text{PARALLEL})$, the tree has
|
||||||
|
depth $O(\log_2 \text{PARALLEL})$, drastically reducing the combinational path.
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,level distance=11mm,
|
||||||
|
every node/.style={fnreg,minimum width=8mm},
|
||||||
|
level 1/.style={sibling distance=30mm},
|
||||||
|
level 2/.style={sibling distance=15mm},
|
||||||
|
level 3/.style={sibling distance=8mm},
|
||||||
|
edge from parent/.style={fnarrowT,draw}]
|
||||||
|
\node[fnblockD]{sum}
|
||||||
|
child {node[fnblockT]{$+$}
|
||||||
|
child {node[fnblockT]{$+$}
|
||||||
|
child {node{$p_0$}} child {node{$p_1$}}}
|
||||||
|
child {node[fnblockT]{$+$}
|
||||||
|
child {node{$p_2$}} child {node{$p_3$}}}}
|
||||||
|
child {node[fnblockT]{$+$}
|
||||||
|
child {node[fnblockT]{$+$}
|
||||||
|
child {node{$p_4$}} child {node{$p_5$}}}
|
||||||
|
child {node[fnblockT]{$+$}
|
||||||
|
child {node{$p_6$}} child {node{$p_7$}}}};
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
\begin{center}\footnotesize\itshape\color{fnGrey}
|
||||||
|
Example with PARALLEL=8: 3 levels. PARALLEL=16 $\to$ 4 levels; PARALLEL=32 $\to$ 5
|
||||||
|
levels.\end{center}
|
||||||
|
|
||||||
|
\begin{fnnote}[PARALLEL as a power of two]
|
||||||
|
The tree is designed for \code{PARALLEL} as a power of two (8, 16, 32\ldots). This is
|
||||||
|
also the value used in all project configurations.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\section{\texttt{neuron\_parallel} --- neuron FSM}
|
||||||
|
\code{neuron\_parallel} processes \code{N\_INPUTS} inputs in groups of \code{PARALLEL},
|
||||||
|
maintaining the accumulator from one group to the next. At the end it adds the bias,
|
||||||
|
applies the activation and saturates to INT8. The number of groups is
|
||||||
|
$\text{GROUPS}=\text{N\_INPUTS}/\text{PARALLEL}$.
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=4mm,start chain=going below,
|
||||||
|
every node/.style={on chain,fnblock,minimum width=46mm}]
|
||||||
|
\node[fnblockA]{\code{start}};
|
||||||
|
\node{group 0 $\to$ accumulate};
|
||||||
|
\node{group 1 $\to$ accumulate};
|
||||||
|
\node[draw=none,fill=none]{\vdots};
|
||||||
|
\node{group GROUPS$-$1 $\to$ accumulate};
|
||||||
|
\node{$+$ bias};
|
||||||
|
\node[fnblockA]{activation (ACT\_RELU / ACT\_NONE)};
|
||||||
|
\node[fnblockT]{INT8 saturation};
|
||||||
|
\node[fnblockD]{\code{done}, \code{y}};
|
||||||
|
\foreach \i [count=\j from 2] in {1,...,8}
|
||||||
|
\draw[fnarrow] (chain-\i) -- (chain-\j);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
\subsection{Parameter guard (elaboration-time)}
|
||||||
|
If \code{PARALLEL} does not exactly divide \code{N\_INPUTS} two failures occur, both
|
||||||
|
confirmed empirically in \code{sim/parameter\_sweep\_tb.v}:
|
||||||
|
\begin{itemize}
|
||||||
|
\item integer division truncates \code{GROUPS} and the excess inputs are never read
|
||||||
|
$\to$ \textbf{wrong} result, with no error and no warning;
|
||||||
|
\item if \code{PARALLEL > N\_INPUTS}, \code{GROUPS=0} and the terminal condition is
|
||||||
|
never satisfied $\to$ the neuron \textbf{hangs} (busy high, done never asserted).
|
||||||
|
\end{itemize}
|
||||||
|
The solution does not modify the validated datapath: a \code{generate} block
|
||||||
|
instantiates a deliberately undefined module when
|
||||||
|
$\text{N\_INPUTS} \bmod \text{PARALLEL}\neq0$, forcing an error at \emph{elaboration}
|
||||||
|
both in simulation and in synthesis. For valid configurations the branch is never
|
||||||
|
elaborated.
|
||||||
|
|
||||||
|
\begin{lstlisting}[caption={\texttt{rtl/neuron\_parallel.v} --- parameter guard}]
|
||||||
|
generate
|
||||||
|
if (N_INPUTS == 0 || N_INPUTS % PARALLEL != 0) begin : PARAMETER_ERROR
|
||||||
|
neuron_parallel_requires_N_INPUTS_multiple_of_PARALLEL
|
||||||
|
invalid_parameter_combination();
|
||||||
|
end
|
||||||
|
endgenerate
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\begin{fnnote}[Edge case \texttt{N\_INPUTS=0} (fixed 2026-09-04)]
|
||||||
|
The original condition (\code{N\_INPUTS \% PARALLEL != 0}) does not catch
|
||||||
|
\code{N\_INPUTS=0}, since $0 \bmod \text{PARALLEL}=0$ for any \code{PARALLEL}: the module
|
||||||
|
elaborated successfully (both in simulation and in real Yosys synthesis) while leaving
|
||||||
|
\code{x\_bus}/\code{w\_bus} undriven and \code{start} silently ineffective. Found during
|
||||||
|
the re-certification campaign (\code{docs/validation/bugs.md}, BUG-002) and fixed by
|
||||||
|
extending the guard as above --- \code{N\_INPUTS=0} now fails elaboration exactly like the
|
||||||
|
other degenerate cases.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\section{Activation functions}
|
||||||
|
\code{neuron\_parallel} accepts a 2-bit \code{activation} port. The default is
|
||||||
|
\code{ACT\_RELU}, the only behavior that existed before the port was introduced, so
|
||||||
|
every pre-existing caller remains unchanged.
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{2.6cm} C{1.4cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Encoding} & \thd{Value} & \thd{Behavior} \\
|
||||||
|
\midrule
|
||||||
|
\code{ACT\_NONE} & \code{2'd0} & Linear: no clamp to zero, bilateral saturation to the INT8 range $[-128,+127]$. \\
|
||||||
|
\rowa \code{ACT\_RELU} & \code{2'd1} & $\max(0,x)$, then positive saturation to $+127$ (default; also the fallback for reserved encodings). \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\section{INT8 saturation}
|
||||||
|
After bias and activation, the 32-bit accumulator is reduced to INT8:
|
||||||
|
\[
|
||||||
|
y=\begin{cases}
|
||||||
|
+127 & \text{if } \mathrm{final\_acc} > 127\\
|
||||||
|
-128 & \text{if } \mathrm{final\_acc} < -128 \ \text{(ACT\_NONE only)}\\
|
||||||
|
0 & \text{if } \mathrm{final\_acc}\le 0 \ \text{(ACT\_RELU only)}\\
|
||||||
|
\mathrm{final\_acc}[7:0] & \text{otherwise}
|
||||||
|
\end{cases}
|
||||||
|
\]
|
||||||
|
|
||||||
|
\section{\texttt{layer} --- neurons in parallel}
|
||||||
|
\code{layer} instantiates \code{N\_NEURONS} neurons that share the input vector
|
||||||
|
\code{x\_bus} but have distinct weights and bias; \code{busy} is the OR and \code{done}
|
||||||
|
the AND of the neurons' signals. It is the module used in the datapath benchmarks
|
||||||
|
(ch.~\ref{ch:impl}), where all neurons work simultaneously. The addressing convention
|
||||||
|
is neuron-major: the weights of neuron $n$ occupy
|
||||||
|
\code{weights\_bus[n*N\_INPUTS*DATA\_WIDTH +: N\_INPUTS*DATA\_WIDTH]}.
|
||||||
@@ -0,0 +1,66 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\gdef \LT@v {\LT@entry
|
||||||
|
{1}{97.35826pt}\LT@entry
|
||||||
|
{1}{63.21504pt}\LT@entry
|
||||||
|
{1}{311.74265pt}}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {4}Parameters and configurability}{9}{chapter.4}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:param}{{4}{9}{Parameters and configurability}{chapter.4}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {4.1}Build parameters (synthesis-time)}{9}{section.4.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {4.2}Runtime network width}{9}{section.4.2}\protected@file@percent }
|
||||||
|
\gdef \LT@vi {\LT@entry
|
||||||
|
{1}{168.49014pt}\LT@entry
|
||||||
|
{1}{97.35826pt}\LT@entry
|
||||||
|
{1}{206.46754pt}}
|
||||||
|
\gdef \LT@vii {\LT@entry
|
||||||
|
{1}{68.9055pt}\LT@entry
|
||||||
|
{1}{68.9055pt}\LT@entry
|
||||||
|
{1}{68.9055pt}\LT@entry
|
||||||
|
{1}{265.59944pt}}
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {4.2.1}Measured savings}{10}{subsection.4.2.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {4.3}Characterized configurations}{10}{section.4.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {4.4}Build versus runtime summary}{10}{section.4.4}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/04-parametri}{
|
||||||
|
\setcounter{page}{11}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{4}
|
||||||
|
\setcounter{section}{4}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{3}
|
||||||
|
\setcounter{LT@tables}{7}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{13}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{7}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{0}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{0}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{25}
|
||||||
|
\setcounter{lstlisting}{0}
|
||||||
|
}
|
||||||
@@ -0,0 +1,88 @@
|
|||||||
|
\chapter{Parameters and configurability}
|
||||||
|
\label{ch:param}
|
||||||
|
|
||||||
|
\section{Build parameters (synthesis-time)}
|
||||||
|
The hardware architecture is fixed at synthesis through the following Verilog
|
||||||
|
parameters. They determine the datapath contained in the bitstream and its capacity
|
||||||
|
\emph{ceiling}.
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.0cm} C{1.8cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Parameter} & \thd{Default} & \thd{Meaning} \\
|
||||||
|
\midrule
|
||||||
|
\code{DATA\_WIDTH} & 8 & Data width (INT8). \\
|
||||||
|
\rowa \code{ACC\_WIDTH} & 32 & Accumulator width (INT32). \\
|
||||||
|
\code{N\_INPUTS} & 32 / 256 & Maximum number of inputs per neuron (benchmark baseline: 256). \\
|
||||||
|
\rowa \code{N\_NEURONS} & 1 / 4 & Maximum number of neurons per layer. \\
|
||||||
|
\code{PARALLEL} & 8 & Simultaneous hardware MACs per neuron; must divide \code{N\_INPUTS} and should be a power of two. \\
|
||||||
|
\rowa \code{N\_LAYERS} & 4 & Maximum number of layers chainable by \code{layer\_sequencer}. \\
|
||||||
|
\code{ADDR\_WIDTH} & 23 & Byte-address width (8~MB). \\
|
||||||
|
\rowa \code{MEM\_DATA\_WIDTH} & 16 & Width of the physical PSRAM data bus. \\
|
||||||
|
\code{CLK\_FREQ\_MHZ} & 80 & Frequency used in the PSRAM timing formulas (must be aligned to the real oscillator). \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\begin{fnwarn}[\texttt{N\_INPUTS} \% \texttt{PARALLEL} constraint]
|
||||||
|
\code{PARALLEL} must divide \code{N\_INPUTS} exactly, otherwise the elaboration guard
|
||||||
|
fires (§\ref{ch:datapath}). The same constraint applies at runtime to
|
||||||
|
\code{n\_inputs\_real}.
|
||||||
|
\end{fnwarn}
|
||||||
|
|
||||||
|
\section{Runtime network width}
|
||||||
|
A single bitstream serves any topology \emph{up to} the build maximum. The actual width
|
||||||
|
of each execution is a separate value, set by the host:
|
||||||
|
\begin{itemize}
|
||||||
|
\item \code{n\_inputs\_real} --- inputs actually used in this execution (must be a
|
||||||
|
multiple of \code{PARALLEL});
|
||||||
|
\item \code{n\_neurons\_real} --- neurons actually computed in this execution.
|
||||||
|
\end{itemize}
|
||||||
|
Both default to the build maximum, so any caller that leaves them unconnected processes
|
||||||
|
the full width as before the ports were introduced.
|
||||||
|
|
||||||
|
\begin{fnnote}[Real early termination]
|
||||||
|
This is not mere address bookkeeping: the two values directly bound the hardware loops
|
||||||
|
(X/W reads of \code{neuron\_memory}, MAC group count of \code{neuron\_parallel} and the
|
||||||
|
length of the ping-pong copy for \code{RUN\_NETWORK}). A narrower layer actually
|
||||||
|
\emph{computes} and \emph{copies} faster and does not require zero-padding of the RAM
|
||||||
|
for the unused tail: data beyond \code{n\_inputs\_real}/\code{n\_neurons\_real} is never
|
||||||
|
read.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
This lets a network taper within a single chained execution, for example
|
||||||
|
$256\to64\to16\to4$, with each layer declaring its own actual width in the descriptor
|
||||||
|
table (ch.~\ref{ch:seq}).
|
||||||
|
|
||||||
|
\subsection{Measured savings}
|
||||||
|
Early termination was measured end-to-end:
|
||||||
|
\begin{tabularx}{\textwidth}{L{5.5cm} C{3.0cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Test} & \thd{Cycles} & \thd{Comparison} \\
|
||||||
|
\midrule
|
||||||
|
\code{neuron\_parallel\_tb.v} (T7) & 3 vs 6 & reduced vs full, with ``garbage'' data in the skipped lanes (proof that they are not read). \\
|
||||||
|
\rowa \code{neuron\_memory\_tb.v} (T5) & 209 vs 788 & 8-of-32 vs full 32, through the real PSRAM stack. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\section{Characterized configurations}
|
||||||
|
Some combinations validated in simulation and/or synthesis:
|
||||||
|
\begin{tabularx}{\textwidth}{C{2.0cm} C{2.0cm} C{2.0cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{N\_INPUTS} & \thd{N\_NEURONS} & \thd{PARALLEL} & \thd{Notes} \\
|
||||||
|
\midrule
|
||||||
|
32 & 4 & 8 & First functional parametric test (Phase~1). \\
|
||||||
|
\rowa 256 & 4 & 2/4/8/16 & Datapath benchmark sweep (Phase~7). \\
|
||||||
|
32 & 1..3 & 8 & Single/multi-neuron memory integration (Phase~3). \\
|
||||||
|
\rowa 4 & 4 & 2 & End-to-end 2-layer \code{RUN\_NETWORK} test over real SPI. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\section{Build versus runtime summary}
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\footnotesize,node distance=6mm]
|
||||||
|
\node[fnblockD,minimum width=54mm,minimum height=15mm](b){\textbf{BUILD (synthesis)}\\[2pt]
|
||||||
|
{\scriptsize N\_INPUTS, N\_NEURONS, N\_LAYERS,}\\{\scriptsize PARALLEL, DATA\_WIDTH, ACC\_WIDTH}\\{\scriptsize $\Rightarrow$ machine ceiling}};
|
||||||
|
\node[fnblockT,right=16mm of b,minimum width=54mm,minimum height=15mm](r){\textbf{RUNTIME (host, SPI)}\\[2pt]
|
||||||
|
{\scriptsize n\_inputs\_real, n\_neurons\_real,}\\{\scriptsize activation, num\_layers, weights/bias}\\{\scriptsize $\Rightarrow$ actual network}};
|
||||||
|
\draw[fnbus] (b) -- node[fnlbl,above]{$\le$} (r);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
@@ -0,0 +1,68 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {5}Memory subsystem}{11}{chapter.5}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:mem}{{5}{11}{Memory subsystem}{chapter.5}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {5.1}Memory chain}{11}{section.5.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {5.2}\texttt {int8\_memory\_access} --- byte/word conversion}{11}{section.5.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {5.3}\texttt {memory\_interface} --- handshake}{11}{section.5.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {5.4}\texttt {psram\_controller} --- physical bus}{11}{section.5.4}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {5.4.1}Timing}{12}{subsection.5.4.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {5.5}Read page mode}{12}{section.5.5}\protected@file@percent }
|
||||||
|
\newlabel{sec:pagemode}{{5.5}{12}{Read page mode}{section.5.5}{}}
|
||||||
|
\gdef \LT@viii {\LT@entry
|
||||||
|
{1}{103.04872pt}\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{260.52805pt}}
|
||||||
|
\gdef \LT@ix {\LT@entry
|
||||||
|
{1}{159.95424pt}\LT@entry
|
||||||
|
{1}{104.12057pt}\LT@entry
|
||||||
|
{1}{104.12057pt}\LT@entry
|
||||||
|
{1}{104.12057pt}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {5.6}Address map and conventions}{13}{section.5.6}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {5.6.1}PSRAM physical addressing}{13}{subsection.5.6.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {5.7}Bandwidth}{13}{section.5.7}\protected@file@percent }
|
||||||
|
\newlabel{sec:bandwidth}{{5.7}{13}{Bandwidth}{section.5.7}{}}
|
||||||
|
\@setckpt{chapters/05-memoria}{
|
||||||
|
\setcounter{page}{15}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{5}
|
||||||
|
\setcounter{section}{7}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{2}
|
||||||
|
\setcounter{LT@tables}{9}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{19}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{7}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{0}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{0}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{35}
|
||||||
|
\setcounter{lstlisting}{0}
|
||||||
|
}
|
||||||
@@ -0,0 +1,187 @@
|
|||||||
|
\chapter{Memory subsystem}
|
||||||
|
\label{ch:mem}
|
||||||
|
|
||||||
|
\section{Memory chain}
|
||||||
|
The compute engine works with addresses and data at the \emph{byte} level (INT8), while
|
||||||
|
the PSRAM is a 16-bit word device. Three cascaded modules realize the conversion and
|
||||||
|
the physical access:
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=8mm]
|
||||||
|
\node[fnblockD,minimum width=30mm,minimum height=12mm](nm){byte-level master\\{\scriptsize \code{neuron\_memory} / \code{spi\_engine} / \code{layer\_sequencer}}};
|
||||||
|
\node[fnblockT,right=10mm of nm,minimum width=28mm,minimum height=12mm](ia){\code{int8\_memory\_access}\\{\scriptsize byte $\leftrightarrow$ 16-bit word}};
|
||||||
|
\node[fnblock,right=10mm of ia,minimum width=26mm,minimum height=12mm](mi){\code{memory\_interface}\\{\scriptsize IDLE/WAIT FSM}};
|
||||||
|
\node[fnblockA,below=9mm of mi,minimum width=26mm,minimum height=12mm](pc){\code{psram\_controller}\\{\scriptsize async 70\,ns physical bus}};
|
||||||
|
\node[fnblock,left=10mm of pc,minimum width=26mm,minimum height=12mm](ps){PSRAM\\{\scriptsize 8\,MB 4M$\times$16}};
|
||||||
|
\draw[fnbus] (nm)--node[fnlbl,above]{req/wr/addr}(ia);
|
||||||
|
\draw[fnbus] (ia)--node[fnlbl,above]{16-bit}(mi);
|
||||||
|
\draw[fnbus] (mi)--(pc);
|
||||||
|
\draw[fnbus] (pc)--node[fnlbl,above]{DQ/A/ctrl}(ps);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
\section{\texttt{int8\_memory\_access} --- byte/word conversion}
|
||||||
|
Converts the INT8 interface (byte address) into the word interface. The byte address is
|
||||||
|
divided by two (\code{addr>>1}) to obtain the word address; the least significant bit
|
||||||
|
selects the byte:
|
||||||
|
\begin{itemize}
|
||||||
|
\item \code{addr[0]=0} $\to$ low byte: \code{lb\_n=0}, \code{ub\_n=1}, data on DQ[7:0];
|
||||||
|
\item \code{addr[0]=1} $\to$ high byte: \code{lb\_n=1}, \code{ub\_n=0}, data on DQ[15:8].
|
||||||
|
\end{itemize}
|
||||||
|
On read it extracts the correct byte from \code{mem\_rdata}. The FSM has two states
|
||||||
|
(IDLE, WAIT) and returns \code{ready} as a one-cycle pulse.
|
||||||
|
|
||||||
|
\section{\texttt{memory\_interface} --- handshake}
|
||||||
|
Two-state FSM that serializes a single transaction: in IDLE, on the \code{req} request,
|
||||||
|
it latches \code{wr/addr/wdata/lb\_n/ub\_n} and emits a one-cycle \code{mem\_req} pulse
|
||||||
|
toward the controller; in WAIT it waits for \code{mem\_ready}, captures \code{rdata} on
|
||||||
|
read and asserts \code{ready}. It guarantees the ``one transaction at a time'' contract.
|
||||||
|
|
||||||
|
\section{\texttt{psram\_controller} --- physical bus}
|
||||||
|
Asynchronous parallel PSRAM bus controller, with support for the chip's read
|
||||||
|
\textbf{page mode} (\S~\ref{sec:pagemode}). The main state machine is:
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize]
|
||||||
|
\node[fnstate](init) at (0,0){INIT};
|
||||||
|
\node[fnstate](idle) at (3.2,0){IDLE};
|
||||||
|
\node[fnstate](read) at (7,2.7){READ};
|
||||||
|
\node[fnstate](popen) at (11,2.7){PAGE\\OPEN};
|
||||||
|
\node[fnstate](write) at (7,-2.7){WRITE};
|
||||||
|
\node[fnstate](ww) at (11,-2.7){WRITE\\WAIT};
|
||||||
|
\draw[fnarrow] (init)--node[fnlbl,above]{INIT\_CYCLES + CR load}(idle);
|
||||||
|
\draw[fnarrow] (idle)--node[fnlbl,above,sloped]{req \& !wr}(read);
|
||||||
|
\draw[fnarrow] (idle)--node[fnlbl,below,sloped]{req \& wr}(write);
|
||||||
|
\draw[fnarrow] (read)--node[fnlbl,above]{ready}(popen);
|
||||||
|
\draw[fnarrowT] (popen) to[bend left=25] node[fnlbl,below]{req \& !wr}(read);
|
||||||
|
\draw[fnarrow] (popen) to[bend right=20] node[fnlbl,above,sloped]{req \& wr}(write);
|
||||||
|
\draw[fnarrow] (popen) to[out=-100,in=15,looseness=1.15] node[fnlbl,pos=0.55]{tCEM timeout}(idle);
|
||||||
|
\draw[fnarrow] (write)--node[fnlbl,above]{ACCESS\_CYCLES}(ww);
|
||||||
|
\draw[fnarrow] (ww) to[out=160,in=-70] node[fnlbl,pos=0.5,left]{ready}(idle);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
From INIT the controller automatically goes through a configuration-register load
|
||||||
|
sub-sequence (\code{STATE\_CR\_INIT}, 4 steps) before reaching IDLE for the first
|
||||||
|
time --- see \S~\ref{sec:pagemode}. The PAGE~OPEN~$\to$~WRITE transition
|
||||||
|
(bottom-right arrow) internally passes through two transit micro-states,
|
||||||
|
\code{STATE\_PAGE\_CLOSE} and \code{STATE\_PAGE\_REOPEN} (one cycle each): the
|
||||||
|
first forces CE\#/OE\# high for at least one cycle before the controller starts
|
||||||
|
driving the data bus, avoiding contention with the PSRAM's still-active output
|
||||||
|
($\geq t_{HZ}$); the second restarts the already-latched transaction exactly as
|
||||||
|
IDLE would. They are not drawn as separate nodes to keep the figure readable.
|
||||||
|
|
||||||
|
\subsection{Timing}
|
||||||
|
\begin{fnspec}[Timing formulas]
|
||||||
|
$\text{ACCESS\_CYCLES}=\lceil (70\times \text{CLK\_FREQ\_MHZ})/1000\rceil$ \quad
|
||||||
|
(random-access latency, $t_{AA}$/$t_{RC}$ = 70~ns)\\[3pt]
|
||||||
|
$\text{PAGE\_CYCLES}=\lceil (20\times \text{CLK\_FREQ\_MHZ})/1000\rceil$ \quad
|
||||||
|
(same-page continuation, $t_{APA}$/$t_{PC}$ = 20~ns)\\[3pt]
|
||||||
|
$\text{INIT\_CYCLES}=150\times \text{CLK\_FREQ\_MHZ}$ \quad
|
||||||
|
(power-up initialization, $t_{PU}$ = 150~\textmu s)\\[3pt]
|
||||||
|
$\text{PAGE\_TIMEOUT\_CYCLES}=\lceil (6000\times \text{CLK\_FREQ\_MHZ})/1000\rceil$ \quad
|
||||||
|
(automatic page close, safety margin under $t_{CEM}$ = 8~\textmu s)
|
||||||
|
\end{fnspec}
|
||||||
|
The data bus is tri-state driven: \code{psram\_dq = dq\_oe ? dq\_out : Z}. On read
|
||||||
|
\code{dq\_oe=0}; on write \code{dq\_oe=1} during the \code{we\_n} pulse. A WRITE\_WAIT
|
||||||
|
state keeps \code{ce\_n/lb\_n/ub\_n} active for the final hold before release.
|
||||||
|
|
||||||
|
\begin{fnwarn}[This is not QSPI]
|
||||||
|
This is a classic asynchronous-SRAM interface, \textbf{not} QSPI: most commercial
|
||||||
|
serial/QSPI ``PSRAM'' parts are not compatible with this controller without a rewrite.
|
||||||
|
See ch.~\ref{ch:hw} for the recommended part (parallel ISSI).
|
||||||
|
\end{fnwarn}
|
||||||
|
|
||||||
|
\section{Read page mode}
|
||||||
|
\label{sec:pagemode}
|
||||||
|
The recommended chip (ch.~\ref{ch:hw}) is ``asynchronous/\textbf{page mode}'': once
|
||||||
|
an initial random access at $t_{AA}$~=~70~ns has been done, further reads inside the
|
||||||
|
same 16-word page (address bits above \code{A[3]} unchanged) only cost
|
||||||
|
$t_{APA}$/$t_{PC}$~=~20~ns, because CE\#/OE\# stay asserted and only the address bus
|
||||||
|
changes. Page mode is \textbf{disabled by default} at power-up (bit~7 of the
|
||||||
|
configuration register, CR~=~\texttt{0x0070} by default) and must be explicitly
|
||||||
|
enabled.
|
||||||
|
|
||||||
|
\begin{itemize}
|
||||||
|
\item \textbf{Enable at boot}: right after INIT, the controller runs the
|
||||||
|
datasheet's ``software-access sequence'' (2 dummy reads + 2 writes, \texttt{0x0000}
|
||||||
|
unlock then real CR \texttt{0x00F0} = default with the Page bit set) at the
|
||||||
|
chip's highest address --- it reuses exactly the same READ/WRITE logic as every
|
||||||
|
other transaction, so it goes through the same timing checks.
|
||||||
|
\item \textbf{Page bursts}: after a READ the controller no longer closes CE\#/OE\#
|
||||||
|
(PAGE~OPEN state). A following read in the same page only waits PAGE\_CYCLES; a
|
||||||
|
read crossing into a different page still avoids a CE\# toggle but pays a full
|
||||||
|
ACCESS\_CYCLES for that one word (any change at \code{A[4]} or above requires a
|
||||||
|
new $t_{AA}$). A counter closes the page before the $t_{CEM}$ limit with a
|
||||||
|
safety margin.
|
||||||
|
\item \textbf{Only a WRITE closes the page.} Changes to \code{lb\_n}/\code{ub\_n}
|
||||||
|
do \emph{not} close it: \code{int8\_memory\_access} alternates these signals on
|
||||||
|
nearly every access (byte-granular access over the 16-bit bus), so treating them
|
||||||
|
as a close condition --- the first implementation attempt --- made the real
|
||||||
|
workload \emph{slower}, not faster (measured: 53.25$\to$61.25 cycles/edge on
|
||||||
|
\code{graph\_engine}'s gather); removed, corrected to 53.25$\to$37.53
|
||||||
|
cycles/edge (bandwidth +42\%, \S~\ref{sec:bandwidth}).
|
||||||
|
\end{itemize}
|
||||||
|
|
||||||
|
\begin{fnwarn}[No benefit without a sequential pattern]
|
||||||
|
Page mode only speeds up accesses that stay in the same page (or nearly) while the
|
||||||
|
controller is waiting for a new request with the page still open. Isolated,
|
||||||
|
scattered accesses (a random address every time) still pay a full ACCESS\_CYCLES,
|
||||||
|
plus a small close/reopen overhead if preceded by a WRITE or a $t_{CEM}$ timeout:
|
||||||
|
it is not a universal win, it depends on the caller's access pattern.
|
||||||
|
\end{fnwarn}
|
||||||
|
|
||||||
|
Real Fmax (\code{nextpnr-ecp5}, ch.~\ref{ch:impl}) on the integrated
|
||||||
|
\code{spi\_neuron\_top} system with Type~\#2 enabled: \textbf{75.73~MHz} at
|
||||||
|
\code{PARALLEL}=2 (was 55.59~MHz before page mode was added) and
|
||||||
|
\textbf{65.13~MHz} at \code{PARALLEL}=8, both still FAIL against the 80~MHz
|
||||||
|
target but not regressed. The critical path stays, in both cases, entirely
|
||||||
|
inside \code{u\_graph\_engine.u\_neuron} (the \code{mac8}/\code{neuron\_parallel}
|
||||||
|
accumulate chain, ch.~\ref{ch:impl}) --- \code{psram\_controller} never appears
|
||||||
|
in the critical path despite page mode's resource growth.
|
||||||
|
|
||||||
|
\section{Address map and conventions}
|
||||||
|
The addressing space is \code{ADDR\_WIDTH}=23~bits (\emph{byte} address), for a full
|
||||||
|
8~MB. The regions do not have hardwired addresses: their bases are registers set by the
|
||||||
|
host via \op{SET\_BASE} (single-layer path) or read from the descriptor table
|
||||||
|
(multi-layer path).
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.2cm} L{3.4cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Region} & \thd{Base} & \thd{Content / convention} \\
|
||||||
|
\midrule
|
||||||
|
Input $X$ & \code{x\_base} & Shared input vector, read once per invocation. \\
|
||||||
|
\rowa Weights $W$ & \code{w\_base} & Neuron-major: weights of neuron $n$ at \code{w\_base + n*N\_INPUTS} bytes. \\
|
||||||
|
Bias & \code{bias\_addr} & One byte per neuron: bias of neuron $n$ at \code{bias\_addr + n}. \\
|
||||||
|
\rowa Descriptor table & \code{table\_base} & \code{N\_LAYERS} 11-byte entries (ch.~\ref{ch:seq}). \\
|
||||||
|
Ping-pong buffers A/B & \code{buf\_a\_base} / \code{buf\_b\_base} & Intermediate outputs between layers. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\subsection{PSRAM physical addressing}
|
||||||
|
The recommended PSRAM is 4M$\times$16 (8~MB), which requires a 22-bit word address
|
||||||
|
(A0--A21). \code{int8\_memory\_access} computes \code{addr>>1}, turning the 23-bit byte
|
||||||
|
address into a 22-bit word address that maps exactly onto A0--A21; bit~22 of
|
||||||
|
\code{psram\_a} is therefore always 0 and 22 real address lines remain on the PCB.
|
||||||
|
|
||||||
|
\section{Bandwidth}
|
||||||
|
\label{sec:bandwidth}
|
||||||
|
Measured on \code{graph\_engine}'s edge-list gather (ch.~\ref{ch:grafo}), by
|
||||||
|
difference between two graph sizes to isolate the per-edge cost from the fixed
|
||||||
|
per-neuron overhead (\code{sim/graph\_engine\_bandwidth\_tb.v}):
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{5.2cm} Y Y Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{} & \thd{Before (no page mode)} & \thd{After (page mode)} & \thd{$\Delta$} \\
|
||||||
|
\midrule
|
||||||
|
Cycles/edge & 53.25 & 37.53 & $-29.5\%$ \\
|
||||||
|
\rowa Bandwidth @80\,MHz & 6.01\,MB/s & 8.53\,MB/s & $+41.9\%$ \\
|
||||||
|
Bandwidth @16\,MHz\textsuperscript{*} & 1.20\,MB/s & 1.71\,MB/s & $+41.9\%$ \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
\textsuperscript{*}recommended real oscillator (ch.~\ref{ch:hw}).
|
||||||
|
|
||||||
|
The model still remains memory-bound by construction: \code{neuron\_memory} reads
|
||||||
|
$X$ once and re-reads $W$/bias for each neuron (ch.~\ref{ch:seq}), one neuron at a
|
||||||
|
time; page mode reduces the per-byte cost of a sequential access, it does not
|
||||||
|
eliminate the access pattern itself.
|
||||||
@@ -0,0 +1,57 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {6}Memory, multi-neuron and multi-layer}{15}{chapter.6}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:seq}{{6}{15}{Memory, multi-neuron and multi-layer}{chapter.6}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {6.1}\texttt {neuron\_memory} --- memory/neuron bridge}{15}{section.6.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {6.2}\texttt {layer\_sequencer} --- multi-layer network}{15}{section.6.2}\protected@file@percent }
|
||||||
|
\gdef \LT@x {\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{57.52458pt}\LT@entry
|
||||||
|
{1}{306.05219pt}}
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {6.2.1}Ping-pong buffers}{16}{subsection.6.2.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {6.2.2}Descriptor table}{16}{subsection.6.2.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {6.3}Hierarchy of the \texttt {busy}/\texttt {done} signals}{16}{section.6.3}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/06-sequencer}{
|
||||||
|
\setcounter{page}{17}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{6}
|
||||||
|
\setcounter{section}{3}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{1}
|
||||||
|
\setcounter{LT@tables}{10}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{21}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{7}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{0}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{0}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{41}
|
||||||
|
\setcounter{lstlisting}{0}
|
||||||
|
}
|
||||||
@@ -0,0 +1,111 @@
|
|||||||
|
\chapter[Memory, multi-neuron and multi-layer]{Memory integration, multi-neuron and multi-layer}
|
||||||
|
\label{ch:seq}
|
||||||
|
|
||||||
|
\section{\texttt{neuron\_memory} --- memory/neuron bridge}
|
||||||
|
\code{neuron\_memory} connects the compute datapath to memory and manages the loop over
|
||||||
|
the neurons. It reads the $X$ vector only once (shared input), then for each neuron
|
||||||
|
re-reads $W$ and bias from RAM and feeds them to a single reused instance of
|
||||||
|
\code{neuron\_parallel}: the design is memory-bound, one neuron computed at a time,
|
||||||
|
without duplicating the datapath. The output is \code{y\_bus}, packed neuron-major
|
||||||
|
(\code{DATA\_WIDTH*N\_NEURONS} bits).
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=13mm]
|
||||||
|
\node[fnstate](idle){IDLE};
|
||||||
|
\node[fnstate,right=of idle](rx){READ\_X};
|
||||||
|
\node[fnstate,right=of rx](rw){READ\_W};
|
||||||
|
\node[fnstate,below=10mm of rw](rb){READ\_BIAS};
|
||||||
|
\node[fnstate,left=of rb](sn){START\_N};
|
||||||
|
\node[fnstate,left=of sn](wn){WAIT\_N};
|
||||||
|
\draw[fnarrow] (idle)--node[fnlbl,above]{start}(rx);
|
||||||
|
\draw[fnarrow] (rx)--node[fnlbl,above]{X read}(rw);
|
||||||
|
\draw[fnarrow] (rw)--(rb);
|
||||||
|
\draw[fnarrow] (rb)--(sn);
|
||||||
|
\draw[fnarrow] (sn)--(wn);
|
||||||
|
\draw[fnarrow] (wn) to[bend left=18] node[fnlbl,above]{next neuron}(rw);
|
||||||
|
\draw[fnarrow] (wn) to[bend right=28] node[fnlbl,below]{last neuron: done}(idle);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
The states are IDLE, READ\_X, READ\_W, READ\_BIAS, START\_N, WAIT\_N. After the last
|
||||||
|
neuron the FSM returns to IDLE and asserts \code{done}. The count of neurons and inputs
|
||||||
|
actually processed is given by \code{n\_neurons\_real}/\code{n\_inputs\_real}
|
||||||
|
(ch.~\ref{ch:param}).
|
||||||
|
|
||||||
|
\section{\texttt{layer\_sequencer} --- multi-layer network}
|
||||||
|
\code{layer\_sequencer} chains up to \code{N\_LAYERS} executions of the same
|
||||||
|
\code{neuron\_memory} instance, realizing a dense feed-forward network \emph{without}
|
||||||
|
touching the validated compute core. It reads a descriptor table written by the host and
|
||||||
|
alternates the two output buffers in RAM (ping-pong).
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=13mm]
|
||||||
|
\node[fnstate](i){IDLE};
|
||||||
|
\node[fnstate,right=of i](rd){READ\\DESC};
|
||||||
|
\node[fnstate,right=of rd](rw){READ\\WAIT};
|
||||||
|
\node[fnstate,below=10mm of rw](sl){START\\LAYER};
|
||||||
|
\node[fnstate,left=of sl](wl){WAIT\\LAYER};
|
||||||
|
\node[fnstate,left=of wl](ci){COPY\\ISSUE};
|
||||||
|
\node[fnstate,below=9mm of ci](cw){COPY\\WAIT};
|
||||||
|
\draw[fnarrow] (i)--node[fnlbl,above]{run\_start}(rd);
|
||||||
|
\draw[fnarrow] (rd)--(rw);
|
||||||
|
\draw[fnarrow] (rw)--(sl);
|
||||||
|
\draw[fnarrow] (sl)--(wl);
|
||||||
|
\draw[fnarrow] (wl)--(ci);
|
||||||
|
\draw[fnarrow] (ci)--(cw);
|
||||||
|
\draw[fnarrow] (cw) to[bend left=15] node[fnlbl,left]{next layer}(rd);
|
||||||
|
\draw[fnarrow] (cw) to[bend right=12] node[fnlbl,below]{last: seq\_done}(i);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
\subsection{Ping-pong buffers}
|
||||||
|
Layer~0 reads the external input \code{x\_base}. Layer $k>0$ reads from the buffer
|
||||||
|
written by layer $k-1$; the output of each layer is copied into the other buffer,
|
||||||
|
alternating A and B. The final output remains both in \code{y\_bus} (readable with
|
||||||
|
\op{READ\_OUTPUT}) and in the ping-pong buffer into which it was copied.
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=7mm]
|
||||||
|
\node[fnblockA,minimum width=18mm](x){X\\\code{x\_base}};
|
||||||
|
\node[fnblockD,right=10mm of x,minimum width=20mm](l0){Layer 0};
|
||||||
|
\node[fnblock,right=10mm of l0,minimum width=18mm](ba){buf A};
|
||||||
|
\node[fnblockD,right=10mm of ba,minimum width=20mm](l1){Layer 1};
|
||||||
|
\node[fnblock,right=10mm of l1,minimum width=18mm](bb){buf B};
|
||||||
|
\node[fnblockD,right=10mm of bb,minimum width=20mm](l2){Layer 2};
|
||||||
|
\draw[fnarrow] (x)--(l0); \draw[fnarrow] (l0)--(ba);
|
||||||
|
\draw[fnarrow] (ba)--(l1); \draw[fnarrow] (l1)--(bb);
|
||||||
|
\draw[fnarrow] (bb)--(l2);
|
||||||
|
\draw[fnarrowT,dashed] (l2.south) to[bend left=25] node[fnlbl,below]{copy into buf A} (ba.south);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
\subsection{Descriptor table}
|
||||||
|
Written by the host into RAM at \code{table\_base} with \op{WRITE\_RAM}; \code{N\_LAYERS}
|
||||||
|
entries of 11 bytes each, MSB-first:
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.4cm} C{1.6cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Field} & \thd{Bytes} & \thd{Meaning} \\
|
||||||
|
\midrule
|
||||||
|
\code{w\_base} & 3 & Weight base of the layer. \\
|
||||||
|
\rowa \code{bias\_addr} & 3 & Bias base of the layer. \\
|
||||||
|
\code{activation} & 1 & Layer activation (low 2 bits, cf. \code{ACT\_*}). \\
|
||||||
|
\rowa \code{n\_inputs\_real} & 2 & Actual inputs of the layer (multiple of \code{PARALLEL}). \\
|
||||||
|
\code{n\_neurons\_real} & 2 & Actual neurons of the layer. \\
|
||||||
|
\midrule
|
||||||
|
\rowh \thd{Total} & \thd{11} & per entry/layer \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\begin{fnnote}[Copy proportional to the actual width]
|
||||||
|
The sequencer copies exactly \code{n\_neurons\_real} bytes of \code{y\_bus} into the
|
||||||
|
ping-pong buffer (not the full build width): a narrower layer is copied faster, without
|
||||||
|
zero-padding in RAM. Each activation is read per-layer from the table, independent of the
|
||||||
|
\code{activation} register of the single-layer path.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\section{Hierarchy of the \texttt{busy}/\texttt{done} signals}
|
||||||
|
In the multi-layer path, \code{STATUS.busy} is the OR of the single-layer and sequencer
|
||||||
|
busy signals, while \code{STATUS.done} latches only at completion of the \emph{last}
|
||||||
|
layer, not at each intermediate layer (ch.~\ref{ch:spi}). The top-level returns control
|
||||||
|
of \code{neuron\_memory} to the direct \op{START} path at the end of the sequence.
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {7}Graph network (Type \#2)}{17}{chapter.7}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:grafo}{{7}{17}{Graph network (Type \#2)}{chapter.7}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {7.1}Two network types}{17}{section.7.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {7.2}Global activation buffer}{17}{section.7.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {7.3}Feed-forward DAG and the \texttt {src\_id < out\_id} rule}{17}{section.7.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {7.4}Data formats}{17}{section.7.4}\protected@file@percent }
|
||||||
|
\gdef \LT@xi {\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{57.52458pt}\LT@entry
|
||||||
|
{1}{306.05219pt}}
|
||||||
|
\gdef \LT@xii {\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{57.52458pt}\LT@entry
|
||||||
|
{1}{306.05219pt}}
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {7.4.1}Type \#2 descriptor (graph)}{18}{subsection.7.4.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {7.4.2}Graph edge (4~bytes, aligned)}{18}{subsection.7.4.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {7.5}\texttt {graph\_engine} --- graph engine}{18}{section.7.5}\protected@file@percent }
|
||||||
|
\gdef \LT@xiii {\LT@entry
|
||||||
|
{1}{85.97733pt}\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{277.59944pt}}
|
||||||
|
\gdef \LT@xiv {\LT@entry
|
||||||
|
{1}{142.88284pt}\LT@entry
|
||||||
|
{1}{329.4331pt}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {7.6}Type \#2 opcodes and registers}{19}{section.7.6}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {7.7}Occupancy (Type \#2 enabled)}{19}{section.7.7}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {7.8}Gather bandwidth (measured)}{20}{section.7.8}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {7.9}\texttt {netasm} host assembler}{20}{section.7.9}\protected@file@percent }
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {7.1}{\ignorespaces Pseudo-assembly example (graph)}}{20}{lstlisting.7.1}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/06b-grafo}{
|
||||||
|
\setcounter{page}{21}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{7}
|
||||||
|
\setcounter{section}{9}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{4}
|
||||||
|
\setcounter{LT@tables}{14}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{31}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{11}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{0}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{0}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{53}
|
||||||
|
\setcounter{lstlisting}{1}
|
||||||
|
}
|
||||||
@@ -0,0 +1,190 @@
|
|||||||
|
\chapter[Graph network (Type \#2)]{Two-level configuration: graph network (Type \#2)}
|
||||||
|
\label{ch:grafo}
|
||||||
|
|
||||||
|
\section{Two network types}
|
||||||
|
The engine exposes two \emph{network types} selectable by the host, with the same start
|
||||||
|
command dispatching to the correct engine:
|
||||||
|
|
||||||
|
\begin{itemize}
|
||||||
|
\item \textbf{Type \#1 --- classic network (dense).} Layers with neurons per layer, fully
|
||||||
|
connected between consecutive layers. It is the \code{layer\_sequencer} path
|
||||||
|
(ch.~\ref{ch:seq}), started by \op{RUN\_NETWORK}. Connections are \emph{implicit by
|
||||||
|
position}: nothing is enumerated, only the weights are defined, addressed as
|
||||||
|
\code{w\_base + k*n\_inputs + j}.
|
||||||
|
\item \textbf{Type \#2 --- arbitrary graph (sparse).} Starting from the input neuron ids,
|
||||||
|
each neuron's connections up to the output are defined through a per-neuron \emph{sparse
|
||||||
|
edge-list}. Connections are \emph{explicit by enumeration}: each connection is an edge
|
||||||
|
\code{(src\_id, weight)}; if it is not in the list, it does not exist.
|
||||||
|
\end{itemize}
|
||||||
|
|
||||||
|
\begin{fnnote}[The difference in one line]
|
||||||
|
Dense: you define the \emph{weights} by position in a matrix. Graph: you define each
|
||||||
|
\emph{connection} as an edge \code{(src\_id, weight)} in a per-neuron list. The two
|
||||||
|
descriptor tables share the same 11-byte format but different fields; the
|
||||||
|
\code{net\_type} register tells the engine which interpretation to use.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\section{Global activation buffer}
|
||||||
|
Type \#2 introduces an \textbf{activation buffer} indexed by \emph{signal id}, one INT8
|
||||||
|
byte per id, implemented in \textbf{on-chip \code{DP16KD} block RAM}
|
||||||
|
(\code{rtl/act\_buffer.v}). Ids \code{0..N\_in-1} are the inputs; each neuron writes its
|
||||||
|
own output into its own id. The source gather reads from here with \emph{single-cycle
|
||||||
|
random access}: this is what makes the graph cheap, because it is the access that PSRAM
|
||||||
|
(70~ns, sequential) could not accelerate.
|
||||||
|
|
||||||
|
\begin{fnspec}[V1 sizing]
|
||||||
|
\code{N\_TOTAL}=4096 signals, 16-bit id (room to 65\,536 without changing the format).
|
||||||
|
Buffer = 4~KB, i.e. 2 \code{DP16KD} blocks out of 108. The real constraint becomes the
|
||||||
|
PSRAM edge capacity ($\approx$2\,M edges at 4~B), not block RAM.
|
||||||
|
\end{fnspec}
|
||||||
|
|
||||||
|
\section{Feed-forward DAG and the \texttt{src\_id < out\_id} rule}
|
||||||
|
The graph is a feed-forward DAG: every connection points to an \textbf{already-computed}
|
||||||
|
id (\code{src\_id < out\_id}). Neurons are processed in ascending id order, so that when a
|
||||||
|
neuron is computed all its sources are ready in the buffer. Cycles and recurrence are out
|
||||||
|
of scope for V1. The rule is checked at two levels: by the host assembler (compile time)
|
||||||
|
and by a runtime guard in \code{graph\_engine} (\code{STATUS.err}), in the same philosophy
|
||||||
|
as the elaboration guard on \code{N\_INPUTS \% PARALLEL}.
|
||||||
|
|
||||||
|
\section{Data formats}
|
||||||
|
Both descriptors are 11~bytes/entry, MSB-first, at \code{table\_base}.
|
||||||
|
|
||||||
|
\subsection{Type \#2 descriptor (graph)}
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.4cm} C{1.6cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Field} & \thd{Bytes} & \thd{Meaning} \\
|
||||||
|
\midrule
|
||||||
|
\code{conn\_ptr} & 3 & Byte address in PSRAM of the neuron's edge block. \\
|
||||||
|
\rowa \code{n\_conn} & 2 & Real connections (pre-padding). \\
|
||||||
|
\code{out\_id} & 2 & Id into which the neuron's output is written. \\
|
||||||
|
\rowa \code{activation} & 1 & \code{ACT\_RELU} / \code{ACT\_NONE} (low 2 bits). \\
|
||||||
|
\code{bias} & 1 & Neuron bias (INT8). \\
|
||||||
|
\rowa \code{reserved} & 2 & 0. \\
|
||||||
|
\midrule
|
||||||
|
\rowh \thd{Total} & \thd{11} & entries in ascending \code{out\_id} order \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\subsection{Graph edge (4~bytes, aligned)}
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.4cm} C{1.6cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Field} & \thd{Bytes} & \thd{Meaning} \\
|
||||||
|
\midrule
|
||||||
|
\code{src\_id} & 2 & Source id (uint16 BE). \\
|
||||||
|
\rowa \code{weight} & 1 & Weight (INT8). \\
|
||||||
|
\code{reserved} & 1 & 0 (4-byte alignment). \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\begin{fnnote}[Padding to \texttt{PARALLEL}]
|
||||||
|
An arbitrary \code{n\_conn} is not a multiple of \code{PARALLEL}: the neuron's edge-list
|
||||||
|
is padded up to the multiple with \textbf{zero-weight} edges (waste
|
||||||
|
$\le$\code{PARALLEL}$-1$ per neuron). This keeps the datapath and its guard intact.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\section{\texttt{graph\_engine} --- graph engine}
|
||||||
|
\code{rtl/graph\_engine.v} orchestrates Type \#2 \textbf{reusing \code{neuron\_parallel}
|
||||||
|
unmodified}, as \code{neuron\_memory} does for the dense case. Key difference: between the
|
||||||
|
two modes only the \emph{X addressing} changes. In Type \#1 the input is contiguous
|
||||||
|
(\code{x\_base + i}); in Type \#2 it is a gather (\code{act\_buf[src\_id]}). The arithmetic
|
||||||
|
core is untouched.
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=4mm,start chain=going below,
|
||||||
|
every node/.style={on chain,fnblock,minimum width=52mm}]
|
||||||
|
\node[fnblockA]{\code{COPY\_INPUTS}: PSRAM \code{x\_base} $\to$ \code{act\_buf[0..N\_in-1]}};
|
||||||
|
\node{\code{READ\_DESC}: descriptor of neuron k};
|
||||||
|
\node{\code{READ\_EDGES}: stream edges + gather \code{act\_buf[src\_id]}};
|
||||||
|
\node{\code{START\_N} / \code{WAIT\_N}: group of \code{PARALLEL} $\to$ \code{neuron\_parallel}};
|
||||||
|
\node[fnblockT]{\code{WRITE\_ACT}: y $\to$ \code{act\_buf[out\_id]}};
|
||||||
|
\node{next neuron (id order)};
|
||||||
|
\node[fnblockD]{\code{WRITE\_OUTPUTS}: last \code{n\_out} $\to$ PSRAM \code{out\_base}};
|
||||||
|
\foreach \i [count=\j from 2] in {1,...,6}
|
||||||
|
\draw[fnarrow] (chain-\i) -- (chain-\j);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
The outputs are the \textbf{last \code{n\_out}} ids: in a DAG with the
|
||||||
|
\code{src\_id < out\_id} ordering the output neurons (sinks, not reused as sources)
|
||||||
|
naturally end up with the highest ids. At the end \code{graph\_engine} copies these
|
||||||
|
\code{n\_out} bytes into a PSRAM region at \code{out\_base}, which the host reads back with
|
||||||
|
\op{READ\_RAM}.
|
||||||
|
|
||||||
|
\section{Type \#2 opcodes and registers}
|
||||||
|
The type is selected with a new opcode; \op{RUN\_NETWORK} dispatches on the
|
||||||
|
\code{net\_type} register (details in ch.~\ref{ch:spi}).
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{2.6cm} L{3.4cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Opcode / sel} & \thd{Name} & \thd{Function} \\
|
||||||
|
\midrule
|
||||||
|
\op{0x11} & SET\_NET\_TYPE & \code{type(1B)}: \code{0x01}=dense (\#1), \code{0x02}=graph (\#2). Default after \op{RESET}=dense. \\
|
||||||
|
\rowa \code{SET\_BASE sel 9} & num\_neurons\_graph & Number of graph neurons (uint16). \\
|
||||||
|
\code{SET\_BASE sel 10} & n\_out & Number of output ids (uint16). \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\begin{fnnote}[Zero regression on Type \#1]
|
||||||
|
With \code{net\_type=dense} (the default value after \op{RESET}) the \#1 path is
|
||||||
|
bit-identical to before: \op{RUN\_NETWORK} keeps its \code{num\_layers(1B)} payload and the
|
||||||
|
framing of the existing opcodes does not change.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\section{Occupancy (Type \#2 enabled)}
|
||||||
|
Yosys synthesis of the full \code{spi\_neuron\_top} system with Type \#2 enabled
|
||||||
|
(\code{PARALLEL}=2):
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{4.6cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Resource} & \thd{Use} \\
|
||||||
|
\midrule
|
||||||
|
\code{DP16KD} (block RAM) & 2 (activation buffer) \\
|
||||||
|
\rowa \code{MULT18X18D} (DSP) & 4 (2 \code{neuron\_memory} + 2 \code{graph\_engine}) \\
|
||||||
|
LUT4 & 2619 \\
|
||||||
|
\rowa TRELLIS\_FF & 2467 \\
|
||||||
|
\code{\$\_TBUF\_} (PSRAM bus) & 16 \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
The device (108 \code{DP16KD}, 72 DSP, $\approx$44k LUT/FF) stays well below saturation:
|
||||||
|
Type \#2 adds a complete mode at a contained resource cost. LUT4/TRELLIS\_FF grew from an
|
||||||
|
earlier measurement (2367/2406) because of the PSRAM page mode added to the controller
|
||||||
|
(ch.~\ref{ch:mem}, \S~5.5) --- under 6\% utilization, no practical impact.
|
||||||
|
|
||||||
|
\section{Gather bandwidth (measured)}
|
||||||
|
The per-edge gather cost was \textbf{isolated} by building two structurally-identical
|
||||||
|
graphs with different edge counts and differencing the cycles: the subtraction cancels the
|
||||||
|
fixed per-neuron overhead and leaves the edge cost alone.
|
||||||
|
|
||||||
|
\begin{fnspec}[Per-edge cost]
|
||||||
|
\textbf{37.53 cycles/edge} with PSRAM page mode enabled (ch.~\ref{ch:mem}, \S~5.5) ---
|
||||||
|
\textbf{53.25 cycles/edge} without it (pre-page-mode baseline, consistent with theory:
|
||||||
|
4~bytes/edge $\times$ $\approx$13 cycles/byte over async PSRAM $\approx$52). At 80~MHz:
|
||||||
|
$\approx$2.13\,M edges/s ($\approx$8.5~MB/s, +42\% vs. baseline); at the real 16~MHz
|
||||||
|
clock: $\approx$426\,k edges/s ($\approx$1.71~MB/s).
|
||||||
|
\end{fnspec}
|
||||||
|
|
||||||
|
Page-mode read (roadmap G7, ch.~\ref{ch:roadmap}) has been implemented and measured: the
|
||||||
|
gather's sequential access benefits directly, cutting the per-edge cost by 29.5\%
|
||||||
|
(53.25$\to$37.53 cycles/edge). Each edge still pays \code{int8\_memory\_access}'s
|
||||||
|
byte-granular access (4 bytes/edge); page mode reduces the cost of each sequential byte,
|
||||||
|
not the number of accesses.
|
||||||
|
|
||||||
|
\section{\texttt{netasm} host assembler}
|
||||||
|
Readable network configuration needs no dedicated FPGA logic: a pseudo-assembly is
|
||||||
|
compiled \emph{on the host} (\code{tools/netasm/}) into the exact bytes of the tables and
|
||||||
|
edges, then loaded with \op{WRITE\_RAM}. The assembler validates at compile time
|
||||||
|
(\code{src\_id < out\_id}, \code{N\_TOTAL} bounds, padding to \code{PARALLEL}),
|
||||||
|
complementing the runtime guard.
|
||||||
|
|
||||||
|
\begin{lstlisting}[language=,caption={Pseudo-assembly example (graph)},basicstyle=\ttfamily\scriptsize]
|
||||||
|
NET graph
|
||||||
|
INPUTS 4 ; ids 0..3
|
||||||
|
NEURON n4 relu bias=2
|
||||||
|
CONN 0 w=5
|
||||||
|
CONN 1 w=-3
|
||||||
|
NEURON n5 none bias=0
|
||||||
|
CONN n4 w=2 ; symbolic reference to n4's output
|
||||||
|
CONN 2 w=7
|
||||||
|
OUTPUT n5
|
||||||
|
END
|
||||||
|
\end{lstlisting}
|
||||||
@@ -0,0 +1,82 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\gdef \LT@xv {\LT@entry
|
||||||
|
{1}{43.2982pt}\LT@entry
|
||||||
|
{1}{80.28644pt}\LT@entry
|
||||||
|
{1}{122.96556pt}\LT@entry
|
||||||
|
{1}{80.28644pt}\LT@entry
|
||||||
|
{1}{125.81102pt}}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {8}SPI host interface}{21}{chapter.8}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:spi}{{8}{21}{SPI host interface}{chapter.8}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {8.1}Physical layer}{21}{section.8.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {8.2}Framing and explicit length}{21}{section.8.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {8.3}Opcode table}{21}{section.8.3}\protected@file@percent }
|
||||||
|
\gdef \LT@xvi {\LT@entry
|
||||||
|
{1}{46.14322pt}\LT@entry
|
||||||
|
{1}{103.04872pt}\LT@entry
|
||||||
|
{1}{323.12401pt}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {8.4}\texttt {SET\_BASE} selectors}{23}{section.8.4}\protected@file@percent }
|
||||||
|
\newlabel{sec:setbase}{{8.4}{23}{\texttt {SET\_BASE} selectors}{section.8.4}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {8.5}\texttt {STATUS.done} sticky / clear-on-read}{24}{section.8.5}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {8.6}Host attention pins (\texttt {data\_ready\_n}, \texttt {irq\_n})}{24}{section.8.6}\protected@file@percent }
|
||||||
|
\gdef \LT@xvii {\LT@entry
|
||||||
|
{1}{57.52458pt}\LT@entry
|
||||||
|
{1}{114.43008pt}\LT@entry
|
||||||
|
{1}{300.36128pt}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {8.7}\texttt {READ\_CONFIG}}{25}{section.8.7}\protected@file@percent }
|
||||||
|
\newlabel{sec:readcfg}{{8.7}{25}{\texttt {READ\_CONFIG}}{section.8.7}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {8.8}Flash subsystem (opcodes 0x40--0x47, completed 2026-09-04)}{25}{section.8.8}\protected@file@percent }
|
||||||
|
\newlabel{sec:flashspi}{{8.8}{25}{Flash subsystem (opcodes 0x40--0x47, completed 2026-09-04)}{section.8.8}{}}
|
||||||
|
\gdef \LT@xviii {\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{363.57677pt}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {8.9}Session sequences}{26}{section.8.9}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {8.9.1}Single-layer path}{26}{subsection.8.9.1}\protected@file@percent }
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {8.1}{\ignorespaces Single-layer session}}{26}{lstlisting.8.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {8.9.2}Multi-layer path (RUN\_NETWORK)}{26}{subsection.8.9.2}\protected@file@percent }
|
||||||
|
\newlabel{sec:run-network}{{8.9.2}{26}{Multi-layer path (RUN\_NETWORK)}{subsection.8.9.2}{}}
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {8.2}{\ignorespaces Multi-layer session}}{26}{lstlisting.8.2}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/07-spi}{
|
||||||
|
\setcounter{page}{28}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{8}
|
||||||
|
\setcounter{section}{9}
|
||||||
|
\setcounter{subsection}{2}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{4}
|
||||||
|
\setcounter{LT@tables}{18}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{50}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{7}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{4}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{-1}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{65}
|
||||||
|
\setcounter{lstlisting}{2}
|
||||||
|
}
|
||||||
@@ -0,0 +1,290 @@
|
|||||||
|
\chapter{SPI host interface}
|
||||||
|
\label{ch:spi}
|
||||||
|
|
||||||
|
\section{Physical layer}
|
||||||
|
The FPGA is always an SPI \textbf{slave}. The v1 protocol uses SPI \textbf{Mode~0}
|
||||||
|
(CPOL=0, CPHA=0), MSB-first, single-SPI. One command per low-CS period; byte~0 of each
|
||||||
|
transaction is the opcode. Multi-byte fields are big-endian.
|
||||||
|
|
||||||
|
\begin{fnspec}[Mode 0 sampling]
|
||||||
|
\code{mosi} is sampled on the \textbf{rising} edge of \code{sclk}; \code{miso} is driven
|
||||||
|
on the \textbf{falling} edge (stable before the master's next sampling). \code{spi\_slave}
|
||||||
|
synchronizes \code{sclk/mosi/cs\_n} with a double flip-flop (3-stage CDC) before every
|
||||||
|
edge detection.
|
||||||
|
\end{fnspec}
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikztimingtable}[timing/dslope=0.1,timing/.style={x=3.4ex,y=2.2ex},
|
||||||
|
xscale=1.0,font=\scriptsize]
|
||||||
|
\sig{CS\_N} & H 1L 16L 1H \\
|
||||||
|
\sig{SCLK} & L 1L {2C(2)}8{2C(2)} 6L \\
|
||||||
|
\sig{MOSI} & U 1U 2D{b7} 2D{b6} 2D{b5} 2D{b4} 2D{b3} 2D{b2} 2D{b1} 2D{b0} 2U \\
|
||||||
|
\sig{MISO} & Z 1Z 16D{data} 1Z \\
|
||||||
|
\end{tikztimingtable}
|
||||||
|
\end{center}
|
||||||
|
\begin{center}\footnotesize\itshape\color{fnGrey}
|
||||||
|
Framing of one byte: CS falls, 8 SCLK pulses, MSB first; MISO in tri-state outside a
|
||||||
|
transaction.\end{center}
|
||||||
|
|
||||||
|
\begin{fnnote}[\texttt{tx\_byte\_req} contract]
|
||||||
|
\code{tx\_byte\_req} is a \emph{prefetch hint}, not a ``byte consumed'' event: a consumer
|
||||||
|
must advance its pointers (RAM address, response byte index) on \code{rx\_valid}, which
|
||||||
|
pulses exactly once per real byte transferred.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\section{Framing and explicit length}
|
||||||
|
The length of RAM transfers is \textbf{explicit}, not delimited by the CS edge:
|
||||||
|
\op{WRITE\_RAM}/\op{READ\_RAM} carry a 2-byte length field, so the SPI controller only
|
||||||
|
needs a byte counter. Byte addresses are 23-bit, carried in a 3-byte field with the most
|
||||||
|
significant bit reserved to 0.
|
||||||
|
|
||||||
|
\section{Opcode table}
|
||||||
|
\renewcommand{\arraystretch}{1.16}
|
||||||
|
\begin{longtable}{C{1.1cm} L{2.4cm} L{3.9cm} L{2.4cm} L{4.0cm}}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Op} & \thd{Name} & \thd{Payload (host$\to$FPGA)} & \thd{Response} & \thd{Function} \\
|
||||||
|
\midrule
|
||||||
|
\endfirsthead
|
||||||
|
\rowh \thd{Op} & \thd{Name} & \thd{Payload} & \thd{Response} & \thd{Function} \\ \midrule
|
||||||
|
\endhead
|
||||||
|
\bottomrule
|
||||||
|
\endfoot
|
||||||
|
\op{0x00} & NOP & --- & --- & No operation (idle/dummy clocking). \\
|
||||||
|
\rowa \op{0x01} & WRITE\_RAM & addr(3B)+len(2B)+data & --- & Writes a block into PSRAM (X, weights, bias, parameters). \\
|
||||||
|
\op{0x02} & READ\_RAM & addr(3B)+len(2B) & \code{len} bytes & Reads a block back from PSRAM. \\
|
||||||
|
\rowa \op{0x0F} & RESET & --- & --- & Synchronous reset of the engine and clearing of the STATUS latch; does not erase PSRAM. \\
|
||||||
|
\op{0x10} & SET\_BASE & sel(1B)+addr(3B) & --- & Sets the bases/registers (see §\ref{sec:setbase}). \\
|
||||||
|
\rowa \op{0x11} & SET\_NET\_TYPE & type(1B) & --- & Network type: \code{0x01}=dense (\#1), \code{0x02}=graph (\#2). Default after RESET=dense. \\
|
||||||
|
\rowa \op{0x20} & START & --- & --- & Starts \code{neuron\_memory} (single-layer path); ignored if busy. \\
|
||||||
|
\op{0x21} & STATUS & --- & 1 byte & bit0=\code{busy} (live), bit1=\code{done} (sticky, clear-on-read), bit2=\code{err} (graph guard), bit3=\code{flash\_err} (sticky, clear-on-read), bit4=\code{flash\_busy} (live); bit7:5=0. \\
|
||||||
|
\rowa \op{0x22} & READ\_OUTPUT & --- & \code{N\_NEURONS} bytes & \code{y\_bus} neuron-major (byte~0 = neuron~0); dense path only (Type \#1). \\
|
||||||
|
\op{0x23} & RUN\_NETWORK & num\_layers(1B) & --- & Starts execution: dispatches on \code{net\_type} to \code{layer\_sequencer} (\#1) or \code{graph\_engine} (\#2); ignored if busy. \\
|
||||||
|
\rowa \op{0x30} & READ\_CONFIG & --- & 11 bytes & Hardware configuration record (§\ref{sec:readcfg}). \\
|
||||||
|
\op{0x40} & FLASH\_READ\_BLOCK & flash\_addr(3B)+psram\_addr(3B)+len(3B) & --- & Raw flash$\to$PSRAM read, bypasses the catalog. \\
|
||||||
|
\rowa \op{0x41} & FLASH\_WRITE\_BLOCK & psram\_addr(3B)+flash\_addr(3B)+len(3B) & --- & Raw PSRAM$\to$flash write (internal erase-before-write + $\leq$256B Page Program loop + WIP poll, transparent to the host), bypasses the catalog. \\
|
||||||
|
\op{0x42} & FLASH\_ERASE & sector\_addr(3B) & --- & Standalone 4~KB sector erase (must be sector-aligned), bypasses the catalog. \\
|
||||||
|
\rowa \op{0x43} & CAT\_READ & --- & --- & Reloads the 16-slot catalog (on-chip registers) from the flash's reserved sector. \\
|
||||||
|
\op{0x44} & CAT\_WRITE\_SLOT & slot\_id(1B)+offset(3B)+len(3B)+type(1B) & --- & Registers/updates the slot's (offset, length, type) in the on-chip catalog and persists it to flash; marks the slot \emph{invalid} until \op{SAVE\_SLOT} confirms it. \\
|
||||||
|
\rowa \op{0x45} & LOAD\_SLOT & slot\_id(1B)+psram\_addr(3B) & --- & Flash$\to$PSRAM for the slot (offset/length from the catalog), verifies the CRC32 live; \code{STATUS.flash\_err} if the slot is invalid or the CRC does not match. \\
|
||||||
|
\op{0x46} & SAVE\_SLOT & slot\_id(1B)+psram\_addr(3B)+len(3B) & --- & PSRAM$\to$flash at the slot's already-registered offset, computes the CRC32 live; on success updates and persists the catalog entry (length, CRC, valid=1). \\
|
||||||
|
\rowa \op{0x47} & CAT\_INSPECT & slot\_id(1B) & 16 bytes & Synchronous read of an already-loaded catalog entry: offset[3]+len[3]+type[1]+valid[1]+CRC32[4]+reserved[4], MSB-first. \\
|
||||||
|
\end{longtable}
|
||||||
|
All flash opcodes are \emph{fire-and-forget}: the host polls \op{STATUS} (bit4=
|
||||||
|
\code{flash\_busy}, bit3=\code{flash\_err}) or the \code{irq\_n}/\code{data\_ready\_n} pins
|
||||||
|
for the outcome, except \op{CAT\_INSPECT}, which responds synchronously.
|
||||||
|
|
||||||
|
The 8 flash opcodes (\op{0x40}--\op{0x47}) are described in full, with design rationale and
|
||||||
|
measured real latencies, in §\ref{sec:flashspi} below.
|
||||||
|
|
||||||
|
\section{\texttt{SET\_BASE} selectors}
|
||||||
|
\label{sec:setbase}
|
||||||
|
\begin{tabularx}{\textwidth}{C{1.2cm} L{3.2cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{sel} & \thd{Register} & \thd{Use} \\
|
||||||
|
\midrule
|
||||||
|
0 & \code{x\_base} & Input base $X$. \\
|
||||||
|
\rowa 1 & \code{w\_base} & Weight base. \\
|
||||||
|
2 & \code{bias\_addr} & Bias base. \\
|
||||||
|
\rowa 3 & \code{table\_base} & Descriptor table base (multi-layer). \\
|
||||||
|
4 & \code{buf\_a\_base} & Ping-pong buffer A. \\
|
||||||
|
\rowa 5 & \code{buf\_b\_base} & Ping-pong buffer B. \\
|
||||||
|
6 & \code{activation} & Activation (low 2 bits) --- single-layer path only. \\
|
||||||
|
\rowa 7 & \code{n\_inputs\_real} & Runtime input width (16-bit BE) --- single-layer. \\
|
||||||
|
8 & \code{n\_neurons\_real} & Runtime neuron width (16-bit BE) --- single-layer. \\
|
||||||
|
\rowa 9 & \code{num\_neurons\_graph} & Number of graph neurons (16-bit BE) --- Type \#2. \\
|
||||||
|
10 & \code{n\_out} & Number of output ids (16-bit BE) --- Type \#2. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
Selectors 6--8 concern only the single-layer/manual path; with \op{RUN\_NETWORK} the
|
||||||
|
equivalent values are read per-layer from the descriptor table.
|
||||||
|
|
||||||
|
\begin{fnwarn}[``real=0'' edge cases fixed (2026-09-04)]
|
||||||
|
The re-certification campaign (\code{docs/validation/bugs.md}) found that several
|
||||||
|
runtime values equal to zero were unguarded, with outcomes ranging from a silently
|
||||||
|
ignored limit to a hang or arbitrary-address PSRAM writes. All five cases below are now
|
||||||
|
safe no-ops, independently verified:
|
||||||
|
\begin{itemize}
|
||||||
|
\item \code{n\_inputs\_real=0} (selector 7): completes in 1 cycle with
|
||||||
|
$y=\text{activation}(\text{bias})$ (BUG-003).
|
||||||
|
\item \code{n\_neurons\_real=0} (selector 8): completes without performing any
|
||||||
|
per-neuron computation, far faster than a full-width run (BUG-004).
|
||||||
|
\item \code{num\_neurons\_graph=0} (selector 9): completes immediately after the input
|
||||||
|
copy, without ever entering the descriptor loop (BUG-006).
|
||||||
|
\item \op{RUN\_NETWORK} with \code{num\_layers=0} (dense path): an immediate no-op ---
|
||||||
|
\textbf{before the fix it executed 256 fabricated layers, reading arbitrary PSRAM data as
|
||||||
|
descriptors} (BUG-005, CRITICAL, see \S\ref{sec:run-network} below).
|
||||||
|
\item \op{SET\_NET\_TYPE} received while a run is in progress: now silently rejected
|
||||||
|
(no effect, no SPI error) instead of remapping the arbiter's multiplexer mid-execution
|
||||||
|
--- \textbf{before the fix it caused a permanent hang of the in-progress engine}
|
||||||
|
(BUG-007, CRITICAL).
|
||||||
|
\end{itemize}
|
||||||
|
Details, evidence, and per-fix verification are in \code{docs/validation/bugs.md}.
|
||||||
|
\end{fnwarn}
|
||||||
|
|
||||||
|
\section{\texttt{STATUS.done} sticky / clear-on-read}
|
||||||
|
In \code{neuron\_memory} the \code{done} signal is a single-cycle pulse. A host polling
|
||||||
|
over SPI (much slower than the FPGA clock) would almost certainly miss a raw one-cycle
|
||||||
|
pulse. The SPI register bank therefore latches \code{done} into a sticky bit on the pulse
|
||||||
|
and clears it when the host reads \op{STATUS} (or \op{RESET}). The \code{busy} bit is
|
||||||
|
instead held at level for the whole computation and is read live.
|
||||||
|
|
||||||
|
\begin{fnwarn}[Race corrected (2026-09-02)]
|
||||||
|
A real race in the sticky mechanism (present since Phase~4) was corrected by latching a
|
||||||
|
\code{status\_snapshot} on acceptance of the \op{STATUS} opcode and conditioning the
|
||||||
|
clearing of the sticky bit on \code{status\_snapshot[1]} (it clears only if the byte
|
||||||
|
actually transmitted showed \code{done=1}). A \code{done} that arrives too late for a
|
||||||
|
snapshot is reported on the next poll instead of being lost.
|
||||||
|
\end{fnwarn}
|
||||||
|
|
||||||
|
\section{Host attention pins (\texttt{data\_ready\_n}, \texttt{irq\_n})}
|
||||||
|
Besides \op{STATUS} polling, the top-level exposes two active-low physical pins (bank 7,
|
||||||
|
ch.~\ref{ch:hw}) that mirror the sticky bits without an SPI transaction, handy for driving
|
||||||
|
a host GPIO/IRQ:
|
||||||
|
\begin{itemize}
|
||||||
|
\item \code{data\_ready\_n} = $\sim$\code{STATUS.done} (sticky): low when a result is ready
|
||||||
|
to read, returns high on the \op{STATUS} read (clear-on-read).
|
||||||
|
\item \code{irq\_n} = $\sim$\code{STATUS.err} (graph guard): low when \code{graph\_engine}'s
|
||||||
|
load-time guard has tripped. It is \textbf{not} clear-on-read: it clears only on \op{RESET}
|
||||||
|
or a fresh graph start, so an error is not missed between polls.
|
||||||
|
\end{itemize}
|
||||||
|
These are additive ports: they touch neither the existing opcodes nor the registers.
|
||||||
|
|
||||||
|
\begin{fnwarn}[\code{flash\_err} has no dedicated pin]
|
||||||
|
\code{STATUS.flash\_err} (bit3) is reported \textbf{only} in the \op{STATUS} byte, by
|
||||||
|
design: reusing \code{irq\_n} would have conflated it with graph-guard errors (two
|
||||||
|
independent error domains on one pin), while a flash operation is always host-initiated
|
||||||
|
with an opcode just issued, so polling \op{STATUS} right after --- already implicit in the
|
||||||
|
``fire-and-forget, then poll \op{STATUS}/\code{data\_ready\_n}'' convention --- is already a
|
||||||
|
natural fit, no extra async pin needed. \code{data\_ready\_n}, on the other hand,
|
||||||
|
\emph{also clears at the end of a flash operation}: it mirrors \code{STATUS.done} (bit1),
|
||||||
|
which now latches on a completed flash op too, not only on \op{RUN\_NETWORK}/\op{START}.
|
||||||
|
\end{fnwarn}
|
||||||
|
|
||||||
|
\section{\texttt{READ\_CONFIG}}
|
||||||
|
\label{sec:readcfg}
|
||||||
|
Fixed \textbf{11-byte} payload: it lets a single host firmware work with different
|
||||||
|
bitstreams without recompiling. The \code{N\_INPUTS}/\code{N\_NEURONS} values report the
|
||||||
|
build \emph{maximum} (the ceiling), not necessarily the currently loaded network.
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{C{1.6cm} L{3.6cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Byte} & \thd{Field} & \thd{Source} \\
|
||||||
|
\midrule
|
||||||
|
0 & \code{ADDR\_WIDTH} (bit) & \code{neuron\_memory.ADDR\_WIDTH} \\
|
||||||
|
\rowa 1--2 & \code{N\_INPUTS} (16-bit BE) & build maximum \\
|
||||||
|
3 & \code{N\_NEURONS} & build maximum \\
|
||||||
|
\rowa 4 & \code{PARALLEL} & build parameter \\
|
||||||
|
5 & \code{DATA\_WIDTH} (bit) & build parameter \\
|
||||||
|
\rowa 6--7 & protocol version (BE) & \code{0x0001} \\
|
||||||
|
8--9 & \code{N\_TOTAL} (16-bit BE) & max graph signals (Type \#2) \\
|
||||||
|
\rowa 10 & capability flag & bit0=\code{GRAPH\_SUPPORTED}=1 \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\section{Flash subsystem (opcodes 0x40--0x47, completed 2026-09-04)}
|
||||||
|
\label{sec:flashspi}
|
||||||
|
The FPGA has \textbf{exclusive} access to the onboard boot/persistence flash (Winbond
|
||||||
|
\code{W25Q128JV}, 16~MB SPI NOR, ch.~\ref{ch:hw} §6/§7) through a dedicated, physically
|
||||||
|
separate SPI master (\code{rtl/spi\_flash\_master.v}), never through direct host access to
|
||||||
|
the flash pins. This is \textbf{not} a filesystem: a fixed-size catalog (16 slots,
|
||||||
|
\code{rtl/flash\_slot\_manager.v}) maps \code{slot\_id}~$\to$~(offset, length, type, valid,
|
||||||
|
CRC32) in a reserved flash sector (sector 0) --- no dynamic allocation, no garbage
|
||||||
|
collection.
|
||||||
|
|
||||||
|
\begin{fnnote}[Layering (each level independently testable)]
|
||||||
|
\begin{itemize}
|
||||||
|
\item \code{rtl/spi\_flash\_master.v} --- raw SPI master toward the flash chip
|
||||||
|
(RDID/READ/WREN/PP/SE/RDSR-1). Fully independent 4-wire bus (\code{sclk}/\code{mosi}/
|
||||||
|
\code{miso}/\code{cs\_n}, all ordinary GPIO --- Phase F7, 2026-09-04): an earlier
|
||||||
|
version reused the boot \code{CCLK} pad via the ECP5 \code{USRMCLK} primitive to save
|
||||||
|
one pin, dropped because it made the ``exclusive flash bus'' claim electrically
|
||||||
|
misleading (SCLK still depended on the same pad as the config engine) and carried an
|
||||||
|
unresolved verification gap (\code{USRMCLKTS} timing never checked against the
|
||||||
|
primary Lattice sysCONFIG Usage Guide).
|
||||||
|
\item \code{rtl/flash\_copy\_engine.v} --- block-streaming engine on top: flash$\to$PSRAM
|
||||||
|
(\code{DIR\_LOAD}), PSRAM$\to$flash with internal erase-before-write + $\leq$256B
|
||||||
|
Page Program loop + WIP polling (\code{DIR\_SAVE}), standalone sector erase
|
||||||
|
(\code{DIR\_ERASE}). A low-priority master (Port D) on \code{rtl/mem\_arbiter.v}:
|
||||||
|
flash operations are ms-scale and never block inference.
|
||||||
|
\item \code{rtl/flash\_slot\_manager.v} --- the slot catalog on top of that, plus a CRC32
|
||||||
|
(\code{rtl/crc32.v}, IEEE~802.3/zlib) computed live over the real byte stream during
|
||||||
|
\op{LOAD\_SLOT}/\op{SAVE\_SLOT}, so a corrupted or partially-written slot (e.g. power
|
||||||
|
lost mid-erase) is detected even when the underlying flash operation itself reported
|
||||||
|
success.
|
||||||
|
\end{itemize}
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\begin{fnwarn}[Sector alignment is mandatory]
|
||||||
|
\op{SAVE\_SLOT} (and the raw \op{FLASH\_WRITE\_BLOCK}/\op{FLASH\_ERASE}) require the target
|
||||||
|
flash address to be 4~KB-sector-aligned --- rejected as an error otherwise, rather than a
|
||||||
|
silent partial-sector read-modify-erase-write (no scratch buffer large enough exists for
|
||||||
|
that, and every real \op{SAVE\_SLOT} already writes a whole, sector-aligned slot by
|
||||||
|
construction).
|
||||||
|
\end{fnwarn}
|
||||||
|
|
||||||
|
Full rationale, every datasheet citation, every adversarial test (CRC mismatch, never-saved
|
||||||
|
slot, page-boundary crossing, simulated power loss, arbiter contention), and the two real
|
||||||
|
bugs found and fixed during bring-up (one pre-existing in \code{psram\_controller.v}, one in
|
||||||
|
the new arbiter request handshake) are in \code{WORKLOG.md} (Phases F1-F6 entries) and
|
||||||
|
\code{docs/FPGA-Neural-Flash-Subsystem-Verification.md} (per-module coverage summary, not
|
||||||
|
repeated here).
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.4cm}Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Operation} & \thd{Measured real latency} \\
|
||||||
|
\midrule
|
||||||
|
ERASE (4~KB sector) & $\approx$400~ms (dominated by the flash chip's own internal tSE, independent of the host clock) \\
|
||||||
|
\rowa SAVE (256~B page, incl. its own erase) & $\approx$403~ms (same, tSE+tPP) \\
|
||||||
|
LOAD (4096~B) & 1.74~ms (2.35~MB/s) @80~MHz; 8.71~ms (0.47~MB/s) @16~MHz (purely SPI-clock-bound) \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
Full measurement methodology in \code{docs/FPGA-Neural-Flash-Subsystem-Verification.md}.
|
||||||
|
|
||||||
|
\section{Session sequences}
|
||||||
|
\subsection{Single-layer path}
|
||||||
|
\begin{lstlisting}[language=,caption={Single-layer session},basicstyle=\ttfamily\scriptsize]
|
||||||
|
RESET -> 0x0F
|
||||||
|
READ_CONFIG -> 0x30 (host learns N_INPUTS/N_NEURONS/...)
|
||||||
|
WRITE_RAM (weights) -> 0x01 ...
|
||||||
|
WRITE_RAM (bias) -> 0x01 ...
|
||||||
|
SET_BASE (X/W/BIAS) -> 0x10 x3
|
||||||
|
WRITE_RAM (input X) -> 0x01 ...
|
||||||
|
START -> 0x20
|
||||||
|
poll STATUS -> 0x21 (until done=1; cleared by this read)
|
||||||
|
READ_OUTPUT -> 0x22
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\subsection{Multi-layer path (RUN\_NETWORK)}
|
||||||
|
\label{sec:run-network}
|
||||||
|
\begin{lstlisting}[language=,caption={Multi-layer session},basicstyle=\ttfamily\scriptsize]
|
||||||
|
WRITE_RAM (descriptor table) -> 0x01 ...
|
||||||
|
WRITE_RAM (weights/bias per layer, X L0) -> 0x01 ...
|
||||||
|
SET_BASE (X/TABLE/BUF_A/BUF_B) -> 0x10 x4
|
||||||
|
RUN_NETWORK(num_layers) -> 0x23 <num_layers>
|
||||||
|
poll STATUS -> 0x21 (until done=1)
|
||||||
|
READ_OUTPUT -> 0x22 (y_bus of the final layer)
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\begin{fnnote}[Out of scope for v1]
|
||||||
|
Dual~SPI and CRC/checksum on host transfers (SPI assumed reliable on a board trace --- not
|
||||||
|
to be confused with the flash catalog's CRC32, §\ref{sec:flashspi}, which protects a
|
||||||
|
different domain: flash$\leftrightarrow$PSRAM persistence, not the host SPI link).
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\begin{fnwarn}[\op{WRITE\_RAM}/\op{READ\_RAM} have no backpressure to the host --- a real risk, not a theoretical one]
|
||||||
|
Every received/produced byte must be fully processed by \code{spi\_engine} before the next
|
||||||
|
SCLK-driven byte boundary arrives --- reasonable for the initial bulk-loading of weights/
|
||||||
|
inputs, not a real-time path. The concrete risk: if a host issues \op{WRITE\_RAM}/
|
||||||
|
\op{READ\_RAM} before \code{psram\_controller.v}'s power-up sequence has completed
|
||||||
|
($\sim$150~\textmu s after reset, \code{STATE\_INIT}+\code{STATE\_CR\_INIT}),
|
||||||
|
\code{spi\_engine} stalls waiting for the very first PSRAM access to complete, while the
|
||||||
|
host --- not slowed by any handshake --- keeps clocking bytes. Bytes received during that
|
||||||
|
stall are \textbf{silently dropped}, with no error and no hang: just wrong data in PSRAM.
|
||||||
|
Found during the flash-subsystem work (\code{WORKLOG.md}, Phase~F5) via a minimal
|
||||||
|
\op{WRITE\_RAM}-only reproduction with no flash opcodes involved at all: it is a general
|
||||||
|
hazard for any host, not specific to the flash opcodes. \textbf{Current mitigation: a host
|
||||||
|
must wait for PSRAM power-up (or otherwise ensure the FPGA has been out of reset for
|
||||||
|
$>$150~\textmu s) before its first \op{WRITE\_RAM}/\op{READ\_RAM}.} Not fixed at the
|
||||||
|
protocol level (would need real backpressure, a larger change) --- declared here as an open
|
||||||
|
risk, not silently worked around.
|
||||||
|
\end{fnwarn}
|
||||||
@@ -0,0 +1,78 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\gdef \LT@xix {\LT@entry
|
||||||
|
{1}{40.45274pt}\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{51.83368pt}\LT@entry
|
||||||
|
{1}{51.83368pt}\LT@entry
|
||||||
|
{1}{219.45667pt}}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {9}Network programming}{28}{chapter.9}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:prog}{{9}{28}{Network programming}{chapter.9}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {9.1}General flow}{28}{section.9.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {9.2}Registers and opcodes involved}{28}{section.9.2}\protected@file@percent }
|
||||||
|
\gdef \LT@xx {\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{363.57677pt}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {9.3}Type \#1 --- dense network}{29}{section.9.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {9.3.1}Memory layout}{29}{subsection.9.3.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {9.3.2}Worked example: a $4\to 4\to 2$ network}{29}{subsection.9.3.2}\protected@file@percent }
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {9.1}{\ignorespaces Dense descriptor table (22 bytes)}}{29}{lstlisting.9.1}\protected@file@percent }
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {9.2}{\ignorespaces SPI session (dense)}}{29}{lstlisting.9.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {9.3.3}Host pseudocode (dense)}{29}{subsection.9.3.3}\protected@file@percent }
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {9.3}{\ignorespaces Encoding and loading a dense network}}{29}{lstlisting.9.3}\protected@file@percent }
|
||||||
|
\gdef \LT@xxi {\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{363.57677pt}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {9.4}Type \#2 --- graph network}{30}{section.9.4}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {9.4.1}Memory layout}{30}{subsection.9.4.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {9.4.2}Worked example}{30}{subsection.9.4.2}\protected@file@percent }
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {9.4}{\ignorespaces Graph descriptors + edges}}{30}{lstlisting.9.4}\protected@file@percent }
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {9.5}{\ignorespaces SPI session (graph)}}{30}{lstlisting.9.5}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {9.4.3}Host pseudocode (graph)}{30}{subsection.9.4.3}\protected@file@percent }
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {9.6}{\ignorespaces Encoding and loading a graph}}{31}{lstlisting.9.6}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {9.4.4}\texttt {netasm} pseudo-assembly}{31}{subsection.9.4.4}\protected@file@percent }
|
||||||
|
\@writefile{lol}{\contentsline {lstlisting}{\numberline {9.7}{\ignorespaces netasm: source and generated bytes}}{31}{lstlisting.9.7}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/07b-programmazione}{
|
||||||
|
\setcounter{page}{32}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{9}
|
||||||
|
\setcounter{section}{4}
|
||||||
|
\setcounter{subsection}{4}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{3}
|
||||||
|
\setcounter{LT@tables}{21}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{52}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{21}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{4}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{-1}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{77}
|
||||||
|
\setcounter{lstlisting}{7}
|
||||||
|
}
|
||||||
@@ -0,0 +1,232 @@
|
|||||||
|
\chapter[Network programming]{Neural network programming}
|
||||||
|
\label{ch:prog}
|
||||||
|
|
||||||
|
This chapter is the practical guide to encoding a network for FPGA-Neural: how it is laid
|
||||||
|
out in memory, which registers are set and how it is started, for both topologies. It
|
||||||
|
assumes the SPI opcodes (ch.~\ref{ch:spi}) and the descriptor formats (ch.~\ref{ch:seq},
|
||||||
|
\ref{ch:grafo}).
|
||||||
|
|
||||||
|
\section{General flow}
|
||||||
|
Whatever the type, the cycle is the same: the host \emph{builds the data structures in
|
||||||
|
RAM}, sets the \emph{base registers}, declares the \emph{network type}, \emph{starts} and
|
||||||
|
\emph{reads back} the result.
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=3mm,start chain=going below,
|
||||||
|
every node/.style={on chain,fnblock,minimum width=64mm}]
|
||||||
|
\node[fnblockA]{1. \op{RESET} --- clears the engine and the STATUS latch};
|
||||||
|
\node{2. \op{SET\_NET\_TYPE} --- dense (\#1) or graph (\#2)};
|
||||||
|
\node{3. \op{WRITE\_RAM} --- tables, weights/edges, bias, input X};
|
||||||
|
\node{4. \op{SET\_BASE} --- base registers (x, table, \ldots)};
|
||||||
|
\node[fnblockT]{5. \op{RUN\_NETWORK} --- dispatch on \code{net\_type}};
|
||||||
|
\node{6. \op{STATUS} polling --- waits for \code{done}};
|
||||||
|
\node[fnblockD]{7. \op{READ\_OUTPUT} / \op{READ\_RAM} --- result};
|
||||||
|
\foreach \i [count=\j from 2] in {1,...,6} \draw[fnarrow] (chain-\i)--(chain-\j);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
\section{Registers and opcodes involved}
|
||||||
|
All base values are set with \op{SET\_BASE} \code{sel(1B)+addr(3B)}. Selectors:
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{C{1.0cm} L{3.4cm} C{1.4cm} C{1.4cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{sel} & \thd{Register} & \thd{Type \#1} & \thd{Type \#2} & \thd{Use} \\
|
||||||
|
\midrule
|
||||||
|
0 & \code{x\_base} & \checkmark & \checkmark & Input base $X$. \\
|
||||||
|
\rowa 3 & \code{table\_base} & \checkmark & \checkmark & Descriptor table. \\
|
||||||
|
4 & \code{buf\_a\_base} & \checkmark & \checkmark\textsuperscript{$\ast$} & Ping-pong A (\#1) / \code{out\_base} reuse (\#2). \\
|
||||||
|
\rowa 5 & \code{buf\_b\_base} & \checkmark & --- & Ping-pong B (\#1). \\
|
||||||
|
9 & \code{num\_neurons\_graph} & --- & \checkmark & Number of graph neurons. \\
|
||||||
|
\rowa 10 & \code{n\_out} & --- & \checkmark & Number of output ids. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
\begin{center}\footnotesize\itshape\color{fnGrey}
|
||||||
|
$\ast$ In Type \#2 the ping-pong buffers are unused: selector 4 is reused as
|
||||||
|
\code{out\_base} (region into which outputs are copied). Selectors 1/2/6/7/8 concern only
|
||||||
|
the manual single-layer path (\op{START}), not \op{RUN\_NETWORK}.\end{center}
|
||||||
|
|
||||||
|
For Type \#1, the \emph{per-layer} \code{w\_base}/\code{bias\_addr} are \textbf{not} set
|
||||||
|
with \op{SET\_BASE}: they are fields of the descriptor table. \op{SET\_NET\_TYPE} defaults
|
||||||
|
to \emph{dense} after \op{RESET}, so a \#1 network works even without issuing it.
|
||||||
|
|
||||||
|
% ======================================================================
|
||||||
|
\section{Type \#1 --- dense network}
|
||||||
|
|
||||||
|
\subsection{Memory layout}
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.4cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Structure} & \thd{Format} \\
|
||||||
|
\midrule
|
||||||
|
Input $X$ & \code{n\_inputs\_real} INT8 bytes at \code{x\_base}. \\
|
||||||
|
\rowa Weights (per layer) & Neuron-major: neuron $k$ at \code{w\_base + k*n\_inputs\_real}, \code{n\_neurons*n\_inputs} bytes. \\
|
||||||
|
Bias (per layer) & One INT8 byte per neuron at \code{bias\_addr}. \\
|
||||||
|
\rowa Descriptor table & \code{num\_layers} 11-byte entries at \code{table\_base}. \\
|
||||||
|
Buffers A/B & Ping-pong intermediate outputs. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
Descriptor (11 bytes, MSB-first): \code{w\_base}(3) $|$ \code{bias\_addr}(3) $|$
|
||||||
|
\code{activation}(1) $|$ \code{n\_inputs\_real}(2) $|$ \code{n\_neurons\_real}(2).
|
||||||
|
|
||||||
|
\subsection{Worked example: a $4\to4\to2$ network}
|
||||||
|
Layer~0: 4 inputs, 4 neurons, ReLU. Layer~1: 4 inputs, 2 neurons, linear
|
||||||
|
(\code{PARALLEL}=2, so each \code{n\_inputs\_real} is a multiple of 2). Chosen addresses:
|
||||||
|
\code{table\_base}=\code{0x000000}, \code{x\_base}=\code{0x001000}, L0 weights/bias at
|
||||||
|
\code{0x002000}/\code{0x002100}, L1 at \code{0x002200}/\code{0x002300}, buffers at
|
||||||
|
\code{0x003000}/\code{0x003100}.
|
||||||
|
|
||||||
|
\begin{lstlisting}[language=,caption={Dense descriptor table (22 bytes)},basicstyle=\ttfamily\scriptsize]
|
||||||
|
Layer 0: 00 20 00 | 00 21 00 | 01 | 00 04 | 00 04
|
||||||
|
w_base bias_addr ReLU n_in=4 n_neu=4
|
||||||
|
Layer 1: 00 22 00 | 00 23 00 | 00 | 00 04 | 00 02
|
||||||
|
w_base bias_addr NONE n_in=4 n_neu=2
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\begin{lstlisting}[language=,caption={SPI session (dense)},basicstyle=\ttfamily\scriptsize]
|
||||||
|
0x0F RESET
|
||||||
|
0x11 01 SET_NET_TYPE = dense
|
||||||
|
0x01 000000 0016 <22-byte table> WRITE_RAM table
|
||||||
|
0x01 002000 0010 <16-byte L0 wts> WRITE_RAM L0 weights (neuron-major)
|
||||||
|
0x01 002100 0004 <4-byte L0 bias>
|
||||||
|
0x01 002200 0008 <8-byte L1 wts>
|
||||||
|
0x01 002300 0002 <2-byte L1 bias>
|
||||||
|
0x01 001000 0004 <x0 x1 x2 x3> WRITE_RAM input X
|
||||||
|
0x10 00 001000 SET_BASE x_base
|
||||||
|
0x10 03 000000 SET_BASE table_base
|
||||||
|
0x10 04 003000 SET_BASE buf_a
|
||||||
|
0x10 05 003100 SET_BASE buf_b
|
||||||
|
0x23 02 RUN_NETWORK num_layers=2
|
||||||
|
0x21 ... poll STATUS until done=1
|
||||||
|
0x22 READ_OUTPUT -> 2 bytes (final layer)
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\subsection{Host pseudocode (dense)}
|
||||||
|
\begin{lstlisting}[language=,caption={Encoding and loading a dense network},basicstyle=\ttfamily\scriptsize]
|
||||||
|
def load_dense(layers, X): # layers in execution order
|
||||||
|
spi(RESET); spi(SET_NET_TYPE, DENSE)
|
||||||
|
table = b""
|
||||||
|
for L in layers: # L: weights[n][k], bias[n], act, n_in, n_out
|
||||||
|
assert L.n_in % PARALLEL == 0
|
||||||
|
w = alloc(L.weights_neuron_major) # k slow, input fast
|
||||||
|
b = alloc(L.bias)
|
||||||
|
table += u24(w)+u24(b)+u8(L.act)+u16(L.n_in)+u16(L.n_out)
|
||||||
|
write_ram(TABLE_BASE, table)
|
||||||
|
write_ram(X_BASE, X)
|
||||||
|
set_base(0, X_BASE); set_base(3, TABLE_BASE)
|
||||||
|
set_base(4, BUF_A); set_base(5, BUF_B)
|
||||||
|
spi(RUN_NETWORK, len(layers))
|
||||||
|
wait_status_done()
|
||||||
|
return read_output(layers[-1].n_out)
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
% ======================================================================
|
||||||
|
\section{Type \#2 --- graph network}
|
||||||
|
|
||||||
|
\subsection{Memory layout}
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.4cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Structure} & \thd{Format} \\
|
||||||
|
\midrule
|
||||||
|
Input $X$ & \code{N\_in} bytes at \code{x\_base}; copied into \code{act\_buf[0..N\_in-1]} at start. \\
|
||||||
|
\rowa Descriptor table & \code{num\_neurons\_graph} 11-byte entries at \code{table\_base}, in ascending \code{out\_id} order. \\
|
||||||
|
Edge blocks & Per neuron: \code{n\_conn} 4-byte edges at \code{conn\_ptr}, padded to a multiple of \code{PARALLEL} (zero-weight edges). \\
|
||||||
|
\rowa Outputs & \code{n\_out} bytes written to \code{out\_base} (=selector 4). \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
Graph descriptor (11 bytes): \code{conn\_ptr}(3) $|$ \code{n\_conn}(2) $|$ \code{out\_id}(2)
|
||||||
|
$|$ \code{activation}(1) $|$ \code{bias}(1) $|$ \code{reserved}(2). \quad
|
||||||
|
Edge (4 bytes): \code{src\_id}(2) $|$ \code{weight}(1) $|$ \code{reserved}(1). \quad
|
||||||
|
Rule: \code{src\_id < out\_id} (feed-forward DAG).
|
||||||
|
|
||||||
|
\subsection{Worked example}
|
||||||
|
4 inputs (ids 0--3). Neuron n4 (\code{out\_id}=4, ReLU, bias=2) connected to ids 0 and 1;
|
||||||
|
neuron n5 (\code{out\_id}=5, linear, bias=0) connected to n4 (id~4) and id~2; output = n5
|
||||||
|
(\code{n\_out}=1). \code{PARALLEL}=2, both have 2 connections (no padding). Addresses:
|
||||||
|
\code{table\_base}=\code{0x000000}, edges at \code{0x000100}, \code{x\_base}=
|
||||||
|
\code{0x001000}, \code{out\_base}=\code{0x002000}.
|
||||||
|
|
||||||
|
\begin{lstlisting}[language=,caption={Graph descriptors + edges},basicstyle=\ttfamily\scriptsize]
|
||||||
|
Descriptors (at 0x000000, 22 bytes):
|
||||||
|
n4: 00 01 00 | 00 02 | 00 04 | 01 | 02 | 00 00
|
||||||
|
conn_ptr n_conn out_id ReLU bias rsv
|
||||||
|
n5: 00 01 08 | 00 02 | 00 05 | 00 | 00 | 00 00
|
||||||
|
conn_ptr n_conn out_id NONE bias rsv
|
||||||
|
|
||||||
|
Edge blocks (at 0x000100, 4 bytes/edge: src_id, weight, rsv):
|
||||||
|
n4 @0x000100: 00 00 05 00 (src=0, w=+5)
|
||||||
|
00 01 FD 00 (src=1, w=-3) ; -3 = 0xFD
|
||||||
|
n5 @0x000108: 00 04 02 00 (src=4, w=+2) ; id4 = n4's output
|
||||||
|
00 02 07 00 (src=2, w=+7)
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\begin{lstlisting}[language=,caption={SPI session (graph)},basicstyle=\ttfamily\scriptsize]
|
||||||
|
0x0F RESET
|
||||||
|
0x11 02 SET_NET_TYPE = graph
|
||||||
|
0x01 000000 0016 <22-byte table> WRITE_RAM descriptors
|
||||||
|
0x01 000100 0010 <16-byte edges> WRITE_RAM edge blocks
|
||||||
|
0x01 001000 0004 <x0 x1 x2 x3> WRITE_RAM input X
|
||||||
|
0x10 00 001000 SET_BASE x_base
|
||||||
|
0x10 03 000000 SET_BASE table_base
|
||||||
|
0x10 04 002000 SET_BASE out_base (sel 4 reuse)
|
||||||
|
0x10 09 000002 SET_BASE num_neurons_graph = 2
|
||||||
|
0x10 0A 000001 SET_BASE n_out = 1
|
||||||
|
0x23 00 RUN_NETWORK (dispatch to graph_engine)
|
||||||
|
0x21 ... poll STATUS (bit2=err if src_id>=out_id)
|
||||||
|
0x02 002000 0001 READ_RAM out_base -> 1 byte (n5 output)
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\subsection{Host pseudocode (graph)}
|
||||||
|
\begin{lstlisting}[language=,caption={Encoding and loading a graph},basicstyle=\ttfamily\scriptsize]
|
||||||
|
def load_graph(neurons, X, n_out): # neurons sorted by ascending out_id
|
||||||
|
spi(RESET); spi(SET_NET_TYPE, GRAPH)
|
||||||
|
edges = b""; table = b""
|
||||||
|
for N in neurons: # N: out_id, conns=[(src_id,w)...], act, bias
|
||||||
|
for (src,_) in N.conns:
|
||||||
|
assert src < N.out_id and src < N_TOTAL # DAG rule
|
||||||
|
conn_ptr = EDGE_BASE + len(edges)
|
||||||
|
padded = pad(N.conns, PARALLEL, fill=(0,0)) # zero-weight edges
|
||||||
|
for (src,w) in padded:
|
||||||
|
edges += u16(src)+i8(w)+u8(0)
|
||||||
|
table += u24(conn_ptr)+u16(len(N.conns))+u16(N.out_id) \
|
||||||
|
+ u8(N.act)+i8(N.bias)+u16(0)
|
||||||
|
write_ram(TABLE_BASE, table); write_ram(EDGE_BASE, edges)
|
||||||
|
write_ram(X_BASE, X)
|
||||||
|
set_base(0, X_BASE); set_base(3, TABLE_BASE); set_base(4, OUT_BASE)
|
||||||
|
set_base(9, len(neurons)); set_base(10, n_out)
|
||||||
|
spi(RUN_NETWORK, 0) # payload ignored in graph
|
||||||
|
wait_status_done()
|
||||||
|
return read_ram(OUT_BASE, n_out)
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\subsection{\texttt{netasm} pseudo-assembly}
|
||||||
|
The readable description is compiled by the host assembler (\code{tools/netasm/}) into
|
||||||
|
exactly the table and edge bytes above. Example equivalent to the worked graph:
|
||||||
|
|
||||||
|
\begin{lstlisting}[language=,caption={netasm: source and generated bytes},basicstyle=\ttfamily\scriptsize]
|
||||||
|
; --- source ---
|
||||||
|
NET graph
|
||||||
|
INPUTS 4 ; ids 0..3
|
||||||
|
NEURON n4 relu bias=2
|
||||||
|
CONN 0 w=5
|
||||||
|
CONN 1 w=-3
|
||||||
|
NEURON n5 none bias=0
|
||||||
|
CONN n4 w=2 ; symbolic reference -> id 4
|
||||||
|
CONN 2 w=7
|
||||||
|
OUTPUT n5
|
||||||
|
END
|
||||||
|
|
||||||
|
; --- the assembler emits ---
|
||||||
|
; assigned ids: n4=4, n5=5 (guarantees src_id < out_id)
|
||||||
|
; descriptors: 00 01 00 00 02 00 04 01 02 00 00
|
||||||
|
; 00 01 08 00 02 00 05 00 00 00 00
|
||||||
|
; edges: 00 00 05 00 00 01 FD 00 (n4)
|
||||||
|
; 00 04 02 00 00 02 07 00 (n5)
|
||||||
|
; registers: table_base, x_base, out_base, num_neurons=2, n_out=1
|
||||||
|
; compile-time checks: src_id<out_id, N_TOTAL, padding to PARALLEL
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\begin{fnnote}[Why two encoding levels]
|
||||||
|
The host pseudocode and \code{netasm} produce the \emph{same bytes}. The former is useful
|
||||||
|
when the network is generated at runtime (e.g. trained weights); the latter when the
|
||||||
|
topology is hand-written or version-controlled as source. In both cases the FPGA receives
|
||||||
|
only tables and data via \op{WRITE\_RAM}: no on-board interpreter.
|
||||||
|
\end{fnnote}
|
||||||
@@ -0,0 +1,54 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\gdef \LT@xxii {\LT@entry
|
||||||
|
{1}{48.98866pt}\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{314.5881pt}}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {10}Arbitration and top-level}{32}{chapter.10}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:top}{{10}{32}{Arbitration and top-level}{chapter.10}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {10.1}\texttt {mem\_arbiter} --- three-port arbiter}{32}{section.10.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {10.2}\texttt {spi\_neuron\_top} --- full integration}{32}{section.10.2}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/08-toplevel}{
|
||||||
|
\setcounter{page}{34}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{10}
|
||||||
|
\setcounter{section}{2}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{1}
|
||||||
|
\setcounter{LT@tables}{22}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{54}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{21}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{4}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{-1}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{80}
|
||||||
|
\setcounter{lstlisting}{0}
|
||||||
|
}
|
||||||
@@ -0,0 +1,78 @@
|
|||||||
|
\chapter[Arbitration and top-level]{Arbitration and top-level integration}
|
||||||
|
\label{ch:top}
|
||||||
|
|
||||||
|
\section{\texttt{mem\_arbiter} --- three-port arbiter}
|
||||||
|
A single byte-level memory master (which feeds the shared chain
|
||||||
|
\code{int8\_memory\_access} $\to$ \code{memory\_interface} $\to$ \code{psram\_controller})
|
||||||
|
is arbitrated among three requesters:
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{C{1.3cm} L{3.4cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Port} & \thd{Master} & \thd{Accesses} \\
|
||||||
|
\midrule
|
||||||
|
A & \code{spi\_engine} & \op{WRITE\_RAM} / \op{READ\_RAM}. \\
|
||||||
|
\rowa B & \code{neuron\_memory} & X/W/bias reads during an execution. \\
|
||||||
|
C & \code{layer\_sequencer} & Descriptor reads + buffer writes between layers. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
Fixed priority \textbf{B $>$ C $>$ A}: an inference in progress is more critical than the
|
||||||
|
sequencer's bookkeeping, which in turn is more critical than a manual SPI access that has
|
||||||
|
just arrived. In normal operation B and C are anyway temporally disjoint
|
||||||
|
(\code{neuron\_memory} requests only during an execution, \code{layer\_sequencer} only in
|
||||||
|
the pauses between layers), so the priority matters mostly for the corner case of a
|
||||||
|
manual \op{WRITE\_RAM}/\op{READ\_RAM} arriving during a multi-layer execution.
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=6mm]
|
||||||
|
\node[fnblock,minimum width=30mm](a){Port A --- \code{spi\_engine}};
|
||||||
|
\node[fnblock,below=4mm of a,minimum width=30mm](b){Port B --- \code{neuron\_memory}};
|
||||||
|
\node[fnblock,below=4mm of b,minimum width=30mm](c){Port C --- \code{layer\_sequencer}};
|
||||||
|
\node[fnblockD,right=16mm of b,minimum width=26mm,minimum height=16mm](arb){\code{mem\_arbiter}\\{\scriptsize B$>$C$>$A}};
|
||||||
|
\node[fnblockT,right=14mm of arb,minimum width=26mm](m){shared memory\\{\scriptsize chain}};
|
||||||
|
\draw[fnarrow] (a)-|(arb.west|-a); \draw[fnarrow] (b)--(arb.west);
|
||||||
|
\draw[fnarrow] (c)-|(arb.west|-c);
|
||||||
|
\draw[fnbus] (arb)--(m);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
Once access is granted, the arbiter retains ownership until the single transaction's
|
||||||
|
\code{m\_ready} pulse, then releases: all three masters emit \code{req} as a clean
|
||||||
|
one-cycle pulse, so a queue-less grant-and-forward design suffices.
|
||||||
|
|
||||||
|
\section{\texttt{spi\_neuron\_top} --- full integration}
|
||||||
|
The top-level connects SPI (\code{spi\_slave}+\code{spi\_engine}), the arbiter, the
|
||||||
|
sequencer, \code{neuron\_memory} and the PSRAM chain. The reset of \code{neuron\_memory}
|
||||||
|
is the OR of the global reset with the soft-reset pulse of the \op{RESET} opcode, so the
|
||||||
|
host can recover the engine over SPI without a physical reset (the RAM stays intact).
|
||||||
|
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=7mm]
|
||||||
|
\node[fnblockA,minimum width=22mm](ss){\code{spi\_slave}};
|
||||||
|
\node[fnblockA,right=8mm of ss,minimum width=22mm](se){\code{spi\_engine}};
|
||||||
|
\node[fnblockT,below=8mm of se,minimum width=26mm](sq){\code{layer\_sequencer}};
|
||||||
|
\node[fnblockD,right=10mm of se,minimum width=24mm](mux){ctrl MUX\\{\scriptsize on \code{seq\_busy}}};
|
||||||
|
\node[fnblock,below=8mm of mux,minimum width=26mm](nm){\code{neuron\_memory}};
|
||||||
|
\node[fnblockD,right=10mm of mux,minimum width=22mm](arb){\code{mem\_arbiter}};
|
||||||
|
\node[fnblockA,right=8mm of arb,minimum width=26mm](mem){PSRAM chain};
|
||||||
|
\draw[fnarrow] (ss)--(se);
|
||||||
|
\draw[fnarrow] (se)--(mux);
|
||||||
|
\draw[fnarrow] (sq)--(mux);
|
||||||
|
\draw[fnarrow] (mux)--(nm);
|
||||||
|
\draw[fnarrow] (se.south) to[bend right=10] (arb.north west);
|
||||||
|
\draw[fnarrow] (nm)--(arb);
|
||||||
|
\draw[fnarrow] (sq.east) to[bend right=20] (arb.south west);
|
||||||
|
\draw[fnbus] (arb)--(mem);
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
The multiplexer switches the control lines of \code{neuron\_memory} between the sequencer
|
||||||
|
(while \code{seq\_busy} is high) and the direct path of \code{spi\_engine} (legacy
|
||||||
|
single-layer mode), returning the engine to the direct path at the end of the sequence.
|
||||||
|
|
||||||
|
\begin{fnnote}[End-to-end verification]
|
||||||
|
\code{spi\_neuron\_top} is verified in simulation with real PSRAM
|
||||||
|
(\code{psram\_model.v}, no mock): RESET/READ\_CONFIG/WRITE\_RAM/READ\_RAM/SET\_BASE/
|
||||||
|
START/STATUS/READ\_OUTPUT and \op{RUN\_NETWORK} are exercised purely over simulated SPI
|
||||||
|
(ch.~\ref{ch:impl}).
|
||||||
|
\end{fnnote}
|
||||||
@@ -0,0 +1,76 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\gdef \LT@xxiii {\LT@entry
|
||||||
|
{1}{154.26378pt}\LT@entry
|
||||||
|
{1}{97.35826pt}\LT@entry
|
||||||
|
{1}{220.69391pt}}
|
||||||
|
\gdef \LT@xxiv {\LT@entry
|
||||||
|
{1}{51.83368pt}\LT@entry
|
||||||
|
{1}{63.21504pt}\LT@entry
|
||||||
|
{1}{51.83368pt}\LT@entry
|
||||||
|
{1}{57.52458pt}\LT@entry
|
||||||
|
{1}{57.52458pt}\LT@entry
|
||||||
|
{1}{54.67912pt}\LT@entry
|
||||||
|
{1}{51.83368pt}}
|
||||||
|
\expandafter \gdef \csname pgfplots@labelstyle@plt:tp\endcsname {fnBlue={\pgfkeysnovalue },mark={square*},thick={\pgfkeysnovalue },mark options={fill=fnBlue}}
|
||||||
|
\expandafter \gdef \csname pgfplots@show@ref@plt:tp\endcsname {\begingroup \def \pgfplots@draw@image {\def \tikz@plot@handler {\pgfkeys {/pgf/plots/@handler options/.cd, start=\relax , end macro=\relax , point macro=\pgfutil@gobble , jump macro=\relax , special macro=\pgfutil@gobble , point macro=\pgf@plot@line@handler , jump=\global \let \pgf@plotstreampoint \pgf@plot@line@handler@move }\begingroup \escapechar =-1 \edef \pgfplotsplothandlername {\string \pgfplothandlerlineto }\pgfmath@smuggleone \pgfplotsplothandlername \endgroup \def \pgfplotsplothandlerLUAfactory {function(axis, pointmetainputhandler) return pgfplots.GenericPlothandler.new("\pgfplotsplothandlername ", axis,pointmetainputhandler) end}\def \pgfplotsplothandlerLUAvisualizerfactory {pgfplots.defaultPlotVisualizerFactory}}\pgfkeysdef {/pgfplots/legend image code}{\draw [/pgfplots/mesh=false,bar width=3pt,bar shift=0pt,mark repeat=2,mark phase=2,] plot coordinates { (0cm,0cm) (0.3cm,0cm) (0.6cm,0cm)};}\pgfkeysdef {/pgfplots/every legend image post}{}\pgfkeyssetvalue {/pgfplots/mark list fill}{.!80!black}\pgfkeysdef {/pgfplots/every crossref picture}{\pgfkeysalso {baseline,yshift=0.3em}}\pgfplots@show@small@legendplots {fnBlue={\pgfkeysnovalue },mark={square*},thick={\pgfkeysnovalue },mark options={fill=fnBlue}}{}}\ifpgfpicture \scope [/pgfplots/every crossref picture]\pgfplots@draw@image \endscope \else \expandafter \ifx \csname tikzappendtofigurename\endcsname \relax \else \begingroup \tikzappendtofigurename {_crossref}\fi \tikz [/pgfplots/every crossref picture]{\pgfplots@draw@image }\expandafter \ifx \csname tikzappendtofigurename\endcsname \relax \else \endgroup \fi \fi \endgroup }
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {11}ECP5 implementation}{34}{chapter.11}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:impl}{{11}{34}{ECP5 implementation}{chapter.11}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {11.1}Flow and verification}{34}{section.11.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {11.2}Datapath benchmark (256$\times $4)}{34}{section.11.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {11.2.1}Fmax and throughput versus parallelism}{35}{subsection.11.2.1}\protected@file@percent }
|
||||||
|
\newlabel{plt:tp}{{\expandafter\protect\csname pgfplots@show@ref@plt:tp\endcsname}{35}{Fmax and throughput versus parallelism}{pgfplotslink.11.2.0}{}}
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {11.2.2}Interpretation}{35}{subsection.11.2.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {11.2.3}Critical path and the 100~MHz limit}{35}{subsection.11.2.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {11.3}Full integrated system}{35}{section.11.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {11.3.1}Cause: the saturation/ReLU carry chain}{35}{subsection.11.3.1}\protected@file@percent }
|
||||||
|
\gdef \LT@xxv {\LT@entry
|
||||||
|
{1}{142.88284pt}\LT@entry
|
||||||
|
{1}{85.97733pt}\LT@entry
|
||||||
|
{1}{80.28644pt}\LT@entry
|
||||||
|
{1}{163.16934pt}}
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {11.3.2}Timing closure (2026-09-03)}{36}{subsection.11.3.2}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/09-implementazione}{
|
||||||
|
\setcounter{page}{37}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{11}
|
||||||
|
\setcounter{section}{3}
|
||||||
|
\setcounter{subsection}{2}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{3}
|
||||||
|
\setcounter{LT@tables}{25}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{62}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{21}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{4}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{-1}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{89}
|
||||||
|
\setcounter{lstlisting}{0}
|
||||||
|
}
|
||||||
@@ -0,0 +1,169 @@
|
|||||||
|
\chapter[ECP5 implementation]{ECP5 implementation and characterization}
|
||||||
|
\label{ch:impl}
|
||||||
|
|
||||||
|
\section{Flow and verification}
|
||||||
|
The project is verified on two complementary planes: functional \textbf{simulation} with
|
||||||
|
Icarus Verilog (signed algebra, products, accumulation, groups, bias, ReLU, saturation,
|
||||||
|
busy/done signals) and real \textbf{implementation} with Yosys (synthesis) $+$
|
||||||
|
nextpnr-ecp5 (place\&route, timing) $+$ Project~Trellis (\code{ecppack}).
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{5.0cm} C{3.0cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Verification stage} & \thd{Outcome} & \thd{Covers} \\
|
||||||
|
\midrule
|
||||||
|
Functional RTL & \PASS & datapath correctness \\
|
||||||
|
\rowa Parametric simulation & \PASS & configuration sweep \\
|
||||||
|
ECP5 synthesis & \PASS & synthesizability, mapping \\
|
||||||
|
\rowa Placement / Routing & \PASS & LUT/FF/DSP, timing \\
|
||||||
|
Bitstream (\code{ecppack}) & \PASS & full flow, 0 errors (P2 and P8) \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\begin{fnnote}[End-to-end toolchain through the bitstream]
|
||||||
|
The full flow RTL $\to$ Yosys $\to$ nextpnr-ecp5 $\to$ \code{ecppack} produces a valid
|
||||||
|
bitstream for P2 and P8, \textbf{0 errors at every stage}. Header verified byte-by-byte:
|
||||||
|
\code{Part: LFE5U-45F-8CABGA381}, the target's real part number, not a placeholder. Only
|
||||||
|
\emph{generation} is verified: no physical-hardware test in this session.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\section{Datapath benchmark (256$\times$4)}
|
||||||
|
Configuration: INT8/INT32, \code{N\_INPUTS}=256, \code{N\_NEURONS}=4, variable
|
||||||
|
\code{PARALLEL}, 80~MHz target, device \code{LFE5U-45F-8BG381C} ($-8$). The test buses
|
||||||
|
are generated \emph{inside} the benchmark wrapper so as not to expose thousands of I/Os;
|
||||||
|
the top-level exposes only \code{clk/rst/start/y\_bus/busy/done}.
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{C{1.4cm} C{1.8cm} C{1.4cm} C{1.6cm} C{1.6cm} C{1.5cm} C{1.4cm}}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{PAR} & \thd{tot MAC} & \thd{DSP} & \thd{Fmax} & \thd{Tcrit} & \thd{80\,MHz} & \thd{LUT4} \\
|
||||||
|
\midrule
|
||||||
|
16 & 64 & 64/72 & 52.13 & 19.18 & \FAIL & $\approx$2531 \\
|
||||||
|
\rowa 8 & 32 & 32/72 & 61.71 & 16.20 & \FAIL & --- \\
|
||||||
|
4 & 16 & 16/72 & 75.01 & 13.33 & \FAIL & 804 \\
|
||||||
|
\rowa 2 & 8 & 8/72 & 87.88 & 11.38 & \PASS & 481 \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
\begin{center}\footnotesize\itshape\color{fnGrey}
|
||||||
|
Fmax and Tcrit in MHz and ns. Total MACs $=$ PARALLEL$\times$4 neurons.\end{center}
|
||||||
|
|
||||||
|
\subsection{Fmax and throughput versus parallelism}
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}
|
||||||
|
\begin{axis}[
|
||||||
|
width=0.62\textwidth,height=6.0cm,
|
||||||
|
axis y line*=left, axis x line=bottom,
|
||||||
|
xlabel={\footnotesize PARALLEL}, ylabel={\footnotesize Fmax [MHz]},
|
||||||
|
xtick={2,4,8,16}, xmode=log, log basis x=2,
|
||||||
|
ymin=40,ymax=95, ytick={40,55,70,85},
|
||||||
|
tick label style={font=\scriptsize}, label style={font=\footnotesize},
|
||||||
|
grid=major, grid style={fnRule!40},
|
||||||
|
legend style={font=\scriptsize,at={(0.5,-0.28)},anchor=north,legend columns=2}]
|
||||||
|
\addplot[fnTeal,mark=*,thick,mark options={fill=fnTeal}]
|
||||||
|
coordinates {(2,87.88)(4,75.01)(8,61.71)(16,52.13)};
|
||||||
|
\addlegendentry{Fmax}
|
||||||
|
\draw[fnAmber,dashed,thick] (axis cs:2,80)--(axis cs:16,80);
|
||||||
|
\node[font=\scriptsize,text=fnAmber] at (axis cs:11,82.5){80 MHz target};
|
||||||
|
\end{axis}
|
||||||
|
\begin{axis}[
|
||||||
|
width=0.62\textwidth,height=6.0cm,
|
||||||
|
axis y line*=right, axis x line=none,
|
||||||
|
xmode=log, log basis x=2, xmin=2,xmax=16,
|
||||||
|
ylabel={\footnotesize throughput [G\,MAC/s]},
|
||||||
|
ymin=0,ymax=3.6, ytick={0,1,2,3},
|
||||||
|
tick label style={font=\scriptsize}, label style={font=\footnotesize}]
|
||||||
|
\addplot[fnBlue,mark=square*,thick,mark options={fill=fnBlue}]
|
||||||
|
coordinates {(2,0.703)(4,1.20)(8,1.97)(16,3.34)};
|
||||||
|
\label{plt:tp}
|
||||||
|
\end{axis}
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
\begin{center}\footnotesize\itshape\color{fnGrey}
|
||||||
|
Fundamental trade-off: as PARALLEL grows, Fmax drops (deeper routing/tree) but the
|
||||||
|
theoretical throughput rises. The blue line (squares) is the throughput
|
||||||
|
$\approx$MAC/cycle$\times$Fmax.\end{center}
|
||||||
|
|
||||||
|
\subsection{Interpretation}
|
||||||
|
Reducing \code{PARALLEL} lowers simultaneous MACs, DSPs, adder-tree depth and routing
|
||||||
|
congestion, so Fmax rises; but the number of groups increases and hence the latency.
|
||||||
|
Frequency alone is not enough to choose: what matters is the overall throughput
|
||||||
|
$\approx$MAC/cycle$\times$frequency.
|
||||||
|
|
||||||
|
\begin{fnnote}[Architectural choices]
|
||||||
|
\code{PARALLEL=8} is the candidate for the throughput-oriented V1: exactly 32~simultaneous
|
||||||
|
MACs with 4 neurons, DSP at $\approx$44\%, leaving resources for controller, buffers, SPI
|
||||||
|
and future pipelines. \code{PARALLEL=2} is the frequency-oriented reference: 87.88~MHz,
|
||||||
|
the only one to exceed the 80~MHz target, but it requires 128 groups for a 256-input
|
||||||
|
neuron.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\subsection{Critical path and the 100~MHz limit}
|
||||||
|
The 100~MHz target is not met (best result 87.88~MHz with P2). The limit is
|
||||||
|
\emph{temporal}, not one of occupancy: with P2 the FPGA is barely used (DSP $\approx$11\%,
|
||||||
|
LUT $\approx$1\%). The critical path runs through weight FF $\to$ \code{MULT18X18D} $\to$
|
||||||
|
products $\to$ adder/carry $\to$ \code{acc\_next} $\to$ ReLU/saturation $\to$ output FF.
|
||||||
|
Exceeding 100~MHz will require one or more internal pipelines, not yet necessary to
|
||||||
|
proceed.
|
||||||
|
|
||||||
|
\section{Full integrated system}
|
||||||
|
Real synthesis of \code{spi\_neuron\_top} (SPI + arbiter + \code{neuron\_memory} +
|
||||||
|
\code{graph\_engine} + PSRAM chain), speed grade $-8$. Before timing closure the integrated
|
||||||
|
system missed the 80~MHz target (P2 $\approx$55~MHz, P8 $\approx$45~MHz), with a critical
|
||||||
|
path entirely inside \code{neuron\_parallel}.
|
||||||
|
|
||||||
|
\subsection{Cause: the saturation/ReLU carry chain}
|
||||||
|
Resource usage is not the cause (device below 10\% everywhere). The integrated system's
|
||||||
|
critical path is the \textbf{\code{CCU2C} carry chain of the saturation/ReLU comparator} in
|
||||||
|
\code{neuron\_parallel.v} --- \emph{not} SPI, arbiter, PSRAM, nor the Type~\#2 modules. The
|
||||||
|
saturation was written as an arithmetic comparison (\code{acc > 127}, \code{acc < -128}),
|
||||||
|
mapped by the synthesizer onto a 32-bit subtractor with a long carry chain.
|
||||||
|
|
||||||
|
\subsection{Timing closure (2026-09-03)}
|
||||||
|
An explicit waiver of the ``datapath untouchable'' rule for a separate timing-closure task,
|
||||||
|
with the single constraint of \textbf{bit-exact equivalence} across the whole regression.
|
||||||
|
Two steps:
|
||||||
|
\begin{itemize}
|
||||||
|
\item \textbf{Step 1 --- saturation/ReLU as bit-test.} A signed 32-bit value fits INT8 iff
|
||||||
|
\code{acc[31:7]} are all equal: an AND/OR reduction over a bit slice instead of a 32-bit
|
||||||
|
carry chain. A correct, bit-exact-verified simplification; real logic gain, but on its own
|
||||||
|
submerged by placement noise.
|
||||||
|
\item \textbf{Step 2 --- pipeline register} between accumulate and activation (\code{+1}
|
||||||
|
cycle of latency per neuron, absorbed by the \code{start}/\code{done} handshake, transparent
|
||||||
|
to callers). This is the decisive step.
|
||||||
|
\end{itemize}
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{4.6cm} C{2.6cm} C{2.4cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Config} & \thd{Before} & \thd{After} & \thd{$\Delta$} \\
|
||||||
|
\midrule
|
||||||
|
P2, real \code{.lpf} & 54.58 & \textbf{75.30} & $+38\%$ \\
|
||||||
|
\rowa P2, 5-seed sweep & 55.59 & 73.38--75.55 & robust \\
|
||||||
|
P8, unconstrained & 45.47 & \textbf{60.26} & $+33\%$ \\
|
||||||
|
\rowa P8, 5-seed sweep & 43.15--50.48 & 60.26--68.87 & non-overlapping \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
\begin{center}\footnotesize\itshape\color{fnGrey}
|
||||||
|
Fmax in MHz, real place\&route (\code{nextpnr-ecp5}). Robust across 5 seeds, not attributable
|
||||||
|
to placement luck.\end{center}
|
||||||
|
|
||||||
|
\begin{fnnote}[Stop criterion and real margin]
|
||||||
|
80~MHz is not reached (75.30~MHz at P2, 94\% of target) but the gain is large and real
|
||||||
|
($+38\%$/$+33\%$). The next step (the \code{MULT18X18D} output register, which would touch
|
||||||
|
\code{mac\_unit.v}) was left out: 80~MHz is \emph{headroom} toward the real \code{.lpf}, not
|
||||||
|
an operating requirement. With the planned 16~MHz oscillator, even the worst measured number
|
||||||
|
($\approx$45~MHz at P8) has $2.8\times$ of margin.
|
||||||
|
\textbf{Superseded 2026-09-04}: after adding the flash subsystem (ch.~\ref{ch:spi}
|
||||||
|
§\ref{sec:flashspi}, ch.~\ref{ch:roadmap}), Fmax for the full system (P2, same real pinout +
|
||||||
|
3 new flash signals) was 66.68~MHz, critical path still on the same
|
||||||
|
\code{neuron\_parallel} accumulator chain identified above --- not a new bottleneck, the
|
||||||
|
difference from 75.30~MHz was placement/routing noise from the added pins/logic.
|
||||||
|
\textbf{Updated again the same day (Phase F7)}: made the flash SPI bus genuinely
|
||||||
|
independent (dropped the \code{CCLK}/\code{USRMCLK} reuse, added a 4th ordinary
|
||||||
|
\code{flash\_sclk} pin), Fmax re-measured \textbf{67.91~MHz} (slight improvement, critical
|
||||||
|
path confirmed still identical). Margin on the 16~MHz oscillator: $4.2\times$.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\begin{fnnote}[Separate future optimization]
|
||||||
|
Independent of timing closure: the \code{x\_mem}/\code{w\_mem} arrays of \code{neuron\_memory}
|
||||||
|
are still inferred as distributed RAM on LUTs instead of \code{DP16KD}. Moving them to block
|
||||||
|
RAM would free LUTs and is a Phase~7 candidate --- but it was not on the critical path
|
||||||
|
resolved here.
|
||||||
|
\end{fnnote}
|
||||||
@@ -0,0 +1,102 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\gdef \LT@xxvi {\LT@entry
|
||||||
|
{1}{131.50148pt}\LT@entry
|
||||||
|
{1}{340.81447pt}}
|
||||||
|
\gdef \LT@xxvii {\LT@entry
|
||||||
|
{1}{397.71999pt}\LT@entry
|
||||||
|
{1}{74.59596pt}}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {12}Hardware design and pinout}{37}{chapter.12}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:hw}{{12}{37}{Hardware design and pinout}{chapter.12}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {12.1}Target device}{37}{section.12.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {12.2}Pin budget}{37}{section.12.2}\protected@file@percent }
|
||||||
|
\gdef \LT@xxviii {\LT@entry
|
||||||
|
{1}{97.35826pt}\LT@entry
|
||||||
|
{1}{40.45274pt}\LT@entry
|
||||||
|
{1}{66.06006pt}\LT@entry
|
||||||
|
{1}{40.45274pt}\LT@entry
|
||||||
|
{1}{248.5392pt}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {12.3}Signal map (top-level \texttt {spi\_neuron\_top}) --- real balls}{38}{section.12.3}\protected@file@percent }
|
||||||
|
\gdef \LT@xxix {\LT@entry
|
||||||
|
{1}{277.59943pt}\LT@entry
|
||||||
|
{1}{57.52458pt}\LT@entry
|
||||||
|
{1}{137.19194pt}}
|
||||||
|
\gdef \LT@xxx {\LT@entry
|
||||||
|
{1}{97.35826pt}\LT@entry
|
||||||
|
{1}{374.95769pt}}
|
||||||
|
\gdef \LT@xxxi {\LT@entry
|
||||||
|
{1}{114.43008pt}\LT@entry
|
||||||
|
{1}{97.35826pt}\LT@entry
|
||||||
|
{1}{260.5276pt}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {12.4}Per-bank allocation (real die geometry)}{40}{section.12.4}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {12.5}PSRAM subsystem}{40}{section.12.5}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {12.5.1}PSRAM connection (FPGA-exclusive)}{40}{subsection.12.5.1}\protected@file@percent }
|
||||||
|
\gdef \LT@xxxii {\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{68.9055pt}\LT@entry
|
||||||
|
{1}{294.67126pt}}
|
||||||
|
\gdef \LT@xxxiii {\LT@entry
|
||||||
|
{1}{97.35826pt}\LT@entry
|
||||||
|
{1}{74.59596pt}\LT@entry
|
||||||
|
{1}{300.36172pt}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {12.6}Clock}{41}{section.12.6}\protected@file@percent }
|
||||||
|
\newlabel{sec:clock}{{12.6}{41}{Clock}{section.12.6}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {12.7}Power}{41}{section.12.7}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {12.8}Configuration and programming}{41}{section.12.8}\protected@file@percent }
|
||||||
|
\newlabel{sec:config}{{12.8}{41}{Configuration and programming}{section.12.8}{}}
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {12.8.1}JTAG (development / debug)}{41}{subsection.12.8.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {12.8.2}Config-SPI to boot flash}{41}{subsection.12.8.2}\protected@file@percent }
|
||||||
|
\gdef \LT@xxxiv {\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{74.59596pt}\LT@entry
|
||||||
|
{1}{288.9808pt}}
|
||||||
|
\gdef \LT@xxxv {\LT@entry
|
||||||
|
{1}{125.81102pt}\LT@entry
|
||||||
|
{1}{125.81102pt}\LT@entry
|
||||||
|
{1}{220.69391pt}}
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {12.8.3}Configuration modes (\texttt {CFGMDN})}{42}{subsection.12.8.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {12.9}Open tasks before schematic capture}{42}{section.12.9}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/10-hardware}{
|
||||||
|
\setcounter{page}{44}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{12}
|
||||||
|
\setcounter{section}{9}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{10}
|
||||||
|
\setcounter{LT@tables}{35}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{70}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{21}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{4}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{-1}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{103}
|
||||||
|
\setcounter{lstlisting}{0}
|
||||||
|
}
|
||||||
@@ -0,0 +1,326 @@
|
|||||||
|
\chapter[Hardware design and pinout]{Hardware design and signal map}
|
||||||
|
\label{ch:hw}
|
||||||
|
|
||||||
|
\begin{fnnote}[Pinout status --- assigned and verified]
|
||||||
|
A real \code{.lpf} now exists (\code{synth/ecp5/spi\_neuron\_top.lpf}) with the top-level's
|
||||||
|
\textbf{57 signals} assigned to concrete CABGA381 balls, \textbf{verified by a full
|
||||||
|
0-error \code{nextpnr-ecp5} place\&route} (no longer \code{-{}-lpf-allow-unconstrained}). The
|
||||||
|
balls come from Project~Trellis's device database (\code{iodb.json}, the same nextpnr uses)
|
||||||
|
and were independently validated against §4.3.2 of the official Lattice datasheet (per-bank
|
||||||
|
GPIO counts: exact match on 6 of 7 banks, off by 1 ball on bank~3, immaterial since no
|
||||||
|
assigned signal uses it). \code{TRELLIS\_IO}: 57/245 (23\%). Current-build Fmax (full system
|
||||||
|
incl. flash subsystem with an independent SPI bus, Phase F7, 2026-09-04) \textbf{67.91~MHz},
|
||||||
|
critical path confirmed still on \code{neuron\_parallel}'s accumulator chain, unchanged from
|
||||||
|
earlier builds (ch.~\ref{ch:impl}). Pin-by-pin summary at the front of the document
|
||||||
|
(pp.~2--3). The boot config-SPI and JTAG balls do not appear here: they are dedicated
|
||||||
|
fixed-function pins with no corresponding RTL port, nextpnr never requires them (0 errors),
|
||||||
|
they matter only for the PCB schematic.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\section{Target device}
|
||||||
|
\begin{tabularx}{\textwidth}{L{4.2cm}Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Parameter} & \thd{Value} \\
|
||||||
|
\midrule
|
||||||
|
Device & Lattice ECP5 \code{LFE5U-45F-8BG381C} \\
|
||||||
|
\rowa Package & CABGA381 (381 balls) \\
|
||||||
|
Speed grade & $-8$ (the fastest of the ECP5 family) \\
|
||||||
|
\rowa Resources & $\approx$44k LUT/FF, 72$\times$\code{MULT18X18D}, \code{DP16KD} block RAM \\
|
||||||
|
Usable I/O & $\approx$232 balls out of 381 (rest: power/ground/NC) \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\section{Pin budget}
|
||||||
|
The project requires about 60 signals out of $\approx$232 usable I/Os: ample margin
|
||||||
|
($>$170 free pins), so the board is not pin-constrained.
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{Y C{2.2cm}}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Function} & \thd{Pins} \\
|
||||||
|
\midrule
|
||||||
|
PSRAM (22 address, 16 data, 6 control) & up to 44 \\
|
||||||
|
\rowa Application SPI (\code{sclk/mosi/miso/cs\_n}) & 4 \\
|
||||||
|
Clock, reset & 2 \\
|
||||||
|
\rowa Host attention pins (\code{irq\_n}, \code{data\_ready\_n}) & 2 \\
|
||||||
|
Flash runtime SPI bus (\code{flash\_sclk/flash\_mosi/flash\_miso/flash\_cs\_n}, ordinary GPIO, fully independent bus --- Phase F7) & 4 \\
|
||||||
|
\rowa JTAG (bring-up / debug, recommended) & 4 \\
|
||||||
|
\midrule
|
||||||
|
\rowh \thd{Total} & \thd{$\approx$60} \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\section{Signal map (top-level \texttt{spi\_neuron\_top}) --- real balls}
|
||||||
|
Real assignment of the top-level's 57 signals, verified by place\&route, \textbf{an
|
||||||
|
individual ball for every bit} (never a bus range). I/O standard: LVCMOS33 (3.3~V I/O
|
||||||
|
supply). The balls come from the real place\&route-verified \code{.lpf}. A compact summary
|
||||||
|
of the same table also appears at the front of the document (pp.~2--3).
|
||||||
|
|
||||||
|
\renewcommand{\arraystretch}{1.1}
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.0cm} C{1.0cm} C{1.9cm} C{1.0cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Signal} & \thd{Dir} & \thd{Ball} & \thd{Bank} & \thd{Function} \\
|
||||||
|
\midrule
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}Clock and reset (bank 7, left edge)}}\\
|
||||||
|
\code{clk} & IN & H5 & 7 & System clock on pad \code{GR\_PCLK7\_0} (dedicated global clock). \\
|
||||||
|
\rowa \code{rst} & IN & B4 & 7 & Global synchronous reset, active high. \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}Application SPI (bank 7, opposite the PSRAM bus)}}\\
|
||||||
|
\code{sclk} & IN & B5 & 7 & SPI clock (CPOL=0, CPHA=0). \\
|
||||||
|
\rowa \code{mosi} & IN & C5 & 7 & Master-Out Slave-In. \\
|
||||||
|
\code{miso} & OUT & A3 & 7 & Master-In Slave-Out (driven on the falling edge). \\
|
||||||
|
\rowa \code{cs\_n} & IN & B3 & 7 & Active-low chip-select. \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}Host attention pins (bank 7, active-low, level)}}\\
|
||||||
|
\code{data\_ready\_n} & OUT & C3 & 7 & Low while a result awaits reading (mirrors \code{STATUS.done}, clear on STATUS read). \\
|
||||||
|
\rowa \code{irq\_n} & OUT & C4 & 7 & Low if the graph load-time guard has tripped (mirrors \code{STATUS.err}); clears only on \code{RESET} or a fresh \code{run\_start}, \emph{not} on a STATUS read. \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}Flash subsystem --- independent SPI bus toward the onboard W25Q128JV (bank 7, Phases F1-F7)}}\\
|
||||||
|
\code{flash\_sclk} & OUT & E3 & 7 & SPI clock toward the flash --- ordinary GPIO, no config primitive involved (Phase F7). \\
|
||||||
|
\rowa \code{flash\_mosi} & OUT & D3 & 7 & Master-Out Slave-In toward the flash. \\
|
||||||
|
\code{flash\_miso} & IN & D5 & 7 & Master-In Slave-Out from the flash. \\
|
||||||
|
\rowa \code{flash\_cs\_n} & OUT & E4 & 7 & Flash chip-select, active low. \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}PSRAM address bus \code{psram\_a[21:0]} --- 22 individual balls (bank 2)}}\\
|
||||||
|
\code{psram\_a[0]} & OUT & E16 & 2 & PSRAM A0 \\
|
||||||
|
\rowa \code{psram\_a[1]} & OUT & F16 & 2 & PSRAM A1 \\
|
||||||
|
\code{psram\_a[2]} & OUT & D18 & 2 & PSRAM A2 \\
|
||||||
|
\rowa \code{psram\_a[3]} & OUT & E17 & 2 & PSRAM A3 \\
|
||||||
|
\code{psram\_a[4]} & OUT & E18 & 2 & PSRAM A4 \\
|
||||||
|
\rowa \code{psram\_a[5]} & OUT & F18 & 2 & PSRAM A5 \\
|
||||||
|
\code{psram\_a[6]} & OUT & F17 & 2 & PSRAM A6 \\
|
||||||
|
\rowa \code{psram\_a[7]} & OUT & G16 & 2 & PSRAM A7 \\
|
||||||
|
\code{psram\_a[8]} & OUT & G18 & 2 & PSRAM A8 \\
|
||||||
|
\rowa \code{psram\_a[9]} & OUT & H16 & 2 & PSRAM A9 \\
|
||||||
|
\code{psram\_a[10]} & OUT & H17 & 2 & PSRAM A10 \\
|
||||||
|
\rowa \code{psram\_a[11]} & OUT & H18 & 2 & PSRAM A11 \\
|
||||||
|
\code{psram\_a[12]} & OUT & J16 & 2 & PSRAM A12 \\
|
||||||
|
\rowa \code{psram\_a[13]} & OUT & J17 & 2 & PSRAM A13 \\
|
||||||
|
\code{psram\_a[14]} & OUT & C20 & 2 & PSRAM A14 \\
|
||||||
|
\rowa \code{psram\_a[15]} & OUT & D19 & 2 & PSRAM A15 \\
|
||||||
|
\code{psram\_a[16]} & OUT & E19 & 2 & PSRAM A16 \\
|
||||||
|
\rowa \code{psram\_a[17]} & OUT & E20 & 2 & PSRAM A17 \\
|
||||||
|
\code{psram\_a[18]} & OUT & F19 & 2 & PSRAM A18 \\
|
||||||
|
\rowa \code{psram\_a[19]} & OUT & F20 & 2 & PSRAM A19 \\
|
||||||
|
\code{psram\_a[20]} & OUT & G20 & 2 & PSRAM A20 \\
|
||||||
|
\rowa \code{psram\_a[21]} & OUT & H20 & 2 & PSRAM A21 \\
|
||||||
|
\code{psram\_a[22]} & OUT & P18 & 3 & Always 0 (byte$\to$word shift): NC on the board. \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}PSRAM data bus \code{psram\_dq[15:0]} --- 16 individual balls (banks 2 and 3)}}\\
|
||||||
|
\rowa \code{psram\_dq[0]} & IO & K18 & 2 & PSRAM DQ0 \\
|
||||||
|
\code{psram\_dq[1]} & IO & C18 & 2 & PSRAM DQ1 (dual-function ball, used as ordinary GPIO). \\
|
||||||
|
\rowa \code{psram\_dq[2]} & IO & D17 & 2 & PSRAM DQ2 \\
|
||||||
|
\code{psram\_dq[3]} & IO & D20 & 2 & PSRAM DQ3 \\
|
||||||
|
\rowa \code{psram\_dq[4]} & IO & G19 & 2 & PSRAM DQ4 \\
|
||||||
|
\code{psram\_dq[5]} & IO & J18 & 2 & PSRAM DQ5 \\
|
||||||
|
\rowa \code{psram\_dq[6]} & IO & J19 & 2 & PSRAM DQ6 \\
|
||||||
|
\code{psram\_dq[7]} & IO & J20 & 2 & PSRAM DQ7 \\
|
||||||
|
\rowa \code{psram\_dq[8]} & IO & K19 & 2 & PSRAM DQ8 \\
|
||||||
|
\code{psram\_dq[9]} & IO & K20 & 2 & PSRAM DQ9 \\
|
||||||
|
\rowa \code{psram\_dq[10]} & IO & L17 & 3 & PSRAM DQ10 \\
|
||||||
|
\code{psram\_dq[11]} & IO & M18 & 3 & PSRAM DQ11 \\
|
||||||
|
\rowa \code{psram\_dq[12]} & IO & M17 & 3 & PSRAM DQ12 \\
|
||||||
|
\code{psram\_dq[13]} & IO & N16 & 3 & PSRAM DQ13 \\
|
||||||
|
\rowa \code{psram\_dq[14]} & IO & N18 & 3 & PSRAM DQ14 \\
|
||||||
|
\code{psram\_dq[15]} & IO & P17 & 3 & PSRAM DQ15 (bidirectional tri-state data bus, \code{dq\_oe} = direction). \\
|
||||||
|
\multicolumn{5}{l}{\textit{\color{fnDark}PSRAM control (bank 3)}}\\
|
||||||
|
\rowa \code{psram\_ce\_n} & OUT & N17 & 3 & Chip enable, active low. \\
|
||||||
|
\code{psram\_oe\_n} & OUT & R16 & 3 & Output enable (read). \\
|
||||||
|
\rowa \code{psram\_we\_n} & OUT & R17 & 3 & Write enable (write). \\
|
||||||
|
\code{psram\_lb\_n} & OUT & T16 & 3 & Lower-byte enable (DQ[7:0]). \\
|
||||||
|
\rowa \code{psram\_ub\_n} & OUT & N19 & 3 & Upper-byte enable (DQ[15:8]). \\
|
||||||
|
\code{psram\_zz\_n} & OUT & N20 & 3 & Sleep/snooze (inactive=high in operation). \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
\renewcommand{\arraystretch}{1.25}
|
||||||
|
|
||||||
|
\begin{fnnote}[Board signals not exposed as RTL ports]
|
||||||
|
Not ports of \code{spi\_neuron\_top} but required at board level: the \textbf{configuration
|
||||||
|
SPI} lines to the onboard NOR flash (\code{PROGRAMN}/\code{INITN}/\code{DONE}/\code{CCLK}\ldots,
|
||||||
|
the datasheet's ``Miscellaneous Dedicated Pins'') and the 4 \textbf{JTAG} lines
|
||||||
|
(\code{TCK}/\code{TMS}/\code{TDI}/\code{TDO}), the \textbf{oscillator} on the \code{PCLK}
|
||||||
|
pad, the \textbf{power supplies}. Their ball numbers are not in the Lattice datasheet
|
||||||
|
(separate file) but are not needed here: dedicated pins with no RTL port, nextpnr never
|
||||||
|
requires them (0 errors), they matter only for the PCB schematic.
|
||||||
|
\end{fnnote}
|
||||||
|
|
||||||
|
\begin{fnwarn}[Application SPI separate from configuration SPI]
|
||||||
|
The application SPI (\code{sclk/mosi/miso/cs\_n}) must land on ordinary I/Os,
|
||||||
|
\textbf{never} on the configuration-SPI pins: the config-SPI clock pin is not reusable as
|
||||||
|
a general-purpose input after configuration without a board-level workaround. Keeping them
|
||||||
|
physically separate avoids that problem.
|
||||||
|
\end{fnwarn}
|
||||||
|
|
||||||
|
\section{Per-bank allocation (real die geometry)}
|
||||||
|
The placement follows the die-edge geometry (from Trellis's \code{globals.json},
|
||||||
|
ball~$\to$~(col,row)~$\to$~bank): banks \textbf{2 and 3} sit contiguously along the chip's
|
||||||
|
\textbf{right} edge and together hold the entire PSRAM bus (44+1 signals) --- exactly the
|
||||||
|
``one or two adjacent banks'' recommended. Bank \textbf{7} (\textbf{left} edge, physically
|
||||||
|
opposite the PSRAM bus) holds the application SPI and clock/reset, deliberately on the far
|
||||||
|
side so the two buses do not cross. \code{clk} is on the dedicated pad \code{H5}
|
||||||
|
(\code{GR\_PCLK7\_0}). Where a bank ran out of plain balls (part of \code{psram\_dq}), the
|
||||||
|
next dual-function ball was used as ordinary GPIO, confirmed usable by the real place\&route.
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{Y C{1.6cm} L{4.4cm}}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Signal group} & \thd{\# pins} & \thd{Bank (real)} \\
|
||||||
|
\midrule
|
||||||
|
PSRAM addresses \code{psram\_a[21:0]} & 22 & bank 2 (right edge) \\
|
||||||
|
\rowa PSRAM data \code{psram\_dq[15:0]} & 16 & banks 2 + 3 (adjacent) \\
|
||||||
|
PSRAM control (ce/oe/we/lb/ub/zz) & 6 & bank 3 \\
|
||||||
|
\rowa Application SPI & 4 & bank 7 (left edge) \\
|
||||||
|
Host attention pins (\code{irq\_n}, \code{data\_ready\_n}) & 2 & bank 7 \\
|
||||||
|
\rowa Independent flash SPI bus (\code{flash\_sclk/flash\_mosi/flash\_miso/flash\_cs\_n}) & 4 & bank 7 \\
|
||||||
|
Clock / reset & 2 & bank 7, \code{clk} on \code{GR\_PCLK7\_0} \\
|
||||||
|
\rowa Boot config SPI / JTAG & --- & dedicated pins (outside RTL, PCB only) \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\section{PSRAM subsystem}
|
||||||
|
The \code{psram\_controller.v} controller implements an \textbf{asynchronous parallel}
|
||||||
|
interface (address bus, 16-bit data, \code{ce\_n/oe\_n/we\_n} and byte-lanes
|
||||||
|
\code{lb\_n/ub\_n}, plus \code{zz\_n}) with an access latency of \textbf{70~ns} wired as
|
||||||
|
$\lceil 70\,\text{ns}\times f_{clk}\rceil$. It is an asynchronous-SRAM-style bus, not QSPI.
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.0cm}Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Role} & \thd{Component} \\
|
||||||
|
\midrule
|
||||||
|
Working memory & ISSI \code{IS66WVE4M16EBLL-70BLI} --- 64\,Mbit parallel PSRAM (4M$\times$16, 8~MB), async, 70~ns, an exact match to the controller timing. \\
|
||||||
|
\rowa Fallback & ISSI \code{IS61WV6416DBLL} / \code{IS61WV102416BLL} (true async SRAM, drop-in on the same signals, \code{zz\_n} inactive, $\sim$10~ns, lower density). \\
|
||||||
|
Persistent storage & Winbond \code{W25Q128JV} --- 16~MB SPI NOR flash for bitstream, weights, bias, network metadata. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\subsection{PSRAM connection (FPGA-exclusive)}
|
||||||
|
The PSRAM is driven \textbf{exclusively by the FPGA} through \code{psram\_controller.v}: no
|
||||||
|
external master touches the bus. The external host (RPi/ESP32/MCU) only speaks SPI to the
|
||||||
|
FPGA and never touches these lines. Pin-by-pin connection FPGA~$\leftrightarrow$~ISSI
|
||||||
|
\code{IS66WVE4M16EBLL-70BLI}:
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.6cm} L{3.0cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{FPGA signal} & \thd{PSRAM pin} & \thd{Function} \\
|
||||||
|
\midrule
|
||||||
|
\code{psram\_a[21:0]} & A0--A21 & Address bus (22 lines, 8~MB word address). \\
|
||||||
|
\rowa \code{psram\_dq[15:0]} & DQ0--DQ15 & Bidirectional data bus (tri-state, \code{dq\_oe}=direction). \\
|
||||||
|
\code{psram\_ce\_n} & CE\# & Chip enable (active low). \\
|
||||||
|
\rowa \code{psram\_oe\_n} & OE\# & Output enable (read). \\
|
||||||
|
\code{psram\_we\_n} & WE\# & Write enable (write). \\
|
||||||
|
\rowa \code{psram\_lb\_n} & LB\# & Lower-byte enable (DQ[7:0]). \\
|
||||||
|
\code{psram\_ub\_n} & UB\# & Upper-byte enable (DQ[15:8]). \\
|
||||||
|
\rowa \code{psram\_zz\_n} & ZZ\# & Sleep/snooze (held high in operation). \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
PSRAM supply: \textbf{3.3~V} (BLL variant), on the same I/O rail as banks 2/3 to which it is
|
||||||
|
wired (ch.~\ref{ch:hw}, real balls). Decoupling per supply pin per the ISSI datasheet.
|
||||||
|
|
||||||
|
\section{Clock}
|
||||||
|
\label{sec:clock}
|
||||||
|
There is no PLL in the RTL yet: \code{CLK\_FREQ\_MHZ} is a \emph{timing parameter} (it
|
||||||
|
feeds the PSRAM access formulas), not a clock generator. The mounted oscillator drives
|
||||||
|
\code{clk} directly. Recommendation: a 16~MHz MEMS oscillator (SiTime SiT2001B family),
|
||||||
|
well below the 67.91~MHz Fmax of the full integrated system (incl. flash subsystem,
|
||||||
|
ch.~\ref{ch:impl}). \code{CLK\_FREQ\_MHZ} must
|
||||||
|
be set to the real value of the mounted oscillator, otherwise the PSRAM timing comes out
|
||||||
|
wrong.
|
||||||
|
|
||||||
|
\section{Power}
|
||||||
|
A \textbf{three-rail} tree (the Lattice eval board's SERDES section is not needed and is
|
||||||
|
omitted: no 1.2~V \code{VCCA}/\code{VCCHTX}):
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.4cm} C{2.0cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Rail} & \thd{Voltage} & \thd{Feeds / regulator} \\
|
||||||
|
\midrule
|
||||||
|
\code{VCC} (core) & 1.1~V & FPGA core logic. Buck \code{TLV62568}, $\geq$600~mA. \\
|
||||||
|
\rowa \code{VCCIO0/2/3/6/7} & 3.3~V & I/O of all used banks + PSRAM. Buck \code{TLV62568}, 1~A. \\
|
||||||
|
\code{VCCAUX} & 2.5~V & FPGA auxiliary. LDO \code{TLV73325}, 10~mA. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
Decoupling: at least one capacitor per supply pin + bulk per rail, per the Lattice ECP5
|
||||||
|
hardware checklist. Input: external 12~V (or match the bucks to the source).
|
||||||
|
|
||||||
|
\section{Configuration and programming}
|
||||||
|
\label{sec:config}
|
||||||
|
Writing the FPGA ``map'' (bitstream) happens through dedicated silicon pins, \textbf{not}
|
||||||
|
RTL top-level ports. Default mode: \textbf{MSPI} --- automatic boot from the NOR flash at
|
||||||
|
power-on (standalone product); JTAG available for development.
|
||||||
|
|
||||||
|
\subsection{JTAG (development / debug)}
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.0cm} C{2.2cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Signal} & \thd{Ball\textsuperscript{$\dagger$}} & \thd{Function} \\
|
||||||
|
\midrule
|
||||||
|
\code{TCK} & T5 & Test clock. \\
|
||||||
|
\rowa \code{TDI} & R5 & Test data in. \\
|
||||||
|
\code{TDO} & V4 & Test data out. \\
|
||||||
|
\rowa \code{TMS} & U5 & Test mode select. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\subsection{Config-SPI to boot flash}
|
||||||
|
The FPGA loads the bitstream from the \textbf{Winbond \code{W25Q128JV}} (128~Mbit SPI NOR,
|
||||||
|
Quad read) at power-on. The flash subsystem (\code{rtl/flash\_slot\_manager.v}, Phases
|
||||||
|
F1-F7, ch.~\ref{ch:impl}) uses the \textbf{same physical flash} for network weights/bias/
|
||||||
|
metadata at runtime, FPGA-exclusive access: after configuration, the FPGA regains control
|
||||||
|
of the chip through a fully independent 4-wire SPI bus, \code{flash\_sclk/flash\_mosi/
|
||||||
|
flash\_miso/flash\_cs\_n} (all ordinary GPIO, pp.~2--3 and §``Signal map'' --- no ECP5
|
||||||
|
config primitive involved, Phase F7) --- this still implies a board-level dual connection
|
||||||
|
(the flash's DI/DO/CS/CLK pins wired both to the dedicated boot pins below and to these 4
|
||||||
|
ordinary balls, since it is the same physical chip serving both roles), not yet captured in
|
||||||
|
a schematic (none exists yet, see the checklist below).
|
||||||
|
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.4cm} C{2.2cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Signal} & \thd{Ball\textsuperscript{$\dagger$}} & \thd{Function} \\
|
||||||
|
\midrule
|
||||||
|
\code{CCLK/MCLK/SCK} & U3 & Configuration clock. \\
|
||||||
|
\rowa \code{DQ0\_MOSI} & W2 & Config data (MOSI). \\
|
||||||
|
\code{DQ1\_MISO} & V2 & Config data (MISO). \\
|
||||||
|
\rowa \code{BUSY\_CSSPIN} & R2 & Flash chip-select. \\
|
||||||
|
\code{DQ2 / DQ3} & Y2 / W1 & Quad-read lines. \\
|
||||||
|
\rowa \code{PROGRAMN} & W3 & Start reconfiguration (button, active low). \\
|
||||||
|
\code{INITN} & V3 & Init / configuration error (LED). \\
|
||||||
|
\rowa \code{DONE} & Y3 & Configuration complete (LED). \\
|
||||||
|
\code{CFGMDN[2:0]} & R4/T4/U4 & Mode select (see below). \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\subsection{Configuration modes (\texttt{CFGMDN})}
|
||||||
|
\begin{tabularx}{\textwidth}{L{4.0cm} C{4.0cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Mode} & \thd{CFGMDN[2:0]} & \thd{Use} \\
|
||||||
|
\midrule
|
||||||
|
MSPI (boot from flash) & \code{010} & \textbf{Default} --- standalone. \\
|
||||||
|
\rowa SSPI (slave SPI) & \code{001} & Config from external host. \\
|
||||||
|
SCM (slave serial) & \code{101} & Serial config. \\
|
||||||
|
\rowa SPCM (slave parallel) & \code{111} & 8-bit parallel config. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\begin{fnwarn}[Configuration balls to verify on the 45F]
|
||||||
|
\textsuperscript{$\dagger$}The JTAG and config-SPI balls listed here are the \emph{reference}
|
||||||
|
from the Lattice eval board (85F device). JTAG and config-SPI are dedicated, largely fixed
|
||||||
|
pins in the ECP5 family, but the exact positions on the \code{LFE5U-45F-8BG381C} target must
|
||||||
|
be confirmed against the Lattice 45F pinout file (Diamond/Radiant or the Trellis database)
|
||||||
|
before committing them to the schematic, as already done for the application signals
|
||||||
|
(ch.~\ref{ch:hw}).
|
||||||
|
\textbf{Distinct from this open item} (do not conflate the two): the flash subsystem's own
|
||||||
|
runtime SPI pins (\code{flash\_sclk}, \code{flash\_mosi}, \code{flash\_miso},
|
||||||
|
\code{flash\_cs\_n} --- Phases F1-F6, made fully independent in Phase F7) \textbf{are} real,
|
||||||
|
pinned, place\&route-verified ordinary GPIO on bank 7 --- \textbf{no pin shared with any
|
||||||
|
ECP5 config primitive}: an earlier version reused the boot \code{CCLK} pad for SCLK via
|
||||||
|
\code{USRMCLK}, dropped in Phase F7 (\code{USRMCLK} utilisation in the current full-system
|
||||||
|
synthesis is 0/1, confirming it is no longer used at all).
|
||||||
|
\end{fnwarn}
|
||||||
|
|
||||||
|
\section{Open tasks before schematic capture}
|
||||||
|
\begin{itemize}
|
||||||
|
\item[\OK] \code{ADDR\_WIDTH}=23 (full 8~MB) across all modules and testbenches.
|
||||||
|
\item[\OK] Real \code{.lpf} with the CABGA381 ball assignment, place\&route-verified with
|
||||||
|
0 errors (\code{synth/ecp5/spi\_neuron\_top.lpf}, 57 signals incl. flash subsystem).
|
||||||
|
\item[\OK] Boot/persistence flash subsystem (Phases F1-F7): SPI master, copy engine,
|
||||||
|
CRC32 slot catalog, fully independent 4-wire SPI bus, real synthesis at 0 errors, Fmax
|
||||||
|
67.91~MHz (\code{WORKLOG.md}).
|
||||||
|
\item[$\square$] Confirm PSRAM/SPI signal integrity at the actually mounted clock.
|
||||||
|
\item[$\square$] Board-level dual-wiring diagram for the flash's DI/DO/CS/CLK pins (dedicated
|
||||||
|
boot pins + the flash subsystem's 4 ordinary balls) --- not yet captured in a schematic.
|
||||||
|
\item[$\square$] Choice of the JTAG connector footprint.
|
||||||
|
\item[$\square$] Schematic capture (KiCad or other): no schematic exists yet for this
|
||||||
|
device/package combination.
|
||||||
|
\end{itemize}
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\gdef \LT@xxxvi {\LT@entry
|
||||||
|
{1}{51.83368pt}\LT@entry
|
||||||
|
{1}{103.04872pt}\LT@entry
|
||||||
|
{1}{80.28644pt}\LT@entry
|
||||||
|
{1}{237.14711pt}}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {13}Quick reference}{44}{chapter.13}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:ref}{{13}{44}{Quick reference}{chapter.13}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {13.1}SPI opcodes}{44}{section.13.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {13.2}STATUS byte}{44}{section.13.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {13.3}SET\_BASE selectors}{44}{section.13.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {13.4}Descriptor table (11 bytes/layer, MSB-first)}{44}{section.13.4}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {13.5}Build parameters}{44}{section.13.5}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/11-registri}{
|
||||||
|
\setcounter{page}{45}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{13}
|
||||||
|
\setcounter{section}{5}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{1}
|
||||||
|
\setcounter{LT@tables}{36}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{70}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{21}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{4}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{-1}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{109}
|
||||||
|
\setcounter{lstlisting}{0}
|
||||||
|
}
|
||||||
@@ -0,0 +1,80 @@
|
|||||||
|
\chapter{Quick reference}
|
||||||
|
\label{ch:ref}
|
||||||
|
|
||||||
|
\section{SPI opcodes}
|
||||||
|
\begin{tabularx}{\textwidth}{C{1.4cm} L{3.2cm} C{2.4cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Value} & \thd{Name} & \thd{Response} & \thd{Summary} \\
|
||||||
|
\midrule
|
||||||
|
\op{0x00} & NOP & --- & idle \\
|
||||||
|
\rowa \op{0x01} & WRITE\_RAM & --- & PSRAM block write \\
|
||||||
|
\op{0x02} & READ\_RAM & \code{len} B & PSRAM block read \\
|
||||||
|
\rowa \op{0x0F} & RESET & --- & engine reset + STATUS latch \\
|
||||||
|
\op{0x10} & SET\_BASE & --- & set base/register (sel 0..10) \\
|
||||||
|
\rowa \op{0x20} & START & --- & single-layer start \\
|
||||||
|
\op{0x21} & STATUS & 1 B & busy(live)/done(sticky) \\
|
||||||
|
\rowa \op{0x22} & READ\_OUTPUT & N\_NEURONS B & \code{y\_bus} \\
|
||||||
|
\op{0x23} & RUN\_NETWORK & --- & multi-layer start \\
|
||||||
|
\rowa \op{0x30} & READ\_CONFIG & 11 B & configuration record \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\section{STATUS byte}
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize]
|
||||||
|
\foreach \i/\lbl [count=\x from 0] in {7/0,6/0,5/0,4/0,3/0,2/0,1/{done},0/{busy}}{
|
||||||
|
\node[fnreg,minimum width=13mm,minimum height=9mm] (b\x) at (\x*13mm,0) {\lbl};
|
||||||
|
\node[font=\tiny,text=fnGrey,above=0.5mm of b\x] {bit \i};
|
||||||
|
}
|
||||||
|
\node[fill=fnAmber,text=white,rounded corners=1pt,inner sep=1.5pt,font=\tiny]
|
||||||
|
at (b7.center){reserved = 0};
|
||||||
|
\node[fill=fnTeal,text=white,rounded corners=1pt,inner sep=1.5pt,font=\tiny]
|
||||||
|
at (b6.center){};
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
\code{done} is sticky, clear-on-read; \code{busy} is live.
|
||||||
|
|
||||||
|
\section{SET\_BASE selectors}
|
||||||
|
\begin{multicols}{2}\footnotesize
|
||||||
|
\begin{itemize}
|
||||||
|
\item 0 --- \code{x\_base}
|
||||||
|
\item 1 --- \code{w\_base}
|
||||||
|
\item 2 --- \code{bias\_addr}
|
||||||
|
\item 3 --- \code{table\_base}
|
||||||
|
\item 4 --- \code{buf\_a\_base}
|
||||||
|
\columnbreak
|
||||||
|
\item 5 --- \code{buf\_b\_base}
|
||||||
|
\item 6 --- \code{activation} (single-layer)
|
||||||
|
\item 7 --- \code{n\_inputs\_real} (single-layer)
|
||||||
|
\item 8 --- \code{n\_neurons\_real} (single-layer)
|
||||||
|
\item 9 --- \code{num\_neurons\_graph} (Type \#2)
|
||||||
|
\item 10 --- \code{n\_out} (Type \#2)
|
||||||
|
\end{itemize}
|
||||||
|
\end{multicols}
|
||||||
|
|
||||||
|
\section{Descriptor table (11 bytes/layer, MSB-first)}
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}[font=\scriptsize,node distance=0mm]
|
||||||
|
\node[fnreg,minimum width=20mm,minimum height=8mm](a){\code{w\_base}\\3B};
|
||||||
|
\node[fnreg,minimum width=20mm,minimum height=8mm,right=0mm of a](b){\code{bias\_addr}\\3B};
|
||||||
|
\node[fnreg,minimum width=14mm,minimum height=8mm,right=0mm of b](c){\code{act}\\1B};
|
||||||
|
\node[fnreg,minimum width=22mm,minimum height=8mm,right=0mm of c](d){\code{n\_inputs\_real}\\2B};
|
||||||
|
\node[fnreg,minimum width=22mm,minimum height=8mm,right=0mm of d](e){\code{n\_neurons\_real}\\2B};
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
|
|
||||||
|
\section{Build parameters}
|
||||||
|
\begin{multicols}{2}\footnotesize
|
||||||
|
\begin{itemize}
|
||||||
|
\item \code{DATA\_WIDTH} --- 8 (INT8)
|
||||||
|
\item \code{ACC\_WIDTH} --- 32 (INT32)
|
||||||
|
\item \code{N\_INPUTS} --- max inputs
|
||||||
|
\item \code{N\_NEURONS} --- max neurons
|
||||||
|
\item \code{PARALLEL} --- simultaneous MACs
|
||||||
|
\columnbreak
|
||||||
|
\item \code{N\_LAYERS} --- max layers
|
||||||
|
\item \code{ADDR\_WIDTH} --- 23 (8 MB)
|
||||||
|
\item \code{MEM\_DATA\_WIDTH} --- 16
|
||||||
|
\item \code{CLK\_FREQ\_MHZ} --- PSRAM timing
|
||||||
|
\end{itemize}
|
||||||
|
\end{multicols}
|
||||||
@@ -0,0 +1,60 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\gdef \LT@xxxvii {\LT@entry
|
||||||
|
{1}{46.14322pt}\LT@entry
|
||||||
|
{1}{142.88284pt}\LT@entry
|
||||||
|
{1}{63.21504pt}\LT@entry
|
||||||
|
{1}{220.07484pt}}
|
||||||
|
\gdef \LT@xxxviii {\LT@entry
|
||||||
|
{1}{340.81447pt}\LT@entry
|
||||||
|
{1}{131.50148pt}}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {14}Roadmap and development status}{45}{chapter.14}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:roadmap}{{14}{45}{Roadmap and development status}{chapter.14}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {14.1}Development phases}{45}{section.14.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {14.2}Component status}{45}{section.14.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {14.3}Architectural principle (summary)}{46}{section.14.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {14.4}Long-term vision}{46}{section.14.4}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/12-roadmap}{
|
||||||
|
\setcounter{page}{47}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{14}
|
||||||
|
\setcounter{section}{4}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{2}
|
||||||
|
\setcounter{LT@tables}{38}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{72}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{21}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{4}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{-1}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{114}
|
||||||
|
\setcounter{lstlisting}{0}
|
||||||
|
}
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
\chapter{Roadmap and development status}
|
||||||
|
\label{ch:roadmap}
|
||||||
|
|
||||||
|
\section{Development phases}
|
||||||
|
\begin{tabularx}{\textwidth}{C{1.2cm} L{4.6cm} C{1.8cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Phase} & \thd{Title} & \thd{Status} & \thd{Content} \\
|
||||||
|
\midrule
|
||||||
|
1 & Parametric layer & \OK & inputs/neurons/parallelism, accumulation, bias, ReLU; test 32$\times$4/P=8. \\
|
||||||
|
\rowa 2 & Parameter sweep & \OK & multiple configurations incl. non-multiple and degenerate; elaboration guard added. \\
|
||||||
|
3 & Memory architecture & \OK & \code{neuron\_memory} single/multi-neuron, real PSRAM tested; multi-layer buffers $\to$ Phase~5. \\
|
||||||
|
\rowa 4 & SPI interface & \OK & \code{spi\_slave}+\code{spi\_engine}, 17 opcodes incl. flash subsystem, Fmax checked at full-system level. \\
|
||||||
|
5 & Multi-layer network & \OK$^\dagger$ & \code{layer\_sequencer}, configurable activations, runtime width; real toolchain checked. \\
|
||||||
|
\rowa 6 & Host software & planned & Linux and ESP32 drivers on the same protocol. \\
|
||||||
|
7 & Optimization & in progress & timing closure done (55$\to$75~MHz); PSRAM page-mode done (gather bandwidth +42\%); $x$/$w$ block RAM remains. \\
|
||||||
|
\rowa 8 & Hardware training (opt.) & future & backprop, gradients, weight update. \\
|
||||||
|
9 & Flash subsystem (F1-F7) & \OK & dedicated SPI master, flash$\leftrightarrow$PSRAM copy engine, 16-slot catalog with CRC32, fully independent 4-wire SPI bus (F7), 8 opcodes (\op{0x40}--\op{0x47}, ch.~\ref{ch:spi} §\ref{sec:flashspi}); real synthesis 0 errors. \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
\begin{center}\footnotesize\itshape\color{fnGrey}
|
||||||
|
$\dagger$ RTL, unit tests and end-to-end over simulated SPI complete; timing closure done:
|
||||||
|
75.30~MHz (P2) / 60.26~MHz (P8) at the time of Phase~5, bit-exact across the whole
|
||||||
|
regression; Fmax of the full system after Phase~9 (incl. independent flash subsystem):
|
||||||
|
\textbf{67.91~MHz} (ch.~\ref{ch:impl}).\end{center}
|
||||||
|
|
||||||
|
\section{Component status}
|
||||||
|
\begin{tabularx}{\textwidth}{Y C{4.2cm}}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Component} & \thd{Status} \\
|
||||||
|
\midrule
|
||||||
|
Parametric neural layer & \OK{} working \\
|
||||||
|
\rowa Parametric inputs/neurons/parallelism & \OK \\
|
||||||
|
Accumulation, bias, ReLU & \OK \\
|
||||||
|
\rowa 32$\times$4 / P=8 validation & \OK \\
|
||||||
|
Dedicated RAM (interface + controller + INT8 access) & \OK{} tested on real PSRAM \\
|
||||||
|
\rowa SPI interface (17 opcodes incl. RUN\_NETWORK + flash) & \OK{} Fmax at full-system level \\
|
||||||
|
Dual SPI & future \\
|
||||||
|
\rowa Multi-layer engine & \OK{} timing closure 75.30~MHz (P2) at the time of Phase~5 \\
|
||||||
|
Configurable activations (ACT\_NONE/ACT\_RELU) & \OK \\
|
||||||
|
\rowa Runtime network width (one bitstream, any topology) & \OK{} measured savings \\
|
||||||
|
Type \#2 graph network (act\_buffer, graph\_engine, netasm) & \OK{} RTL + tests + synthesis \\
|
||||||
|
\rowa CABGA381 pinout (real \code{.lpf}, 57 signals incl. flash) & \OK{} place\&route-verified, 0 errors \\
|
||||||
|
PSRAM page-mode (G7) & \OK{} done (37.53 cycles/edge, bandwidth +42\%) \\
|
||||||
|
\rowa Flash subsystem (SPI master, copy engine, CRC32 catalog, independent bus F7) & \OK{} real synthesis 0 errors, Fmax 67.91~MHz \\
|
||||||
|
Real bitstream (\code{ecppack}, P2/P8) & \OK{} 0 errors, part LFE5U-45F-8CABGA381 \\
|
||||||
|
\rowa Linux / ESP32 host driver & planned \\
|
||||||
|
Hardware training & future \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\section{Architectural principle (summary)}
|
||||||
|
\begin{fnspec}[Foundation of the project]
|
||||||
|
The FPGA implements the neural machine and owns its own RAM; the host configures and uses
|
||||||
|
the machine. A build fixes the \emph{ceiling} (max layers, max width, PARALLEL); the host
|
||||||
|
configures the \emph{actual} network --- number of layers, per-layer width, per-layer
|
||||||
|
activation, trained parameters --- entirely at runtime, over SPI, into the FPGA's local
|
||||||
|
memory. A single bitstream serves any topology up to that ceiling.
|
||||||
|
\end{fnspec}
|
||||||
|
|
||||||
|
\section{Long-term vision}
|
||||||
|
The final goal is a reusable hardware block integrable into different future projects:
|
||||||
|
the host platform can change (Linux, ESP32, MCU, PC) without changing the fundamental
|
||||||
|
architecture of the engine. The FPGA becomes a dedicated neural computation peripheral,
|
||||||
|
optimized for the topology required by each application.
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
\relax
|
||||||
|
\providecommand\hyper@newdestlabel[2]{}
|
||||||
|
\gdef \LT@xxxix {\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{68.9055pt}\LT@entry
|
||||||
|
{1}{294.67126pt}}
|
||||||
|
\gdef \LT@xl {\LT@entry
|
||||||
|
{1}{114.43008pt}\LT@entry
|
||||||
|
{1}{108.73918pt}\LT@entry
|
||||||
|
{1}{249.14668pt}}
|
||||||
|
\@writefile{toc}{\contentsline {chapter}{\numberline {A}Modules and toolchain}{47}{appendix.A}\protected@file@percent }
|
||||||
|
\@writefile{lof}{\addvspace {10\p@ }}
|
||||||
|
\@writefile{lot}{\addvspace {10\p@ }}
|
||||||
|
\newlabel{ch:appmod}{{A}{47}{Modules and toolchain}{appendix.A}{}}
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {A.1}List of RTL modules}{47}{section.A.1}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {A.2}Ports of the top-level \texttt {spi\_neuron\_top}}{47}{section.A.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {A.3}Toolchain}{47}{section.A.3}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {A.3.1}Main nextpnr parameters}{47}{subsection.A.3.1}\protected@file@percent }
|
||||||
|
\gdef \LT@xli {\LT@entry
|
||||||
|
{1}{165.6447pt}\LT@entry
|
||||||
|
{1}{306.67125pt}}
|
||||||
|
\@writefile{toc}{\contentsline {subsection}{\numberline {A.3.2}Simulation example}{48}{subsection.A.3.2}\protected@file@percent }
|
||||||
|
\@writefile{toc}{\contentsline {section}{\numberline {A.4}Main testbenches}{48}{section.A.4}\protected@file@percent }
|
||||||
|
\@setckpt{chapters/A-moduli}{
|
||||||
|
\setcounter{page}{49}
|
||||||
|
\setcounter{equation}{0}
|
||||||
|
\setcounter{enumi}{0}
|
||||||
|
\setcounter{enumii}{0}
|
||||||
|
\setcounter{enumiii}{0}
|
||||||
|
\setcounter{enumiv}{0}
|
||||||
|
\setcounter{footnote}{0}
|
||||||
|
\setcounter{mpfootnote}{0}
|
||||||
|
\setcounter{part}{0}
|
||||||
|
\setcounter{chapter}{1}
|
||||||
|
\setcounter{section}{4}
|
||||||
|
\setcounter{subsection}{0}
|
||||||
|
\setcounter{subsubsection}{0}
|
||||||
|
\setcounter{paragraph}{0}
|
||||||
|
\setcounter{subparagraph}{0}
|
||||||
|
\setcounter{figure}{0}
|
||||||
|
\setcounter{table}{3}
|
||||||
|
\setcounter{LT@tables}{41}
|
||||||
|
\setcounter{LT@chunks}{1}
|
||||||
|
\setcounter{parentequation}{0}
|
||||||
|
\setcounter{tcbbreakpart}{1}
|
||||||
|
\setcounter{tcblayer}{0}
|
||||||
|
\setcounter{tcolorbox@number}{72}
|
||||||
|
\setcounter{tcbrastercolumn}{1}
|
||||||
|
\setcounter{tcbrasterrow}{1}
|
||||||
|
\setcounter{tcbrasternum}{1}
|
||||||
|
\setcounter{tcbraster}{0}
|
||||||
|
\setcounter{lstnumber}{5}
|
||||||
|
\setcounter{tcblisting}{0}
|
||||||
|
\setcounter{caption@flags}{0}
|
||||||
|
\setcounter{continuedfloat}{0}
|
||||||
|
\setcounter{tikztiming@nrows}{4}
|
||||||
|
\setcounter{tikztimingrows}{0}
|
||||||
|
\setcounter{tikztimingtrans}{-1}
|
||||||
|
\setcounter{tikztimingtranspos}{0}
|
||||||
|
\setcounter{section@level}{0}
|
||||||
|
\setcounter{Item}{0}
|
||||||
|
\setcounter{Hfootnote}{0}
|
||||||
|
\setcounter{bookmark@seq@number}{121}
|
||||||
|
\setcounter{lstlisting}{0}
|
||||||
|
}
|
||||||
@@ -0,0 +1,96 @@
|
|||||||
|
\chapter[Modules and toolchain]{Modules, ports and toolchain}
|
||||||
|
\label{ch:appmod}
|
||||||
|
|
||||||
|
\section{List of RTL modules}
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.4cm} C{2.0cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{File} & \thd{Type} & \thd{Role} \\
|
||||||
|
\midrule
|
||||||
|
\code{rtl/mac\_unit.v} & combinational & single multiply-accumulator \\
|
||||||
|
\rowa \code{rtl/mac8.v} & combinational & parallel MAC + balanced adder tree \\
|
||||||
|
\code{rtl/neuron\_parallel.v} & FSM & neuron: groups, bias, activation, saturation \\
|
||||||
|
\rowa \code{rtl/layer.v} & structural & N\_NEURONS neurons in parallel \\
|
||||||
|
\code{rtl/neuron\_memory.v} & FSM & memory/neuron bridge, neuron loop \\
|
||||||
|
\rowa \code{rtl/layer\_sequencer.v} & FSM & multi-layer sequencing, ping-pong \\
|
||||||
|
\code{rtl/int8\_memory\_access.v} & FSM & byte $\leftrightarrow$ word conversion \\
|
||||||
|
\rowa \code{rtl/memory\_interface.v} & FSM & req/ready handshake \\
|
||||||
|
\code{rtl/psram\_controller.v} & FSM & async 70~ns physical PSRAM bus \\
|
||||||
|
\rowa \code{rtl/mem\_arbiter.v} & arbiter & 3 ports, priority B$>$C$>$A \\
|
||||||
|
\code{rtl/spi\_slave.v} & FSM & SPI Mode 0 physical layer + CDC \\
|
||||||
|
\rowa \code{rtl/spi\_engine.v} & FSM & opcode + register bank \\
|
||||||
|
\code{rtl/act\_buffer.v} & block RAM & DP16KD activation buffer (Type \#2) \\
|
||||||
|
\rowa \code{rtl/graph\_engine.v} & FSM & graph-network engine (Type \#2) \\
|
||||||
|
\code{rtl/spi\_neuron\_top.v} & top & full integration \\
|
||||||
|
\rowa \code{rtl/memory\_model.v} & model & behavioral RAM (sim) \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\section{Ports of the top-level \texttt{spi\_neuron\_top}}
|
||||||
|
See the complete signal-by-signal table in ch.~\ref{ch:hw}. In summary: clock and reset
|
||||||
|
(\code{clk}, \code{rst}); application SPI (\code{sclk}, \code{mosi}, \code{miso},
|
||||||
|
\code{cs\_n}); PSRAM bus (\code{psram\_a[22:0]}, \code{psram\_dq[15:0]},
|
||||||
|
\code{psram\_ce\_n/oe\_n/we\_n/lb\_n/ub\_n/zz\_n}).
|
||||||
|
|
||||||
|
\section{Toolchain}
|
||||||
|
\begin{tabularx}{\textwidth}{L{3.6cm} L{3.4cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Tool} & \thd{Version} & \thd{Use} \\
|
||||||
|
\midrule
|
||||||
|
Yosys & 0.68+post & RTL synthesis $\to$ JSON netlist, ECP5 mapping \\
|
||||||
|
\rowa nextpnr-ecp5 & 0.11.1-19-g8dbcee5 & placement, routing, timing \\
|
||||||
|
Project Trellis & install & \code{ecppack}/\code{ecppll}/\code{ecpbram} \\
|
||||||
|
\rowa Icarus Verilog & \code{-g2012} & functional simulation \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\subsection{Main nextpnr parameters}
|
||||||
|
\begin{lstlisting}[language=,basicstyle=\ttfamily\scriptsize]
|
||||||
|
--45k selects LFE5U-45F
|
||||||
|
--package CABGA381 package
|
||||||
|
--speed 8 speed grade -8
|
||||||
|
--json <netlist> netlist from Yosys
|
||||||
|
--lpf <constraints> pin constraints (currently empty)
|
||||||
|
--lpf-allow-unconstrained allows unconstrained I/Os (benchmark)
|
||||||
|
--freq 80 80 MHz timing target
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\subsection{Simulation example}
|
||||||
|
\begin{lstlisting}[language=,basicstyle=\ttfamily\scriptsize]
|
||||||
|
iverilog -g2012 -Ptb.PARALLEL=16 -o sim/parametric_256x4_p16 \
|
||||||
|
sim/parametric_tb.v rtl/mac_unit.v rtl/mac8.v \
|
||||||
|
rtl/neuron_parallel.v rtl/layer.v
|
||||||
|
vvp sim/parametric_256x4_p16
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\section{Main testbenches}
|
||||||
|
\begin{tabularx}{\textwidth}{L{5.4cm} Y}
|
||||||
|
\toprule
|
||||||
|
\rowh \thd{Testbench} & \thd{Coverage} \\
|
||||||
|
\midrule
|
||||||
|
\code{parametric\_tb.v} & 256$\times$4 datapath, accumulate/bias/ReLU/saturation cases \\
|
||||||
|
\rowa \code{parameter\_sweep\_tb.v} & sweep of valid configurations \\
|
||||||
|
\code{neuron\_parallel\_tb.v} & activations, runtime width (T7) \\
|
||||||
|
\rowa \code{neuron\_memory\_tb.v} / \code{\_multi\_tb.v} & single/multi-neuron memory integration, real PSRAM (T5) \\
|
||||||
|
\code{psram\_controller\_tb.v} & PSRAM controller \\
|
||||||
|
\rowa \code{psram\_page\_mode\_tb.v} & page bursts, page crossing, close on WRITE/$t_{CEM}$ timeout, byte-enable changes (§~5.5) \\
|
||||||
|
\code{spi\_slave\_tb.v} & SPI physical layer (4 tests) \\
|
||||||
|
\rowa \code{spi\_engine\_tb.v} & opcodes, registers (10+ tests) \\
|
||||||
|
\code{spi\_neuron\_top\_tb.v} & end-to-end, real PSRAM over simulated SPI \\
|
||||||
|
\rowa \code{spi\_neuron\_top\_runnetwork\_tb.v} & RUN\_NETWORK 2-layer end-to-end \\
|
||||||
|
\code{layer\_sequencer\_tb.v} & 2-layer sequence, ping-pong, byte-exact copy \\
|
||||||
|
\bottomrule
|
||||||
|
\end{tabularx}
|
||||||
|
|
||||||
|
\vfill
|
||||||
|
\begin{center}
|
||||||
|
\begin{tikzpicture}
|
||||||
|
\node[draw=fnRule,rounded corners=3pt,inner sep=8pt,fill=fnLight,text width=15.5cm]{
|
||||||
|
\footnotesize\color{fnGrey}
|
||||||
|
This datasheet is generated from the RTL code, the documentation and the benchmarks
|
||||||
|
present in the repository \texttt{github.com/manvalan/FPGA-Neural} as of \datasheetdate.
|
||||||
|
The Fmax, resource usage and throughput values are those reported in the repository
|
||||||
|
measurements (real \texttt{.lpf} already assigned and place\&route-verified,
|
||||||
|
ch.~\ref{ch:hw}) and must be re-verified on any substantial RTL change or as the
|
||||||
|
Phase~7 timing closure, still in progress, continues (ch.~\ref{ch:roadmap}).};
|
||||||
|
\end{tikzpicture}
|
||||||
|
\end{center}
|
||||||
@@ -0,0 +1,184 @@
|
|||||||
|
% ======================================================================
|
||||||
|
% FPGA-Neural Datasheet -- preamble / stile
|
||||||
|
% ======================================================================
|
||||||
|
\usepackage[T1]{fontenc}
|
||||||
|
\usepackage[utf8]{inputenc}
|
||||||
|
\usepackage[english]{babel}
|
||||||
|
\usepackage{helvet}
|
||||||
|
\renewcommand{\familydefault}{\sfdefault}
|
||||||
|
\usepackage{courier}
|
||||||
|
\usepackage{microtype}
|
||||||
|
|
||||||
|
\usepackage[a4paper,top=2.4cm,bottom=2.3cm,left=2.2cm,right=2.2cm,headheight=15pt]{geometry}
|
||||||
|
\usepackage[table]{xcolor}
|
||||||
|
\usepackage{graphicx}
|
||||||
|
\usepackage{booktabs}
|
||||||
|
\usepackage{tabularx}
|
||||||
|
\usepackage{longtable}
|
||||||
|
\usepackage{array}
|
||||||
|
\usepackage{ltablex}
|
||||||
|
\keepXColumns
|
||||||
|
\usepackage{multirow}
|
||||||
|
\usepackage{multicol}
|
||||||
|
\usepackage{enumitem}
|
||||||
|
\usepackage{amsmath}
|
||||||
|
\usepackage{amssymb}
|
||||||
|
\usepackage{ragged2e}
|
||||||
|
|
||||||
|
% ---------- Palette ----------------------------------------------------
|
||||||
|
\definecolor{fnDark}{HTML}{0B2E4F} % blu profondo (primario)
|
||||||
|
\definecolor{fnBlue}{HTML}{15629B} % blu medio
|
||||||
|
\definecolor{fnTeal}{HTML}{0E8F8A} % accento teal
|
||||||
|
\definecolor{fnAmber}{HTML}{C9761B} % accento ambra
|
||||||
|
\definecolor{fnRed}{HTML}{B22C34} % fail / warning
|
||||||
|
\definecolor{fnGreen}{HTML}{2E7D32} % pass / ok
|
||||||
|
\definecolor{fnGrey}{HTML}{5B6B78}
|
||||||
|
\definecolor{fnLight}{HTML}{EEF3F7} % sfondo chiaro
|
||||||
|
\definecolor{fnLight2}{HTML}{E2ECF3}
|
||||||
|
\definecolor{fnRule}{HTML}{9FB4C4}
|
||||||
|
\definecolor{codebg}{HTML}{F5F7F9}
|
||||||
|
\definecolor{codekw}{HTML}{15629B}
|
||||||
|
\definecolor{codecom}{HTML}{5B6B78}
|
||||||
|
\definecolor{codestr}{HTML}{0E8F8A}
|
||||||
|
|
||||||
|
% ---------- Titoli -----------------------------------------------------
|
||||||
|
\usepackage{titlesec}
|
||||||
|
\titleformat{\chapter}[display]
|
||||||
|
{\normalfont\bfseries\color{fnDark}}
|
||||||
|
{\filright\Large\color{fnTeal}CHAPTER \thechapter}
|
||||||
|
{6pt}
|
||||||
|
{\Huge\filright}
|
||||||
|
[\vspace{2pt}{\color{fnRule}\titlerule[1.3pt]}]
|
||||||
|
\titlespacing*{\chapter}{0pt}{6pt}{18pt}
|
||||||
|
|
||||||
|
\titleformat{\section}
|
||||||
|
{\normalfont\large\bfseries\color{fnDark}}{\thesection}{0.6em}{}
|
||||||
|
\titleformat{\subsection}
|
||||||
|
{\normalfont\bfseries\color{fnBlue}}{\thesubsection}{0.6em}{}
|
||||||
|
\titleformat{\subsubsection}
|
||||||
|
{\normalfont\bfseries\color{fnGrey}}{\thesubsubsection}{0.6em}{}
|
||||||
|
\titlespacing*{\section}{0pt}{12pt}{4pt}
|
||||||
|
|
||||||
|
% ---------- Header / footer -------------------------------------------
|
||||||
|
\usepackage{fancyhdr}
|
||||||
|
\pagestyle{fancy}
|
||||||
|
\fancyhf{}
|
||||||
|
\renewcommand{\headrulewidth}{0.6pt}
|
||||||
|
\renewcommand{\footrulewidth}{0.4pt}
|
||||||
|
\renewcommand{\headrule}{\hbox to\headwidth{\color{fnRule}\leaders\hrule height \headrulewidth\hfill}}
|
||||||
|
\renewcommand{\footrule}{\hbox to\headwidth{\color{fnRule}\leaders\hrule height \footrulewidth\hfill}}
|
||||||
|
\renewcommand{\chaptermark}[1]{\markboth{#1}{}}
|
||||||
|
\fancyhead[L]{\small\color{fnDark}\textbf{FPGA-Neural}}
|
||||||
|
\fancyhead[R]{\footnotesize\color{fnGrey}\nouppercase{\leftmark}}
|
||||||
|
\fancyfoot[L]{\small\color{fnGrey}Datasheet~\textbf{Rev.\,\datasheetrev}}
|
||||||
|
\fancyfoot[C]{\small\color{fnGrey}manvalan/FPGA-Neural}
|
||||||
|
\fancyfoot[R]{\small\color{fnGrey}\thepage}
|
||||||
|
\fancypagestyle{plain}{\fancyhf{}%
|
||||||
|
\fancyfoot[L]{\small\color{fnGrey}Datasheet~\textbf{Rev.\,\datasheetrev}}%
|
||||||
|
\fancyfoot[C]{\small\color{fnGrey}manvalan/FPGA-Neural}%
|
||||||
|
\fancyfoot[R]{\small\color{fnGrey}\thepage}%
|
||||||
|
\renewcommand{\headrulewidth}{0pt}}
|
||||||
|
|
||||||
|
% ---------- tcolorbox --------------------------------------------------
|
||||||
|
\usepackage[most]{tcolorbox}
|
||||||
|
\tcbuselibrary{skins,breakable}
|
||||||
|
|
||||||
|
% Box "nota"
|
||||||
|
\newtcolorbox{fnnote}[1][Note]{
|
||||||
|
enhanced, breakable, colback=fnLight, colframe=fnTeal,
|
||||||
|
boxrule=0.4pt, left=8pt, right=8pt, top=5pt, bottom=5pt,
|
||||||
|
fonttitle=\bfseries\color{white}, coltitle=white,
|
||||||
|
attach boxed title to top left={xshift=6pt,yshift=-3pt},
|
||||||
|
boxed title style={colback=fnTeal,boxrule=0pt,arc=1pt}, title={#1}}
|
||||||
|
|
||||||
|
% Box "attenzione"
|
||||||
|
\newtcolorbox{fnwarn}[1][Warning]{
|
||||||
|
enhanced, breakable, colback=fnLight, colframe=fnAmber,
|
||||||
|
boxrule=0.4pt, left=8pt, right=8pt, top=5pt, bottom=5pt,
|
||||||
|
fonttitle=\bfseries\color{white}, coltitle=white,
|
||||||
|
attach boxed title to top left={xshift=6pt,yshift=-3pt},
|
||||||
|
boxed title style={colback=fnAmber,boxrule=0pt,arc=1pt}, title={#1}}
|
||||||
|
|
||||||
|
% Box "registro/parametro"
|
||||||
|
\newtcolorbox{fnspec}[1][Specification]{
|
||||||
|
enhanced, breakable, colback=white, colframe=fnBlue,
|
||||||
|
boxrule=0.7pt, left=8pt, right=8pt, top=5pt, bottom=5pt, arc=1.5pt,
|
||||||
|
fonttitle=\bfseries\color{white}, coltitle=white,
|
||||||
|
attach boxed title to top left={xshift=6pt,yshift=-3pt},
|
||||||
|
boxed title style={colback=fnBlue,boxrule=0pt,arc=1pt}, title={#1}}
|
||||||
|
|
||||||
|
% ---------- listings (Verilog) ----------------------------------------
|
||||||
|
\usepackage{listings}
|
||||||
|
\lstdefinestyle{verilog}{
|
||||||
|
language=Verilog,
|
||||||
|
backgroundcolor=\color{codebg},
|
||||||
|
basicstyle=\ttfamily\scriptsize,
|
||||||
|
keywordstyle=\color{codekw}\bfseries,
|
||||||
|
commentstyle=\color{codecom}\itshape,
|
||||||
|
stringstyle=\color{codestr},
|
||||||
|
numbers=left, numberstyle=\tiny\color{fnGrey}, numbersep=7pt,
|
||||||
|
showstringspaces=false, breaklines=true, frame=leftline,
|
||||||
|
framerule=1.2pt, rulecolor=\color{fnTeal},
|
||||||
|
xleftmargin=12pt, framexleftmargin=10pt, tabsize=2,
|
||||||
|
morekeywords={logic,always_ff,always_comb,localparam,signed,genvar,generate,endgenerate}
|
||||||
|
}
|
||||||
|
\lstset{style=verilog}
|
||||||
|
|
||||||
|
% ---------- Tabelle ----------------------------------------------------
|
||||||
|
\newcolumntype{L}[1]{>{\raggedright\arraybackslash}p{#1}}
|
||||||
|
\newcolumntype{C}[1]{>{\centering\arraybackslash}p{#1}}
|
||||||
|
\newcolumntype{R}[1]{>{\raggedleft\arraybackslash}p{#1}}
|
||||||
|
\newcolumntype{Y}{>{\raggedright\arraybackslash}X}
|
||||||
|
\renewcommand{\arraystretch}{1.25}
|
||||||
|
\arrayrulecolor{fnRule}
|
||||||
|
|
||||||
|
% intestazione tabella colorata
|
||||||
|
\newcommand{\thd}[1]{\textbf{\color{white}#1}}
|
||||||
|
\newcommand{\rowh}{\rowcolor{fnDark}}
|
||||||
|
\newcommand{\rowa}{\rowcolor{fnLight}}
|
||||||
|
|
||||||
|
% ---------- Caption ----------------------------------------------------
|
||||||
|
\usepackage{caption}
|
||||||
|
\captionsetup{font=small,labelfont={bf,color=fnTeal},labelsep=period}
|
||||||
|
|
||||||
|
% ---------- TikZ / pgfplots -------------------------------------------
|
||||||
|
\usepackage{tikz}
|
||||||
|
\usetikzlibrary{arrows.meta,positioning,calc,shapes.geometric,shapes.misc,
|
||||||
|
fit,backgrounds,chains,decorations.pathreplacing,decorations.markings,
|
||||||
|
matrix,shadows.blur}
|
||||||
|
\usepackage{pgfplots}
|
||||||
|
\pgfplotsset{compat=1.17}
|
||||||
|
\usepackage{tikz-timing}
|
||||||
|
|
||||||
|
% stili di blocco riusabili
|
||||||
|
\tikzset{
|
||||||
|
fnblock/.style={draw=fnBlue,fill=fnLight,rounded corners=2pt,
|
||||||
|
minimum height=9mm,minimum width=24mm,align=center,font=\small,
|
||||||
|
inner sep=4pt,line width=0.7pt},
|
||||||
|
fnblockT/.style={fnblock,draw=fnTeal,fill=fnLight2},
|
||||||
|
fnblockD/.style={fnblock,draw=fnDark,fill=fnDark,text=white},
|
||||||
|
fnblockA/.style={fnblock,draw=fnAmber,fill=white},
|
||||||
|
fnreg/.style={draw=fnGrey,fill=white,minimum height=8mm,align=center,
|
||||||
|
font=\footnotesize,inner sep=3pt},
|
||||||
|
fnstate/.style={draw=fnBlue,fill=fnLight,circle,minimum size=13mm,
|
||||||
|
align=center,font=\scriptsize,line width=0.7pt},
|
||||||
|
fnarrow/.style={-{Stealth[length=2.6mm]},line width=0.8pt,draw=fnDark},
|
||||||
|
fnarrowT/.style={-{Stealth[length=2.6mm]},line width=0.8pt,draw=fnTeal},
|
||||||
|
fnbus/.style={-{Stealth[length=3mm]},line width=1.6pt,draw=fnBlue},
|
||||||
|
fnlbl/.style={font=\scriptsize\itshape,fill=white,inner sep=1pt,text=fnGrey}
|
||||||
|
}
|
||||||
|
|
||||||
|
% ---------- varie ------------------------------------------------------
|
||||||
|
\newcommand{\reg}[1]{\texttt{\textbf{#1}}}
|
||||||
|
\newcommand{\sig}[1]{\texttt{#1}}
|
||||||
|
\newcommand{\op}[1]{\texttt{\color{fnBlue}#1}}
|
||||||
|
\newcommand{\PASS}{\textcolor{fnGreen}{\textbf{PASS}}}
|
||||||
|
\newcommand{\FAIL}{\textcolor{fnRed}{\textbf{FAIL}}}
|
||||||
|
\newcommand{\OK}{\textcolor{fnGreen}{\textbf{OK}}}
|
||||||
|
\newcommand{\code}[1]{\texttt{#1}}
|
||||||
|
|
||||||
|
\usepackage{enumitem}
|
||||||
|
\setlist{noitemsep,topsep=2pt,leftmargin=1.4em}
|
||||||
|
|
||||||
|
\usepackage[hidelinks,colorlinks=true,linkcolor=fnBlue,urlcolor=fnTeal,
|
||||||
|
citecolor=fnBlue]{hyperref}
|
||||||
Reference in New Issue
Block a user