\relax 
\@writefile{toc}{\string\contentsline\space {section}{\string\numberline\space {1}Introduction and Motivation}{3}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Computing Requirements of Vision.}{3}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Reconfigurable Bit-Parallel Processing.}{4}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Thesis Approach.}{4}}
\citation{Potter85}
\citation{Cloud88}
\citation{Blank90}
\citation{Kim93}
\@writefile{toc}{\string\contentsline\space {section}{\string\numberline\space {2}Previous and Related Work}{5}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {2.1}Architectural Design}{5}}
\citation{Li89}
\citation{Blevins88}
\citation{Shu89}
\citation{Rushton89}
\citation{Thacker94}
\citation{terasys}
\citation{execube}
\citation{berkeley-dsp}
\citation{pam_performance_assessment}
\citation{Audet92}
\@writefile{lot}{\string\contentsline\space {table}{\string\numberline\space {1}{\ignorespaces Architectural performance comparison}}{7}}
\newlabel{tab:perfcomp}{{1}{7}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {2.2}Architectural Analysis}{7}}
\citation{Herbordt94b}
\@writefile{toc}{\string\contentsline\space {section}{\string\numberline\space {3}Reconfigurable Bit-Parallel Architecture}{9}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {3.1}The Basic Idea of RBP}{9}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {1}{\ignorespaces RBP Organization}}{9}}
\newlabel{fig:rbporg}{{1}{9}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {3.2}RBP Performance Model}{9}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {3.3}Abacus PE Architecture}{10}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Activity Register}{10}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {2}{\ignorespaces Silicon Efficiency Ratios}}{11}}
\newlabel{fig:effgraph}{{2}{11}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {3}{\ignorespaces Processing Element Block Diagram}}{11}}
\newlabel{fig:block}{{3}{11}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {4}{\ignorespaces Network cell}}{12}}
\newlabel{fig:network}{{4}{12}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Network}{12}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Background I/O}{12}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {3.4}RBP Algorithms}{12}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Logical Shift.}{12}}
\citation{Bolotski93}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {5}{\ignorespaces Shift}}{13}}
\newlabel{fig:shift}{{5}{13}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Sum Accumulation.}{13}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {6}{\ignorespaces Accumulate}}{13}}
\newlabel{fig:accum}{{6}{13}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Addition.}{13}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {7}{\ignorespaces Addition}}{14}}
\newlabel{fig:addition}{{7}{14}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Match.}{14}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Comparison.}{14}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Multiplication.}{14}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {8}{\ignorespaces Compare}}{15}}
\newlabel{fig:compare}{{8}{15}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Mesh Move.}{15}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Scan Operations.}{15}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {9}{\ignorespaces A +-reduce operation.}}{16}}
\newlabel{fig:tree-reduce}{{9}{16}}
\@writefile{lot}{\string\contentsline\space {table}{\string\numberline\space {2}{\ignorespaces Single-chip arithmetic performance}}{16}}
\newlabel{tab:perf}{{2}{16}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {3.5}Performance Summary.}{16}}
\@writefile{toc}{\string\contentsline\space {section}{\string\numberline\space {4}The Abacus-1 Chip Implementation}{17}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {4.1}PE Core}{17}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Processing Element}{17}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Network}{17}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {10}{\ignorespaces Network Accelerator}}{17}}
\newlabel{fig:netaccel}{{10}{17}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {11}{\ignorespaces Chip Photograph}}{18}}
\newlabel{fig:chip}{{11}{18}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Chip Organization.}{19}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Input/Output Interface}{19}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {4.2}Off-Chip Interfaces}{19}}
\@writefile{toc}{\string\contentsline\space {subsubsection}{\string\numberline\space {4.2.1}System Synchronization}{19}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {12}{\ignorespaces DLL Arbiter}}{20}}
\newlabel{fig:arbit}{{12}{20}}
\@writefile{toc}{\string\contentsline\space {subsubsection}{\string\numberline\space {4.2.2}Instruction Distribution}{20}}
\@writefile{toc}{\string\contentsline\space {subsubsection}{\string\numberline\space {4.2.3}Inter-Chip Mesh Communication}{20}}
\citation{automatic-impedance-control}
\@writefile{toc}{\string\contentsline\space {subsubsection}{\string\numberline\space {4.2.4}Image I/O}{21}}
\@writefile{toc}{\string\contentsline\space {subsubsection}{\string\numberline\space {4.2.5}External Memory Interface}{21}}
\citation{Horn86}
\citation{Harris86}
\citation{Little88}
\@writefile{toc}{\string\contentsline\space {section}{\string\numberline\space {5}System-Level Design}{22}}
\@writefile{toc}{\string\contentsline\space {section}{\string\numberline\space {6}Parallel Algorithm Implementation and Performance}{22}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {6.1}Basic Vision Algorithms}{22}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Edge Detection}{22}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Surface Reconstruction}{22}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Optical Flow}{22}}
\citation{Levialdi72}
\citation{Leighton92}
\citation{Ziavras93}
\citation{Choudhary90}
\citation{Cypher90}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {13}{\ignorespaces Edge Detection Example}}{23}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Performance Summary}{23}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {6.2}Early vision subset of the DARPA IU Benchmark}{23}}
\citation{Leighton92}
\citation{Levialdi72}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {14}{\ignorespaces Optical Flow Example}}{24}}
\newlabel{fig:optflow}{{14}{24}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Connected Component Labelling: Broadcast}{24}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Connected Component Labelling: Shrinking}{24}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {15}{\ignorespaces Optical Flow Field}}{25}}
\newlabel{fig:optvec}{{15}{25}}
\citation{Cypher89}
\@writefile{toc}{\string\contentsline\space {paragraph}{K-curvature Tracking and Corner Detection}{26}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {16}{\ignorespaces K-curvature definition}}{26}}
\newlabel{fig:k-curv-def}{{16}{26}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Median Filter}{26}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Gradient Magnitude}{26}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Hough Transform}{26}}
\citation{Leighton92}
\citation{Quinn87}
\citation{Schnorr86}
\citation{Herbordt94}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {6.3}Sorting and Routing Algorithms}{27}}
\@writefile{toc}{\string\contentsline\space {paragraph}{ShearSort and RevSort}{27}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Mesh Greedy Routing Algorithm}{27}}
\@writefile{toc}{\string\contentsline\space {section}{\string\numberline\space {7}Architectural Tradeoffs}{28}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {7.1}Slice Optimization}{28}}
\@writefile{lot}{\string\contentsline\space {table}{\string\numberline\space {3}{\ignorespaces Architectural and Algorithmic Parameters}}{28}}
\newlabel{tab:param}{{3}{28}}
\newlabel{eq:proctime}{{3}{29}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {7.2}Background Loading}{30}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {7.3}Conclusion}{30}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {17}{\ignorespaces Probability of an on-chip hit sequence of length $k$ for different values of $p$, the miss probability. Note that for high $p$ almost all the area under the curve is near small values of $k$, and therefore long load times. }}{31}}
\newlabel{fig:seqprob}{{17}{31}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {18}{\ignorespaces Effective latency vs latency due to background loading as a function of $L$ and $p$. The advantage of background loading occurs in a relatively narrow zone: where the frequency of loads is very low to begin with, and where latency is low. For low values of $L$, the ratio may not be really important, since it approaches internal access time.}}{32}}
\newlabel{fig:leff}{{18}{32}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {19}{\ignorespaces Execution time as function of $m$. The dip in the middle is the effect of background loading. Without it, the graph would have continued in a straight line. Notice that the dip occurs at a non-useful $m$ value, as it is difficult to evenly partition a 16-bit value.}}{33}}
\newlabel{fig:execfuncm}{{19}{33}}
\@writefile{lof}{\string\contentsline\space {figure}{\string\numberline\space {20}{\ignorespaces Execution time as function of m and bandwidth}}{34}}
\newlabel{fig:execfuncm2}{{20}{34}}
\@writefile{toc}{\string\contentsline\space {section}{\string\numberline\space {8}Next Generation: Abacus-2}{35}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {8.1}Lessons of Abacus-1}{35}}
\@writefile{toc}{\string\contentsline\space {section}{\string\numberline\space {9}Beyond SIMD: Abacus-3}{36}}
\@writefile{toc}{\string\contentsline\space {paragraph}{M-SIMD Operation.}{36}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Systolic Operation.}{36}}
\@writefile{toc}{\string\contentsline\space {paragraph}{SIMD Element or Programmable Logic?}{37}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Universal Computing Element?}{37}}
\@writefile{toc}{\string\contentsline\space {section}{\string\numberline\space {10}Conclusions}{37}}
\@writefile{toc}{\string\contentsline\space {section}{\string\numberline\space {A}Rough Work}{38}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {A.1}Scan and Reduce Primitives}{38}}
\@writefile{toc}{\string\contentsline\space {subsubsection}{\string\numberline\space {A.1.1}Unpipelined loop}{38}}
\@writefile{toc}{\string\contentsline\space {subsubsection}{\string\numberline\space {A.1.2}Pipelined Loop}{39}}
\@writefile{toc}{\string\contentsline\space {subsubsection}{\string\numberline\space {A.1.3}Conclusions}{40}}
\@writefile{toc}{\string\contentsline\space {subsection}{\string\numberline\space {A.2}Starting Computing During Background Loading}{40}}
\@writefile{toc}{\string\contentsline\space {paragraph}{Start Work Early?}{40}}
\bibstyle{ieeetr}
\bibdata{arch,transit,dpga}
\bibcite{Potter85}{1}
\bibcite{Cloud88}{2}
\bibcite{Blank90}{3}
\bibcite{Kim93}{4}
\bibcite{Li89}{5}
\bibcite{Blevins88}{6}
\bibcite{Shu89}{7}
\bibcite{Rushton89}{8}
\bibcite{Thacker94}{9}
\bibcite{terasys}{10}
\bibcite{execube}{11}
\bibcite{berkeley-dsp}{12}
\bibcite{pam_performance_assessment}{13}
\bibcite{Audet92}{14}
\bibcite{Herbordt94b}{15}
\bibcite{Bolotski93}{16}
\bibcite{automatic-impedance-control}{17}
\bibcite{Horn86}{18}
\bibcite{Harris86}{19}
\bibcite{Little88}{20}
\bibcite{Levialdi72}{21}
\bibcite{Leighton92}{22}
\bibcite{Ziavras93}{23}
\bibcite{Choudhary90}{24}
\bibcite{Cypher90}{25}
\bibcite{Cypher89}{26}
\bibcite{Quinn87}{27}
\bibcite{Schnorr86}{28}
\bibcite{Herbordt94}{29}
