[
    {
        "id": "authors:yhgsm-50j43",
        "collection": "authors",
        "collection_id": "yhgsm-50j43",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20150121-071331821",
        "type": "article",
        "title": "GROK-LAB: Generating Real On-chip Knowledge for Intra-cluster Delays Using Timing Extraction",
        "author": [
            {
                "family_name": "Gojman",
                "given_name": "Benjamin",
                "clpid": "Gojman-B"
            },
            {
                "family_name": "Nalmela",
                "given_name": "Sirisha",
                "clpid": "Nalmela-S"
            },
            {
                "family_name": "Mehta",
                "given_name": "Nikil",
                "clpid": "Mehta-N"
            },
            {
                "family_name": "Howarth",
                "given_name": "Nicholas",
                "clpid": "Howarth-N"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Timing Extraction identifies the delay of fine-grained components within an FPGA. From these computed delays, the delay of any path can be calculated. Moreover, a comparison of the fine-grained delays allows a detailed understanding of the amount and type of process variation that exists in the FPGA. To obtain these delays, Timing Extraction measures, using only resources already available in the FPGA, the delay of a small subset of the total paths in the FPGA. We apply Timing Extraction to the Logic Array Block (LAB) on an Altera Cyclone III FPGA to obtain a view of the delay down to near-individual LUT SRAM cell granularity, characterizing components with delays on the order of tens to a few hundred picoseconds with a resolution of \u00b13.2ps, matching the expected error bounds. This information reveals that the 65nm process used has, on average, random variation of \u03c3 \u03bc =4.0% with components having an average maximum spread of 83ps. Timing Extraction also shows that as V_(DD) decreases from 1.2V to 0.9V in a Cyclone IV 60nm FPGA, paths slow down, and variation increases from \u03c3 \u03bc =4.3% to \u03c3 \u03bc =5.8%, a clear indication that lowering V_(DD) magnifies the impact of random variation.",
        "doi": "10.1145/2597889",
        "issn": "1936-7406",
        "publisher": "Association for Computing Machinery (ACM)",
        "publication": "ACM Transactions on Reconfigurable Technology Systems",
        "publication_date": "2014-12",
        "series_number": "4",
        "volume": "7",
        "issue": "4",
        "pages": "Art. No. 32"
    },
    {
        "id": "authors:ej0t3-q1g34",
        "collection": "authors",
        "collection_id": "ej0t3-q1g34",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20130328-131459182",
        "type": "book_section",
        "title": "GROK-LAB: generating real on-chip knowledge for intra-cluster delays using timing extraction",
        "book_title": "Proceedings of the ACM/SIGDA international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "Gojman",
                "given_name": "Benjamin",
                "clpid": "Gojman-B"
            },
            {
                "family_name": "Nalmela",
                "given_name": "Sirisha",
                "clpid": "Nalmela-S"
            },
            {
                "family_name": "Mehta",
                "given_name": "Nikil",
                "clpid": "Mehta-N"
            },
            {
                "family_name": "Howarth",
                "given_name": "Nicholas",
                "clpid": "Howarth-N"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Timing Extraction identifies the delay of fine-grained components within an FPGA. From these computed delays, the\ndelay of any path can be calculated. Moreover, a comparison\nof the fine-grained delays allows a detailed understanding of\nthe amount and type of process variation that exists in the\nFPGA. To obtain these delays, Timing Extraction measures,\nusing only resources already available in the FPGA, the delay of a small subset of the total paths in the FPGA. We\napply Timing Extraction to the Logic Array Block (LAB)\non an Altera Cyclone III FPGA to obtain a view of the\ndelay down to near individual LUT granularity, characterizing components with delays on the order of a few hundred\npicoseconds with a resolution of \u00b13.2 ps. This information\nreveals that the 65 nm process used has, on average, random\nvariation of \u03c3/\u00b5 = 4.0% with components having an average maximum spread of 83 ps. Timing Extraction also shows\nthat as VDD decreases from 1.2 V to 0.9 V in a Cyclone IV\n60 nm FPGA, paths slow down and variation increases from\n\u03c3/\u00b5 = 4.3% to \u03c3/\u00b5 5.8%, a clear indication that lowering\nV_(DD) magnifies the impact of random variation.",
        "doi": "10.1145/2435264.2435281",
        "isbn": "978-1-4503-1887-7",
        "publisher": "Association for Computing Machinery (ACM)",
        "place_of_publication": "New York",
        "publication_date": "2013-02",
        "pages": "81-90"
    },
    {
        "id": "authors:kztd8-8x720",
        "collection": "authors",
        "collection_id": "kztd8-8x720",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20120521-105703473",
        "type": "book_section",
        "title": "Limit study of energy & delay benefits of component-specific routing",
        "book_title": "Proceedings of the ACM/SIGDA international symposium on Field Programmable Gate Arrays",
        "author": [
            {
                "family_name": "Mehta",
                "given_name": "Nikil",
                "clpid": "Mehta-N"
            },
            {
                "family_name": "Rubin",
                "given_name": "Raphael",
                "clpid": "Rubin-R"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "As feature sizes scale toward atomic limits, parameter variation continues to increase, leading to increased margins in both delay and energy. The possibility of very slow devices on critical paths forces designers to increase transistor sizes, reduce clock speed and operate at higher voltages than desired in order to meet timing. With post-fabrication configurability, FPGAs have the opportunity to use slow devices on non-critical paths while selecting fast devices for critical paths. To understand the potential benefit we might gain from component-specific mapping, we quantify the margins associated with parameter variation in FPGAs over a wide range of predictive technologies (45nm-12nm) and gate sizes and show how these margins can be significantly reduced by delay-aware, component-specific routing. For the Toronto 20 benchmark set, we show that component-specific routing can eliminate delay margins induced by variation and reduce energy for energy minimal designs by 1.42-1.98\u00d7. We further show that these benefits increase as technology scales.",
        "doi": "10.1145/2145694.2145710",
        "isbn": "978-1-4503-1155-7",
        "publisher": "Association for Computing Machinery (ACM)",
        "place_of_publication": "New York, NY",
        "publication_date": "2012-02",
        "pages": "97-106"
    },
    {
        "id": "authors:7jm3y-6yj13",
        "collection": "authors",
        "collection_id": "7jm3y-6yj13",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20100707-100125364",
        "type": "book_section",
        "title": "Performance comparison of single-precision SPICE Model-Evaluation on FPGA, GPU, Cell, and multi-core processors",
        "book_title": "FPL:  2009 International Conference on Field Programmable Logic and Applications",
        "author": [
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Automated code generation and performance tuning techniques for concurrent architectures such as GPUs, Cell and FPGAs can provide integer factor speedups over multi-core processor organizations for data-parallel, floating-point computation in SPICE model-evaluation. Our Verilog AMS compiler produces code for parallel evaluation of non-linear circuit models suitable for use in SPICE  simulations where the same model is evaluated several times for all the devices in the circuit. Our compiler uses architecture specific parallelization strategies (OpenMP for multi-core, PThreads for Cell, CUDA for GPU, statically scheduled VLIW for FPGA) when producing code for these different architectures. We automatically explore different implementation configurations (e.g. unroll factor, vector length) using our performance-tuner to identify the best possible configuration for each architecture. We demonstrate speedups of 3- 182times for a Xilinx Virtex5 LX 330T, 1.3-33times for an IBM Cell, and 3-131times for an NVIDIA 9600 GT GPU over a 3 GHz Intel Xeon 5160 implementation for a variety of  single-precision  device models.",
        "doi": "10.1109/FPL.2009.5272548",
        "isbn": "978-1-4244-3891-4",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2009-09-29",
        "pages": "65-72"
    },
    {
        "id": "authors:efshf-kx075",
        "collection": "authors",
        "collection_id": "efshf-kx075",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20090831-143856470",
        "type": "article",
        "title": "Fault Secure Encoder and Decoder for NanoMemory Applications",
        "author": [
            {
                "family_name": "Naeimi",
                "given_name": "Helia",
                "clpid": "Naeimi-H"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Memory cells have been protected from soft errors for more than a decade; due to the increase in soft error rate in logic circuits, the encoder and decoder circuitry around the memory blocks have become susceptible to soft errors as well and must also be protected. We introduce a new approach to design fault-secure encoder and decoder circuitry for memory designs. The key novel contribution of this paper is identifying and defining a new class of error-correcting codes whose redundancy makes the design of fault-secure detectors (FSD) particularly simple. We further quantify the importance of protecting encoder and decoder circuitry against transient errors, illustrating a scenario where the system failure rate (FIT) is dominated by the failure rate of the encoder and decoder. We prove that Euclidean geometry low-density parity-check (EG-LDPC) codes have the fault-secure detector capability. Using some of the smaller EG-LDPC codes, we can tolerate bit or nanowire defect rates of 10% and fault rates of 10^(-18) upsets/device/cycle, achieving a FIT rate at or below one for the entire memory system and a memory density of 10^(11) bit/cm^2 with nanowire pitch of 10 nm for memory blocks of 10 Mb or larger. Larger EG-LDPC codes can achieve even higher reliability and lower area overhead.",
        "doi": "10.1109/TVLSI.2008.2009217",
        "issn": "1063-8210",
        "publisher": "IEEE",
        "publication": "IEEE Transactions on Very Large Scale Integration (VLSI) Systems",
        "publication_date": "2009-04",
        "series_number": "4",
        "volume": "17",
        "issue": "4",
        "pages": "473-486"
    },
    {
        "id": "authors:e959s-tn285",
        "collection": "authors",
        "collection_id": "e959s-tn285",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:PAPieeetc09",
        "type": "article",
        "title": "Pipelining Saturated Accumulation",
        "author": [
            {
                "family_name": "Papadantonakis",
                "given_name": "Karl",
                "clpid": "Papadantonakis-Karl"
            },
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "Chan",
                "given_name": "Stephanie",
                "clpid": "Chan-Stephanie"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Aggressive pipelining and spatial parallelism allow integrated circuits (e.g., custom VLSI, ASICs, and FPGAs) to achieve high throughput on many Digital Signal Processing applications. However, cyclic data dependencies in the computation can limit parallelism and reduce the efficiency and speed of an implementation. Saturated accumulation is an important example where such a cycle limits the throughput of signal processing applications. We show how to reformulate saturated addition as an associative operation so that we can use a parallel-prefix calculation to perform saturated accumulation at any data rate supported by the device. This allows us, for example, to design a 16-bit saturated accumulator which can operate at 280 MHz on a Xilinx Spartan-3(XC3S-5000-4) FPGA, the maximum frequency supported by the component's DCM.",
        "doi": "10.1109/TC.2008.110",
        "issn": "0018-9340",
        "publisher": "IEEE",
        "publication": "IEEE Transactions on Computers",
        "publication_date": "2009-02",
        "series_number": "2",
        "volume": "58",
        "issue": "2",
        "pages": "208-219"
    },
    {
        "id": "authors:771ey-q4c65",
        "collection": "authors",
        "collection_id": "771ey-q4c65",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20100507-150521265",
        "type": "book_section",
        "title": "Accelerating SPICE Model-Evaluation using FPGAs",
        "book_title": "Field Programmable Custom Computing Machines, 2009",
        "author": [
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Single-FPGA spatial implementations can provide\nan order of magnitude speedup over sequential microprocessor\nimplementations for data-parallel, floating-point computation in\nSPICE model-evaluation. Model-evaluation is a key component\nof the SPICE circuit simulator and it is characterized by\nlarge irregular floating-point compute graphs. We show how to\nexploit the parallelism available in these graphs on single-FPGA\ndesigns with a low-overhead VLIW-scheduled architecture. Our\narchitecture uses spatial floating-point operators coupled to local\nhigh-bandwidth memories and interconnected by a time-shared\nnetwork. We retime operation inputs in the model-evaluation to\nallow independent scheduling of computation and communication.\nWith this approach, we demonstrate speedups of 2\u201318\u00d7\nover a dual-core 3GHz Intel Xeon 5160 when using a Xilinx\nVirtex 5 LX330T for a variety of SPICE device models.",
        "doi": "10.1109/FCCM.2009.14",
        "isbn": "978-0-7695-3716-0",
        "publisher": "IEEE",
        "publication_date": "2009",
        "pages": "37-44"
    },
    {
        "id": "authors:y8qy7-7pe88",
        "collection": "authors",
        "collection_id": "y8qy7-7pe88",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:NAEnanot08",
        "type": "article",
        "title": "Fault-tolerant sub-lithographic design with rollback recovery",
        "author": [
            {
                "family_name": "Naeimi",
                "given_name": "Helia",
                "clpid": "Naeimi-H"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Shrinking feature sizes and energy levels coupled with high clock rates and decreasing node capacitance lead us into a regime where transient errors in logic cannot be ignored. Consequently, several recent studies have focused on feed-forward spatial redundancy techniques to combat these high transient fault rates. To complement these studies, we analyze fine-grained rollback techniques and show that they can offer lower spatial redundancy factors with no significant impact on system performance for fault rates up to one fault per device per ten million cycles of operation (Pf = 10^-7) in systems with 10^12 susceptible devices. Further, we concretely demonstrate these claims on nanowire-based programmable logic arrays. Despite expensive rollback buffers and general-purpose, conservative analysis, we show the area overhead factor of our technique is roughly an order of magnitude lower than a gate level feed-forward redundancy scheme.",
        "doi": "10.1088/0957-4484/19/11/115708",
        "issn": "0957-4484",
        "publisher": "IOP",
        "publication": "Nanotechnology",
        "publication_date": "2008-03-19",
        "series_number": "11",
        "volume": "19",
        "issue": "11",
        "pages": "115708"
    },
    {
        "id": "authors:y772g-8zq28",
        "collection": "authors",
        "collection_id": "y772g-8zq28",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110817-152509198",
        "type": "book_section",
        "title": "Fault Secure Encoder and Decoder for Memory Applications",
        "book_title": "2007 22nd IEEE International Symposium on Defect and Fault-Tolerance in VLSI Systems",
        "author": [
            {
                "family_name": "Naeimi",
                "given_name": "Helia",
                "clpid": "Naeimi-H"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "We introduce a reliable memory system that can tolerate multiple transient errors in the memory words as well as transient errors in the encoder and decoder (corrector) circuitry. The key novel development is the fault-secure detector (FSD) error-correcting code (ECC) definition and associated circuitry that can detect errors in the received encoded vector despite experiencing multiple transient faults in its circuitry. The structure of the detector is general enough that it can be used for any ECC that follows our FSD-ECC definition. We prove that two known classes of Low-Density Parity-Check Codes have the FSD-ECC property:\nEuclidean Geometry and Projective Geometry codes. We identify a specific FSD-LDPC code that can tolerate up to 33 errors in each memory word or supporting logic that requires only 30% area overhead for memory blocks of 10 Kbits or larger. Larger codes can achieve even higher reliability and lower area overhead. We quantify the importance of protecting encoder and decoder (corrector) circuitry and illustrate a scenario where the system failure rate (FIT) is dominated by the failure rate of the encoder and decoder.",
        "doi": "10.1109/DFT.2007.54",
        "isbn": "978-0-7695-2885-4",
        "publisher": "IEEE",
        "place_of_publication": "Los Alamitos, CA",
        "publication_date": "2007-09",
        "pages": "409-417"
    },
    {
        "id": "authors:qrw48-kg118",
        "collection": "authors",
        "collection_id": "qrw48-kg118",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20100923-111814269",
        "type": "book_section",
        "title": "Optimistic Parallelization of Floating-Point Accumulation",
        "book_title": "18th IEEE Symposium on Computer Arithmetic",
        "author": [
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Floating-point  arithmetic is notoriously non-associative due to the limited precision representation which demands intermediate values be rounded to fit in the available precision. The resulting cyclic dependency in floating-point accumulation inhibits parallelization  of the computation, including efficient use of pipelining. In practice, however, we observe that floating-point operations are \"mostly\" associative. This observation can be exploited to parallelize floating-point accumulation using a form of optimistic  concurrency. In this scheme, we first compute an optimistic  associative approximation to the sum and then relax the computation by iteratively propagating errors until the correct sum is obtained. We map this computation to a network of 16 statically-scheduled, pipelined, double-precision floating-point adders on the Virtex-4 LX160 (-12) device where each floating-point  adder runs at 296 MHz and has a pipeline depth of  10. On this 16 PE design, we demonstrate an average speedup of 6\u00d7 with randomly generated data and 3-7\u00d7 with summations extracted from Conjugate Gradient benchmarks.",
        "doi": "10.1109/ARITH.2007.25",
        "isbn": "978-0-7695-2854-0",
        "publisher": "IEEE",
        "place_of_publication": "Los Alamitos, CA",
        "publication_date": "2007",
        "pages": "205-216"
    },
    {
        "id": "authors:80y0j-71m30",
        "collection": "authors",
        "collection_id": "80y0j-71m30",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110720-093755059",
        "type": "book_section",
        "title": "Novel Design of Three-Dimensional Crossbar for Future Network on Chip based on Post-Silicon Devices",
        "book_title": "2006 1st International Conference on Nano-Networks and Workshops",
        "author": [
            {
                "family_name": "Nomura",
                "given_name": "Kumiko",
                "clpid": "Nomura-K"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Abe",
                "given_name": "Keiko",
                "clpid": "Abe-K"
            },
            {
                "family_name": "Fujita",
                "given_name": "Shinobu",
                "clpid": "Fujita-S"
            }
        ],
        "abstract": "We present a novel 3D crossbar for future Network-on-a-Chip implementations. We introduce a routing algorithm for\nthe 3D crossbar circuit and detail two specific 3D crossbar\ntopologies. We evaluate the defect tolerance of the 3D crossbar\nand quantify the number of extra layers required to support\narbitrary permutations as a function of the defect rate. Further,\nwe estimate the circuit performance and advantages of the 3D\ncrossbar circuit based on post-silicon devices.",
        "doi": "10.1109/NANONET.2006.346226",
        "isbn": "978-1-4244-0390-5",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2006-09",
        "pages": "72-76"
    },
    {
        "id": "authors:nz40y-amm34",
        "collection": "authors",
        "collection_id": "nz40y-amm34",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161213-174431809",
        "type": "article",
        "title": "Radial addressing of nanowires",
        "author": [
            {
                "family_name": "Savage",
                "given_name": "John E.",
                "clpid": "Savage-J-E"
            },
            {
                "family_name": "Rachlin",
                "given_name": "Eric",
                "clpid": "Rachlin-E"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Lieber",
                "given_name": "Charles M.",
                "clpid": "Lieber-C-M"
            },
            {
                "family_name": "Wu",
                "given_name": "Yue",
                "clpid": "Wu-Yue"
            }
        ],
        "abstract": "We introduce radial encoding of nanowires (NWs), a new method of differentiating and controlling NWs by a small set of mesoscale wires for use in crossbar memories. We describe methods of controlling these NWs and give efficient manufacturing algorithms. These new encoding and decoding methods do not suffer from the misalignment characteristic of flow-aligned NWs. They achieve comparable effective pitch and resulting memory density with axially encoded NWs, while avoiding potential cases of address ambiguity and simplifying NW preparation. We also explore hybrid axial/radial encodings and show that they offer no net benefit over pure codes.",
        "doi": "10.1145/1148015.1148018",
        "issn": "1550-4832",
        "publisher": "ACM",
        "publication": "ACM Journal on Emerging Technologies in Computing Systems (JETC)",
        "publication_date": "2006-04",
        "series_number": "2",
        "volume": "2",
        "issue": "2",
        "pages": "129-154"
    },
    {
        "id": "authors:dd5rr-vwy63",
        "collection": "authors",
        "collection_id": "dd5rr-vwy63",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20170109-150041807",
        "type": "book_section",
        "title": "SAT-based optimal hypergraph partitioning with replication",
        "book_title": "Asia and South Pacific Conference on Design Automation, 2006",
        "author": [
            {
                "family_name": "Wrighton",
                "given_name": "Michael G.",
                "clpid": "Wrighton-M-G"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9 M.",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "We propose a methodology for optimal k-way partitioning with replication of directed hypergraphs via Boolean satisfiability. We begin by leveraging the power of existing and emerging SAT solvers to attack traditional logic bipartitioning and show good scaling behavior. We continue to present the first optimal partitioning results that admit generation and assignment of replicated nodes concurrently. Our framework is general enough that we also give the first published optimal results for partitioning with respect to the maximum subdomain degree metric and the sum of external degrees metric. We show that for the bipartitioning case we can feasibly solve problems of up to 150 nodes with simultaneous replication in hundreds of seconds. For other partitioning metrics, we are able to solve problems up to 40 nodes in hundreds of seconds.",
        "doi": "10.1109/ASPDAC.2006.1594782",
        "isbn": "0-7803-9451-8",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2006-01",
        "pages": "789-795"
    },
    {
        "id": "authors:58gtg-gmy17",
        "collection": "authors",
        "collection_id": "58gtg-gmy17",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110630-080922757",
        "type": "book_section",
        "title": "3D Nanowire-Based Programmable Logic",
        "book_title": "2006 1st International Conference on Nano-Networks and Workshops",
        "author": [
            {
                "family_name": "Gojman",
                "given_name": "Benjamin",
                "clpid": "Gojman-B"
            },
            {
                "family_name": "Rubin",
                "given_name": "Raphael",
                "clpid": "Rubin-R"
            },
            {
                "family_name": "Pilotto",
                "given_name": "Concetta",
                "clpid": "Pilotto-C"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Tanamoto",
                "given_name": "Tefsufumi",
                "clpid": "Tanamoto-T"
            }
        ],
        "abstract": "In nanowire-based logic, the semiconducting material (e.g., Si, GaN, SiGe) is grown into individual nanowires rather than being part of the substrate. This offers us the opportunity to stack multiple layers of nanowires to create a three-dimensional logic structure which has high quality semiconductors in all vertical layers. The authors detail a feasible three-dimensional programmable logic architecture which can plausibly be realized from layers of semiconducting nanowires, making only modest assumptions about the control and placement of individual nanowires in the assembly. This shows a natural path for continuing to scale areal logic density once nanowire pitches approach fundamental limits. The authors show that the three dimensional systems are volumetrically efficient, with the surface area reducing roughly in proportion to the number of vertical layers. The authors further show that, on average, delay is reduced 18% from compact layout in three dimensions. For only a 20% area impact, the authors show how to avoid adding any manufacturing steps to physically isolate portions of nanowire layers.",
        "doi": "10.1109/NANONET.2006.346223",
        "isbn": "978-1-4244-0390-5",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2006",
        "pages": "54-58"
    },
    {
        "id": "authors:mnzr9-pqt53",
        "collection": "authors",
        "collection_id": "mnzr9-pqt53",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110223-132654232",
        "type": "book_section",
        "title": "GraphStep: A System Architecture for Sparse-Graph Algorithms",
        "book_title": "FCCM 2006: 14th Annual IEEE Symposium on Field-Programmable Custom Computing Machines",
        "author": [
            {
                "family_name": "deLorimier",
                "given_name": "Michael",
                "clpid": "deLorimier-M"
            },
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "Mehta",
                "given_name": "Nikil",
                "clpid": "Mehta-N"
            },
            {
                "family_name": "Rizzo",
                "given_name": "Dominic",
                "clpid": "Rizzo-D"
            },
            {
                "family_name": "Eslick",
                "given_name": "Ian",
                "clpid": "Eslick-I"
            },
            {
                "family_name": "Rubin",
                "given_name": "Raphael",
                "clpid": "Rubin-R"
            },
            {
                "family_name": "Uribe",
                "given_name": "Tom\u00e1s E.",
                "clpid": "Uribe-T-E"
            },
            {
                "family_name": "Knight",
                "given_name": "Thomas F., Jr.",
                "clpid": "Knight-T-F-Jr"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Many important applications are organized around\nlong-lived, irregular sparse graphs (e.g., data and knowledge\nbases, CAD optimization, numerical problems, simulations). The\ngraph structures are large, and the applications need regular\naccess to a large, data-dependent portion of the graph for each\noperation (e.g., the algorithm may need to walk the graph, visiting\nall nodes, or propagate changes through many nodes in the\ngraph). On conventional microprocessors, the graph structures\nexceed on-chip cache capacities, making main-memory bandwidth\nand latency the key performance limiters. To avoid this\n\"memory wall,\" we introduce a concurrent system architecture\nfor sparse graph algorithms that places graph nodes in small\ndistributed memories paired with specialized graph processing\nnodes interconnected by a lightweight network. This gives us a\nscalable way to map these applications so that they can exploit\nthe high-bandwidth and low-latency capabilities of embedded\nmemories (e.g., FPGA Block RAMs). On typical spreading activation\nqueries on the ConceptNet Knowledge Base, a sample\napplication, this translates into an order of magnitude speedup\nper FPGA compared to a state-of-the-art Pentium processor.",
        "doi": "10.1109/FCCM.2006.45",
        "isbn": "0-7695-2661-6",
        "publisher": "IEEE",
        "place_of_publication": "Los Alamitos, CA",
        "publication_date": "2006",
        "pages": "143-151"
    },
    {
        "id": "authors:qasmy-zr353",
        "collection": "authors",
        "collection_id": "qasmy-zr353",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110225-090013910",
        "type": "book_section",
        "title": "Packet Switched vs. Time Multiplexed FPGA\n Overlay Networks",
        "book_title": "FCCM 2006: 14th Annual IEEE Symposium on Field-Programmable Custom Computing Machines",
        "author": [
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "Mehta",
                "given_name": "Nikil",
                "clpid": "Mehta-N"
            },
            {
                "family_name": "deLorimier",
                "given_name": "Michael",
                "clpid": "deLorimier-M"
            },
            {
                "family_name": "Rubin",
                "given_name": "Raphael",
                "clpid": "Rubin-R"
            },
            {
                "family_name": "Barnor",
                "given_name": "Henry",
                "clpid": "Barnor-H"
            },
            {
                "family_name": "Wilson",
                "given_name": "Michael J.",
                "clpid": "Wilson-M-J"
            },
            {
                "family_name": "Wrighton",
                "given_name": "Michael",
                "clpid": "Wrighton-"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Dedicated, spatially configured FPGA interconnect\nis efficient for applications that require high throughput connections\nbetween processing elements (PEs) but with a limited degree\nof PE interconnectivity (e.g. wiring up gates and datapaths).\nApplications which virtualize PEs may require a large number\nof distinct PE-to-PE connections (e.g. using one PE to simulate\n100s of operators, each requiring input data from thousands of\nother operators), but with each connection having low throughput\ncompared with the PE's operating cycle time. In these highly interconnected\nconditions, dedicating spatial interconnect resources\nfor all possible connections is costly and inefficient. Alternatively,\nwe can time share physical network resources by virtualizing\ninterconnect links, either by statically scheduling the sharing\nof resources prior to runtime or by dynamically negotiating\nresources at runtime. We explore the tradeoffs (e.g. area, route\nlatency, route quality) between time-multiplexed and packet-switched\nnetworks overlayed on top of commodity FPGAs. We\ndemonstrate modular and scalable networks which operate on\na Xilinx XC2V6000-4 at 166MHz. For our applications, time-multiplexed,\noffline scheduling offers up to a 63% performance\nincrease over online, packet-switched scheduling for equivalent\ntopologies. When applying designs to equivalent area, packet-switching\nis up to 2\u00d7 faster for small area designs while time-multiplexing\nis up to 5\u00d7 faster for larger area designs. When\nlimited to the capacity of a XC2V6000, if all communication is\nknown, time-multiplexed routing outperforms packet-switching;\nhowever when the active set of links drops below 40% of the\npotential links, packet-switched routing can outperform time-multiplexing.",
        "doi": "10.1109/FCCM.2006.55",
        "isbn": "0-7695-2661-6",
        "publisher": "IEEE",
        "place_of_publication": "Los Alamitos, CA",
        "publication_date": "2006",
        "pages": "205-216"
    },
    {
        "id": "authors:a0n3m-tak37",
        "collection": "authors",
        "collection_id": "a0n3m-tak37",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110816-113648980",
        "type": "book_section",
        "title": "Pipelining saturated accumulation",
        "book_title": "2005 IEEE International Conference on Field-Programmable Technology",
        "author": [
            {
                "family_name": "Papadantonakis",
                "given_name": "Karl",
                "clpid": "Papadantonakis-Karl"
            },
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "Chang",
                "given_name": "Stephanie",
                "clpid": "Chang-Stephanie"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Aggressive pipelining allows FPGAs to achieve\nhigh throughput on many Digital Signal Processing\napplications. However, cyclic data dependencies\nin the computation can limit pipelining and reduce\nthe efficiency and speed of an FPGA implementation.\nSaturated accumulation is an important example\nwhere such a cycle limits the throughput of signal\nprocessing applications. We show how to reformulate\nsaturated addition as an associative operation\nso that we can use a parallel-prefix calculation to\nperform saturated accumulation at any data rate supported\nby the device. This allows us, for example, to\ndesign a 16-bit saturated accumulator which can operate\nat 280MHz on a Xilinx Spartan-3 (XC3S-5000-4), the maximum frequency supported by the component's DCM.",
        "doi": "10.1109/FPT.2005.1568519",
        "isbn": "0-7803-9407-0",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2005-12",
        "pages": "19-26"
    },
    {
        "id": "authors:h9513-7bc44",
        "collection": "authors",
        "collection_id": "h9513-7bc44",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:DEHieeetn05b",
        "type": "article",
        "title": "Deterministic Addressing of Nanoscale Devices Assembled at Sublithographic Pitches",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Multiple techniques have now been proposed using random addressing to build demultiplexers which interface between the large pitch of lithographically patterned features and the smaller pitch of self-assembled sublithographic nanowires. At the same time, the relatively high defect rates expected for molecular-sized devices and wires dictate that we design architectures with spare components so we can map around defective elements. To accommodate and mask both of these effects, we introduce a programmable addressing scheme which can be used to provide deterministic addresses for decoders built with random nanoscale addressing and potentially defective wires. We describe how this programmable addressing scheme can be implemented with emerging, nanoscale building blocks and show how to build deterministically addressable memory banks. We characterize the area required for this programmable addressing scheme. For 2048 x 2048 memory banks, the area overhead for address correction is less than 33%, delivering net memory densities around 10^11 b/cm^2.",
        "doi": "10.1109/TNANO.2005.858587",
        "issn": "1536-125X",
        "publisher": "IEEE Transactions on Nanotechnology",
        "publication": "IEEE Transactions on Nanotechnology",
        "publication_date": "2005-11-01",
        "series_number": "6",
        "volume": "4",
        "issue": "6",
        "pages": "681-687"
    },
    {
        "id": "authors:61hb7-q0730",
        "collection": "authors",
        "collection_id": "61hb7-q0730",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110815-092459485",
        "type": "book_section",
        "title": "Hybrid CMOS/Nanoelectronic Digital Circuits:\n Devices, Architectures, and Design Automation",
        "book_title": "ICCAD-2005 : International Conference on Computer Aided Design",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Likharev",
                "given_name": "Konstantin K.",
                "clpid": "Likharev-K-K"
            }
        ],
        "abstract": "Physics offers several active devices with nanometerscale\nfootprint, that can be best used in combination with a\nCMOS subsystem. Such hybrid circuits offer the potential for\nhigh defect tolerance combined with unparalleled performance. In this tutorial, we highlight key issues and architectural alternatives for this promising technology and outline the challenges posed by the hybrid circuits pose for design automation.",
        "doi": "10.1109/ICCAD.2005.1560097",
        "isbn": "0-7803-9254-X",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2005-11",
        "pages": "375-382"
    },
    {
        "id": "authors:50x2p-ypq77",
        "collection": "authors",
        "collection_id": "50x2p-ypq77",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20160420-100045980",
        "type": "article",
        "title": "Nanowire-Based Programmable Architectures",
        "author": [
            {
                "family_name": "Dehon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Chemists can now construct wires which are just a few atoms in diameter; these wires can be selectively\nfield-effect gated, and wire crossings can act as diodes with programmable resistance. These\nnew capabilities present both opportunities and challenges for constructing nanoscale computing\nsystems. The tiny feature sizes offer a path to economically scale down to atomic dimensions. However,\nthe associated bottom-up synthesis techniques only produce highly regular structures and\ncome with high defect rates and minimal control during assembly. To exploit these technologies, we\ndevelop nanowire-based architectures which can bridge between lithographic and atomic-scale feature\nsizes and tolerate defective and stochastic assembly of regular arrays to deliver high density\nuniversal computing devices. Using 10nm pitch nanowires, these nanowire-based programmable\narchitectures offer one to two orders of magnitude greater mapped-logic density than defect-free\nlithographic FPGAs at 22nm.",
        "doi": "10.1145/1084748.1084750",
        "issn": "1550-4832",
        "publisher": "ACM",
        "publication": "ACM Journal on Emerging Technologies in Computing Systems (JETC)",
        "publication_date": "2005-07",
        "series_number": "2",
        "volume": "1",
        "issue": "2",
        "pages": "109-162"
    },
    {
        "id": "authors:4gkav-cha21",
        "collection": "authors",
        "collection_id": "4gkav-cha21",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:DEHieeedtc05",
        "type": "article",
        "title": "Seven strategies for tolerating highly defective fabrication",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Naeimi",
                "given_name": "Helia",
                "clpid": "Naeimi-H"
            }
        ],
        "abstract": "In this article we present an architecture that supports fine-grained sparing and resource matching. The base logic structure is a set of interconnected PLAs. The PLAs and their interconnections consist of large arrays of interchangeable nanowires, which serve as programmable product and sum terms and as programmable interconnect links. Each nanowire can have several defective programmable junctions. We can test nanowires for functionality and use only the subset that provides appropriate conductivity and electrical characteristics. We then perform a matching between nanowire junction programmability and application logic needs to use almost all the nanowires even though most of them have defective junctions. We employ seven high-level strategies to achieve this level of defect tolerance.",
        "doi": "10.1109/MDT.2005.94",
        "issn": "0740-3224",
        "publisher": "IEEE",
        "publication": "IEEE Design and Test of Computers",
        "publication_date": "2005-07",
        "series_number": "4",
        "volume": "22",
        "issue": "4",
        "pages": "306-315"
    },
    {
        "id": "authors:mhws8-z1h70",
        "collection": "authors",
        "collection_id": "mhws8-z1h70",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:DEHieeetn05a",
        "type": "article",
        "title": "Nonphotolithographic nanoscale memory density prospects",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Goldstein",
                "given_name": "Seth Copen",
                "clpid": "Goldstein-S-C"
            },
            {
                "family_name": "Kuekes",
                "given_name": "Philip J.",
                "clpid": "Kuekes-P-J"
            }
        ],
        "abstract": "Technologies are now emerging to construct molecular-scale electronic wires and switches using bottom-up self-assembly. This opens the possibility of constructing nanoscale circuits and memories where active devices are just a few nanometers square and wire pitches may be on the order of ten nanometers. The features can be defined at this scale without using photolithography. The available assembly techniques have relatively high defect rates compared to conventional lithographic integrated circuits and can only produce very regular structures. Nonetheless, with proper memory organization, it is reasonable to expect these technologies to provide memory densities in excess of 10/sup 11/ b/cm/sup 2/ with modest active power requirements under 0.6 W/Tb/s for random read operations.",
        "doi": "10.1109/TNANO.2004.837849",
        "issn": "1536-125X",
        "publisher": "IEEE Transactions on Nanotechnology",
        "publication": "IEEE Transactions on Nanotechnology",
        "publication_date": "2005-03-01",
        "series_number": "2",
        "volume": "4",
        "issue": "2",
        "pages": "215-228"
    },
    {
        "id": "authors:40mkr-jd244",
        "collection": "authors",
        "collection_id": "40mkr-jd244",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20160419-162745354",
        "type": "book_section",
        "title": "Design of Programmable Interconnect for Sublithographic Programmable Logic Arrays",
        "book_title": "Proceedings of the 2005 ACM/SIGDA 13th international symposium on Field-programmable gate arrays (FPGA '05)",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Sublithographic Programmable Logic Arrays can be interconnected and restored using nanoscale wires. Building on a hybrid of bottom-up assembly techniques supported by\nconventional lithographic patterning, we show how modestsized PLA logic blocks, which are efficient for implementing logic, can be organized into a segmented, Manhattan mesh interconnection scheme. The resulting programmable architecture has a macro-scale view which is reminiscent of lithographic FPGA and CPLD designs despite the fact that the low-level, sublithographic fabrication techniques used are much more highly constrained than conventional lithography and are prone to high defect rates. Using the Toronto 20 benchmark set, we begin to explore the design space for these sublithographic architectures and show that they may allow us to exploit nanowire building blocks to reach one to\ntwo orders of magnitude greater density than 22nm CMOS\nlithography.",
        "doi": "10.1145/1046192.1046210",
        "isbn": "1-59593-029-9",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2005-02",
        "pages": "127-137"
    },
    {
        "id": "authors:913x4-ydr36",
        "collection": "authors",
        "collection_id": "913x4-ydr36",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161006-130031306",
        "type": "book_section",
        "title": "Floating-Point Sparse Matrix-Vector Multiply for FPGAs",
        "book_title": "FPGA '05 Proceedings of the 2005 ACM/SIGDA 13th international symposium on Field-programmable gate arrays",
        "author": [
            {
                "family_name": "deLorimier",
                "given_name": "Michael",
                "clpid": "deLorimier-M"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Schmit",
                "given_name": "Herman",
                "clpid": "Schmit-H"
            },
            {
                "family_name": "Wilton",
                "given_name": "Steve",
                "clpid": "Wilton-S"
            }
        ],
        "abstract": "Large, high density FPGAs with high local distributed memory bandwidth surpass the peak floating-point performance\nof high-end, general-purpose processors. Microprocessors do\nnot deliver near their peak floating-point performance on efficient algorithms that use the Sparse Matrix-Vector Multiply (SMVM) kernel. In fact, it is not uncommon for microprocessors to yield only 10\u201320% of their peak floating-point performance when computing SMVM. We develop and analyze a scalable SMVM implementation on modern FPGAs\nand show that it can sustain high throughput, near peak,\nfloating-point performance. For benchmark matrices from\nthe Matrix Market Suite we project 1.5 double precision\nGflops/FPGA for a single Virtex II 6000-4 and 12 double\nprecision Gflops for 16 Virtex IIs (750Mflops/FPGA).",
        "doi": "10.1145/1046192.1046203",
        "isbn": "1-59593-029-9",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2005-02",
        "pages": "75-85"
    },
    {
        "id": "authors:t087r-77f32",
        "collection": "authors",
        "collection_id": "t087r-77f32",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110822-103525220",
        "type": "book_section",
        "title": "A greedy algorithm for tolerating defective crosspoints in nanoPLA design",
        "book_title": "2004 IEEE International Conference on Field-Programmable Technology",
        "author": [
            {
                "family_name": "Naeimi",
                "given_name": "Helia",
                "clpid": "Naeimi-H"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Recent developments suggest both plausible fabrication techniques and viable architectures for building sublithographic programmable logic arrays using molecular-scale wires and switches. Designs at this scale will see much higher defect rates than in conventional lithography. However, these defects need not be an impediment to programmable logic design as this scale. We introduce a strategy for tolerating defective crosspoints and develop a linear-time, greedy algorithm for mapping PLA logic around crosspoint defects. We note that P-term fanin must be bounded to guarantee low overhead mapping and develop analytical guidelines for bounding fanin. We further quantify analytical and empirical mapping overhead rates. Including fanin bounding, our greedy mapping algorithm maps a large set of benchmark designs with 13% average overhead for random junction defect rates as high as 20%.",
        "doi": "10.1109/FPT.2004.1393250",
        "isbn": "0-7803-8651-5",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2004-12",
        "pages": "49-56"
    },
    {
        "id": "authors:f2pqv-krn06",
        "collection": "authors",
        "collection_id": "f2pqv-krn06",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:DEHieeetvlsis04b",
        "type": "article",
        "title": "Unifying mesh- and tree-based programmable interconnect",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "We examine the traditional, symmetric, Manhattan mesh design for field-programmable gate-array (FPGA) routing along with tree-of-meshes (ToM) and mesh-of-trees (MoT) based designs. All three networks can provide general routing for limited bisection designs (Rent's rule with p&lt;1) and allow locality exploitation. They differ in their detailed topology and use of hierarchy. We show that all three have the same asymptotic wiring requirements. We bound this tightly by providing constructive mappings between routes in one network and routes in another. For example, we show that a (c,p) MoT design can be mapped to a (2c,p) linear population ToM and introduce a corner turn scheme which will make it possible to perform the reverse mapping from any (c,p) linear population ToM to a (2c,p) MoT augmented with a particular set of corner turn switches. One consequence of this latter mapping is a multilayer layout strategy for N-node, linear population ToM designs that requires only /spl Theta/(N) two-dimensional area for any p when given sufficient wiring layers. We further show upper and lower bounds for global mesh routes based on recursive bisection width and show these are within a constant factor of each other and within a constant factor of MoT and ToM layout area. In the process we identify the parameters and characteristics which make the networks different, making it clear there is a unified design continuum in which these networks are simply particular regions.",
        "doi": "10.1109/TVLSI.2004.834237",
        "issn": "1063-8210",
        "publisher": "IEEE Transactions on Very Large Scale Integration (VLSI) Systems",
        "publication": "IEEE Transactions on Very Large Scale Integration (VLSI) Systems",
        "publication_date": "2004-10-01",
        "series_number": "10",
        "volume": "12",
        "issue": "10",
        "pages": "1051-1065"
    },
    {
        "id": "authors:43w8j-je549",
        "collection": "authors",
        "collection_id": "43w8j-je549",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:DEHieeetvlsis04a",
        "type": "article",
        "title": "Design of FPGA interconnect for multilevel metallization",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Rubin",
                "given_name": "Raphael",
                "clpid": "Rubin-R"
            }
        ],
        "abstract": "How does multilevel metallization impact the design of field-programmable gate arrays (FPGA) interconnect? The availability of a growing number of metal layers presents the opportunity to use wiring in the third dimension to reduce area and switch requirements. Unfortunately, traditional FPGA wiring schemes are not designed to exploit these additional metal layers. We introduce an alternate topology, based on Leighton's mesh-of-trees (MoT), which carefully exploits hierarchy to allow additional metal layers to support arbitrary device scaling. When wiring layers grow sufficiently fast with aggregate network size (N), our network requires only O(N) area; this is in stark contrast to traditional, Manhattan FPGA routing schemes where switching requirements alone grow superlinearly in N. In practice, we show that, even for the admittedly small designs in the Toronto \"FPGA Place and Route Challenge,\" arity-4 MoT networks require 26% fewer switches than the standard, Manhattan FPGA routing scheme.",
        "doi": "10.1109/TVLSI.2004.827562",
        "issn": "1063-8210",
        "publisher": "IEEE Transactions on Very Large Scale Integration (VLSI) Systems",
        "publication": "IEEE Transactions on Very Large Scale Integration (VLSI) Systems",
        "publication_date": "2004-10-01",
        "series_number": "10",
        "volume": "12",
        "issue": "10",
        "pages": "1038-1050"
    },
    {
        "id": "authors:vyd9b-vsa67",
        "collection": "authors",
        "collection_id": "vyd9b-vsa67",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110817-151253900",
        "type": "book_section",
        "title": "Design patterns for reconfigurable computing",
        "book_title": "12th Annual IEEE Symposium on Field-Programmable Custom Computing Machines",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Adams",
                "given_name": "Joshua",
                "clpid": "Adams-J"
            },
            {
                "family_name": "DeLorimier",
                "given_name": "Michael",
                "clpid": "DeLorimier-M"
            },
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "Matsuda",
                "given_name": "Yuki",
                "clpid": "Matsuda-Y"
            },
            {
                "family_name": "Naeimi",
                "given_name": "Helia",
                "clpid": "Naeimi-H"
            },
            {
                "family_name": "Vanier",
                "given_name": "Michael",
                "clpid": "Vanier-M"
            },
            {
                "family_name": "Wrighton",
                "given_name": "Michael",
                "clpid": "Wrighton-M"
            }
        ],
        "contributor": [
            {
                "family_name": "Pocek",
                "given_name": "Kenneth L.",
                "clpid": "Pocek-K-L"
            },
            {
                "family_name": "Arnold",
                "given_name": "Jeffrey M.",
                "clpid": "Arnold-J-M"
            }
        ],
        "abstract": "It is valuable to identify and catalog design patterns\nfor reconfigurable computing. These design patterns are\ncanonical solutions to common and recurring design challenges which arise in reconfigurable systems and applications. The catalog can form the basis for creating designs, for educating new designers, for understanding the needs of tools and languages, and for discussing reconfigurable design. Tying application and implementation lessons to the expansion and refinement of this catalog will make those lessons more relevant to the design community. In this paper, we articulate this role for design patterns in reconfigurable computing, provide a few example patterns, offer a starting point for the contents of the catalog, and discuss the potential benefits of this effort.",
        "doi": "10.1109/FCCM.2004.29",
        "isbn": "0-7695-2230-0",
        "publisher": "IEEE",
        "place_of_publication": "Los Alamitos, CA",
        "publication_date": "2004-04",
        "pages": "13-23"
    },
    {
        "id": "authors:0az87-65x26",
        "collection": "authors",
        "collection_id": "0az87-65x26",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161006-125146350",
        "type": "book_section",
        "title": "Nanowire-Based Sublithographic Programmable Logic Arrays",
        "book_title": "FPGA '04 Proceedings of the 2004 ACM/SIGDA 12th international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Wilson",
                "given_name": "Michael J.",
                "clpid": "Wilson-M-J"
            }
        ],
        "contributor": [
            {
                "family_name": "Tessier",
                "given_name": "Russ",
                "clpid": "Tessier-R"
            },
            {
                "family_name": "Schmit",
                "given_name": "Herman",
                "clpid": "Schmit-H"
            }
        ],
        "abstract": "How can Programmable Logic Arrays (PLAs) be built without relying on lithography to pattern their smallest features? In this paper, we detail designs which exploit emerging, bottom-up material synthesis techniques to build PLAs\nusing molecular-scale nanowires. Our new designs accommodate technologies where the only post-fabrication programmable element is a non-restoring diode. We introduce\nstochastic techniques which allow us to restore the diode\nlogic at the nanoscale so that it can be cascaded and interconnected for general logic evaluation. Under conservative assumptions using 10nm nanowires and 90nm lithographic support, we project yielded logic density around 500,000nm^2/or term for a 60 or-term array; a complete 60-term, two-level PLA is roughly the same size as a single 4-LUT logic block in 22nm lithography. Each\nor term is comparable in area to a 4-transistor hardwired gate at 22nm. Mapping sample datapaths and conventional programmable logic benchmarks, we estimate that each 60-\nor-term PLA plane will provide equivalent logic to 5\u201310 4-input LUTs.",
        "doi": "10.1145/968280.968299",
        "isbn": "1-58113-829-6",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2004-02",
        "pages": "123-132"
    },
    {
        "id": "authors:6kk7e-36813",
        "collection": "authors",
        "collection_id": "6kk7e-36813",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161005-174825781",
        "type": "book_section",
        "title": "What is the Right Model for Programming and Using Modern FPGAs?",
        "book_title": "FPGA '04 Proceedings of the 2004 ACM/SIGDA 12th international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Hutchings",
                "given_name": "Brad",
                "clpid": "Hutchings-B"
            },
            {
                "family_name": "Rudusky",
                "given_name": "Daryl",
                "clpid": "Rudusky-D"
            },
            {
                "family_name": "Hwang",
                "given_name": "James",
                "clpid": "Hwang-James"
            },
            {
                "family_name": "Nikhil",
                "clpid": "Nikhil-"
            },
            {
                "family_name": "Raje",
                "given_name": "Salil",
                "clpid": "Raje-S"
            },
            {
                "family_name": "Stoica",
                "given_name": "Adrian",
                "clpid": "Stoica-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Tessier",
                "given_name": "Russ",
                "clpid": "Tessier-R"
            },
            {
                "family_name": "Schmit",
                "given_name": "Herman",
                "clpid": "Schmit-H"
            }
        ],
        "abstract": "Traditionally, FPGAs have been the bastard step-brother of ASICs. They have been forced to act like ASICs and fit themselves into the ASIC development model. This has meant ignoring their unique strengths: reprogrammability, late-binding and run-time reconfiguration. Today, however, FPGAs are becoming more acceptable for their own merits. The majority of new design starts are FPGA designs. As FPGAs rise from under the shadow of their aging brother, should they continue to try to wear his hand-me-downs? Or is it time to develop more suitable models that lets them shine? At the same time, the old ASIC model is not even serving ASICs well, and new models for developing ASICs are emerging. All of this may encourage us to rethink how we should be programming FPGA-based systems. Possibilities include:\n\n\u2022 Using the traditional, ASIC model -- it's tried and true, has demonstrated success\n\n\u2022 Using the traditional, sequential processor model -- Compile programs from C down to FPGAs...perhaps evolving FPGAs to better support\n\n\u2022 Using the emerging C-level design tools\n\n\u2022 Using concurrent and/or streaming models (incl. CSP, Matlab/simulink, Ptolemy, SCORE, Stream-C) Using a biologically inspired model (neural networks, genetic programming...)",
        "doi": "10.1145/968280.968281",
        "isbn": "1-58113-829-6",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2004-02",
        "pages": "119"
    },
    {
        "id": "authors:ttd4p-35r90",
        "collection": "authors",
        "collection_id": "ttd4p-35r90",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20130816-103300863",
        "type": "article",
        "title": "Saliency on a chip: a digital approach with an FPGA",
        "author": [
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "Walther",
                "given_name": "Dirk B.",
                "clpid": "Walther-D-B"
            },
            {
                "family_name": "Koch",
                "given_name": "Christof",
                "orcid": "0000-0001-6482-8067",
                "clpid": "Koch-C"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Selective-visual-attention algorithms have\nbeen successfully implemented in analog\nVLSI circuits.1 However, in addition to\nthe usual issues of analog VLSI\u2014such as\nthe need to fi ne-tune a large number of biases\u2014\nthese implementations lack the spatial\nresolution and pre-processing capabilities\nto be truly useful for image-processing\napplications. Here we take an alternative\napproach and implement a neuro-mimetic\nalgorithm for selective visual attention in\ndigital hardware.",
        "issn": "1548-5625",
        "publisher": "Institute of Neuromorphic Engineering",
        "publication": "The Neuromorphic Engineer",
        "publication_date": "2004",
        "series_number": "2",
        "volume": "1",
        "issue": "2",
        "pages": "9-11"
    },
    {
        "id": "authors:kas05-j5h06",
        "collection": "authors",
        "collection_id": "kas05-j5h06",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:DEHieeetn03b",
        "type": "article",
        "title": "Stochastic assembly of sublithographic nanoscale interfaces",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Lincoln",
                "given_name": "Patrick",
                "clpid": "Lincoln-P"
            },
            {
                "family_name": "Savage",
                "given_name": "John E.",
                "clpid": "Savage-J-E"
            }
        ],
        "abstract": "We describe a technique for addressing individual nanoscale wires with microscale control wires without using lithographic-scale processing to define nanoscale dimensions. Such a scheme is necessary to exploit sublithographic nanoscale storage and computational devices. Our technique uses modulation doping to address individual nanowires and self-assembly to organize them into nanoscale-pitch decoder arrays. We show that if coded nanowires are chosen at random from a sufficiently large population, we can ensure that a large fraction of the selected nanowires have unique addresses. For example, we show that N lines can be uniquely addressed over 99% of the time using no more than /spl lceil/2.2log/sub 2/(N)/spl rceil/+11 address wires. We further show a hybrid decoder scheme that only needs to address N=O(W/sub litho-pitch//W/sub nano-pitch/) wires at a time through this stochastic scheme; as a result, the number of unique codes required for the nanowires does not grow with decoder size. We give an O(N/sup 2/) procedure to discover the addresses which are present. We also demonstrate schemes that tolerate the misalignment of nanowires which can occur during the self-assembly process.",
        "doi": "10.1109/TNANO.2003.816658",
        "issn": "1536-125X",
        "publisher": "IEEE Transactions on Nanotechnology",
        "publication": "IEEE Transactions on Nanotechnology",
        "publication_date": "2003-09-01",
        "series_number": "3",
        "volume": "2",
        "issue": "3",
        "pages": "165-174"
    },
    {
        "id": "authors:hfykf-p5s68",
        "collection": "authors",
        "collection_id": "hfykf-p5s68",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:DEHieeetn03a",
        "type": "article",
        "title": "Array-based architecture for FET-based, nanoscale electronics",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Advances in our basic scientific understanding at the molecular and atomic level place us on the verge of engineering designer structures with key features at the single nanometer scale. This offers us the opportunity to design computing systems at what may be the ultimate limits on device size. At this scale, we are faced with new challenges and a new cost structure which motivates different computing architectures than we found efficient and appropriate in conventional very large scale integration (VLSI). We sketch a basic architecture for nanoscale electronics based on carbon nanotubes, silicon nanowires, and nano-scale FETs. This architecture can provide universal logic functionality with all logic and signal restoration operating at the nanoscale. The key properties of this architecture are its minimalism, defect tolerance, and compatibility with emerging bottom-up nanoscale fabrication techniques. The architecture further supports micro-to-nanoscale interfacing for communication with conventional integrated circuits and bootstrap loading.",
        "doi": "10.1109/TNANO.2003.808508",
        "issn": "1536-125X",
        "publisher": "IEEE Transactions on Nanotechnology",
        "publication": "IEEE Transactions on Nanotechnology",
        "publication_date": "2003-03-01",
        "series_number": "1",
        "volume": "2",
        "issue": "1",
        "pages": "23-32"
    },
    {
        "id": "authors:jbpnj-0k057",
        "collection": "authors",
        "collection_id": "jbpnj-0k057",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20170109-150810119",
        "type": "book_section",
        "title": "Hardware-assisted simulated annealing with application for fast FPGA placement",
        "book_title": "FPGA '03 Proceedings of the 2003 ACM/SIGDA eleventh international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "Wrighton",
                "given_name": "Michael G.",
                "clpid": "Wrighton-M-G"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9 M.",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Trimberger",
                "given_name": "Steve",
                "clpid": "Trimberger-S"
            }
        ],
        "abstract": "To truly exploit FPGAs for rapid turn-around development and prototyping, placement times must be reduced to seconds; late-bound, reconfigurable computing applications may demand placement times as short as microseconds. In this paper, we show how a systolic structure can accelerate placement by assigning one processing element to each possible location for an FPGA LUT from a design netlist. We demonstrate that our technique approaches the same quality point as traditional simulated annealing as measured by a simple linear wirelength metric. Experimental results look ahead to compare quality against VPR's fast placer when considering the minimum channel width required to route as the primary optimization criteria. Preliminary results from an FPGA implementation show the feasibility of accelerating simulated annealing by three orders of magnitude using this approach. This means we can place the largest design in the University of Toronto's \"FPGA Placement and Routing Challenge\" in around 4ms.",
        "doi": "10.1145/611817.611824",
        "isbn": "1-58113-651-X",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2003-02",
        "pages": "33-42"
    },
    {
        "id": "authors:xz519-c3423",
        "collection": "authors",
        "collection_id": "xz519-c3423",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161213-164304748",
        "type": "book_section",
        "title": "Design of FPGA interconnect for multilevel metalization",
        "book_title": "FPGA '03 Proceedings of the 2003 ACM/SIGDA eleventh international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "Rubin",
                "given_name": "Raphael",
                "clpid": "Rubin-R"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Trimberger",
                "given_name": "Steve",
                "clpid": "Trimberger-S"
            }
        ],
        "abstract": "How does multilevel metalization impact the design of FPGA interconnect? The availability of a growing number of metal layers presents the opportunity to use wiring in the third-dimension to reduce switch requirements. Unfortunately, traditional FPGA wiring schemes are not designed to exploit these additional metal layers. We introduce an alternate topology, based on Leighton's Mesh-of-Trees, which carefully exploits hierarchy to allow additional metal layers to support arbitrary device scaling. When wiring layers grow sufficiently fast with aggregate network size (N), our network requires only O(N) area; this is in stark contrast to traditional, Manhattan FPGA routing schemes where switching requirements alone grow superlinearly in N. In practice, we show that, even for the admittedly small designs in the Toronto \"FPGA Place and Route Challenge,\" the Mesh-of-Trees networks require 10% less switches than the standard, Manhattan FPGA routing scheme.",
        "doi": "10.1145/611817.611841",
        "isbn": "1-58113-651-X",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2003-02",
        "pages": "154-163"
    },
    {
        "id": "authors:pb7bz-0jq38",
        "collection": "authors",
        "collection_id": "pb7bz-0jq38",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161107-154348196",
        "type": "book_section",
        "title": "Stochastic, spatial routing for hypergraphs, trees, and meshes",
        "book_title": "FPGA '03 Proceedings of the 2003 ACM/SIGDA eleventh international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "Huang",
                "given_name": "Randy",
                "clpid": "Huang-Randy"
            },
            {
                "family_name": "Wawrzynek",
                "given_name": "John",
                "clpid": "Wawrzynek-J"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Trimberger",
                "given_name": "Steve",
                "clpid": "Trimberger-S"
            }
        ],
        "abstract": "FPGA place and route is time consuming, often serving as the major obstacle inhibiting a fast edit-compile-test loop in prototyping and development and the major obstacle preventing late-bound hardware and design mapping for reconfigurable systems. Previous work showed that hardware-assisted routing can accelerate fanout-free routing on Fat-Trees by three orders of magnitude with modest modifications to the network itself. In this paper, we show how these techniques can be applied to any FPGA and how they can be implemented on top of LUT networks in cases where modification of the FPGA itself is not justified. We further show how to accommodate fanout and how to achieve comparable route quality to software-based methods. For a tree network, we estimate an FPGA implementation of our routing logic could route the Toronto Place and Route Benchmarks at least two orders of magnitude faster than a software Pathfinder while achieving within 3% of the aggregate quality. Preliminary results on small mesh benchmarks achieve within one track of vpr-fast.",
        "doi": "10.1145/611817.611830",
        "isbn": "1-58113-651-X",
        "publisher": "ACM",
        "place_of_publication": "New York, CA",
        "publication_date": "2003-02",
        "pages": "78-87"
    },
    {
        "id": "authors:bnmwd-wvz15",
        "collection": "authors",
        "collection_id": "bnmwd-wvz15",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20160812-112140221",
        "type": "book_section",
        "title": "Molecular electronics: devices, systems and tools for gigagate, gigabit chips",
        "book_title": "IEEE/ACM International Conference on Computer Aided Design, 2002. ICCAD 2002",
        "author": [
            {
                "family_name": "Butts",
                "given_name": "Michael",
                "clpid": "Butts-M"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Goldstein",
                "given_name": "Seth Copen",
                "clpid": "Goldstein-S-C"
            }
        ],
        "abstract": "New electronics technologies are emerging which may carry us beyond the limits of lithographic processing down to molecular-scale feature sizes. Devices and interconnects can be made from a variety of molecules and materials including bistable and switchable organic molecules, carbon nanotubes, and, single-crystal semiconductor nanowires. They can be self-assembled into organized structures and attached onto lithographic substrates. This tutorial reviews emerging molecular-scale electronics technology for CAD and system designers and highlights where ICCAD research can help support this technology.",
        "doi": "10.1109/ICCAD.2002.1167569",
        "isbn": "0-7803-7607-2",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2002-11",
        "pages": "443-440"
    },
    {
        "id": "authors:h511y-26m36",
        "collection": "authors",
        "collection_id": "h511y-26m36",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20200127-123636117",
        "type": "book_section",
        "title": "Very Large Scale Spatial Computing",
        "book_title": "Unconventional Models of Computation",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Calude",
                "given_name": "C. S.",
                "clpid": "Calude-C-S"
            }
        ],
        "abstract": "The early decades of computing were marked by limited re sources. However, as we enter the twenty-first century, silicon is offering enormous computing resources on a single die and molecular-scale devices appear plausible offering a path to even greater capacities. Exploiting the capacities of these modern and future devices demands different compu tational models and radical shifts in the way we organize, capture, and optimize computations. A key shift is toward spatially organized compu tation. A natural consequence is that the dominant effects which govern our computing space change from the total number of operations and temporal locality to interconnect complexity and spatial locality. Old computational models which hide, ignore, or obfuscate communication and emphasize temporal sequences inhibit the exploitation of these mod ern capacities, motivating the need for new models which make commu nication and spatial organization more apparent.",
        "doi": "10.1007/3-540-45833-6_3",
        "isbn": "978-3-540-44311-7",
        "publisher": "Springer",
        "place_of_publication": "Berlin",
        "publication_date": "2002-10-10",
        "pages": "27-37"
    },
    {
        "id": "authors:9nkea-5nk06",
        "collection": "authors",
        "collection_id": "9nkea-5nk06",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161129-173323435",
        "type": "book_section",
        "title": "Analysis of quasi-static scheduling techniques in a virtualized reconfigurable machine",
        "book_title": "FPGA '02 Proceedings of the 2002 ACM/SIGDA tenth international symposium on Field-programmable gate arrays",
        "author": [
            {
                "family_name": "Markovskiy",
                "given_name": "Yury",
                "clpid": "Markovskiy-Y"
            },
            {
                "family_name": "Caspi",
                "given_name": "Eylon",
                "clpid": "Caspi-E"
            },
            {
                "family_name": "Huang",
                "given_name": "Randy",
                "clpid": "Huang-Randy"
            },
            {
                "family_name": "Yeh",
                "given_name": "Joseph",
                "clpid": "Yeh-Joseph"
            },
            {
                "family_name": "Chu",
                "given_name": "Michael",
                "clpid": "Chu-Michael"
            },
            {
                "family_name": "Wawrzynek",
                "given_name": "John",
                "clpid": "Wawrzynek-J"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Schlag",
                "given_name": "Martine",
                "clpid": "Schlag-"
            },
            {
                "family_name": "Trimberger",
                "given_name": "Steve",
                "clpid": "Trimberger-S"
            }
        ],
        "abstract": "he SCORE compute model uses fixed-size, virtual compute and memory pages connected by stream links to capture the definition of a computation abstracted from the detailed size of the physical hardware. When the number of physical compute pages is smaller than the number of virtual compute pages in the abstract computation graph, the design is time-multiplexed onto the available physical hardware. A key component of this strategy is an automatic scheduler that selects the temporal sequencing of virtual resources onto the physical device. We describe a quasi-static scheduling strategy that retains the full semantic power of the dynamic SCORE flow graph while taking advantage of static scheduling techniques at program load time to hoist most of the computational work out of the inner scheduling loops. This strategy reduces online scheduling work per reconfiguration epoch by an order of magnitude. In addition, a more global perspective available from offline-scheduling improves schedule quality, resulting in a net reduction of total execution time by 46-81%.",
        "doi": "10.1145/503048.503077",
        "isbn": "1-58113-452-5",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2002-02",
        "pages": "196-205"
    },
    {
        "id": "authors:y9ayh-1fq31",
        "collection": "authors",
        "collection_id": "y9ayh-1fq31",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:DEHcomputer00",
        "type": "article",
        "title": "The density advantage of configurable computing",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "More and more, field-programmable gate arrays (FPGAs) are accelerating computing applications. The absolute performance achieved by these configurable machines has been impressive-often one to two orders of magnitude greater than processor-based alternatives. Configurable computing is one of the fastest, most economical ways to solve problems such as RSA (Rivest-Shamir-Adelman) decryption, DNA sequence matching, signal processing, emulation, and cryptographic attacks. But questions remain as to why FPGAs have been so much more successful than their microprocessor and DSP counterparts. Do FPGA architectures have inherent advantages? Or are these examples just flukes of technology and market pricing? Will advantages increase, decrease, or remain the same as technology advances? Is there some generalization that accounts for the advantages in these cases? \n\nThe author attempts to answer these questions and to see how configurable computing fits into the arsenal of structures used to build general, programmable computing platforms.",
        "doi": "10.1109/2.839320",
        "issn": "0018-9162",
        "publisher": "Computer",
        "publication": "Computer",
        "publication_date": "2000-04-01",
        "series_number": "4",
        "volume": "33",
        "issue": "4",
        "pages": "41-49"
    },
    {
        "id": "authors:fcn9b-c3e17",
        "collection": "authors",
        "collection_id": "fcn9b-c3e17",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:MANcompute97",
        "type": "article",
        "title": "Seeking solutions in configurable computing",
        "author": [
            {
                "family_name": "Mangione-Smith",
                "given_name": "William H.",
                "clpid": "Mangione-Smith-W-H"
            },
            {
                "family_name": "Hutchings",
                "given_name": "Brad",
                "clpid": "Hutchings-B"
            },
            {
                "family_name": "Andrews",
                "given_name": "David",
                "clpid": "Andrews-D"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Ebeling",
                "given_name": "Carl",
                "clpid": "Ebeling-C"
            },
            {
                "family_name": "Hartenstein",
                "given_name": "Reiner",
                "clpid": "Hartenstein-R"
            },
            {
                "family_name": "Mencer",
                "given_name": "Oskar",
                "clpid": "Mencer-O"
            },
            {
                "family_name": "Morris",
                "given_name": "John",
                "clpid": "Morris-J"
            },
            {
                "family_name": "Palem",
                "given_name": "Krishna",
                "clpid": "Palem-K"
            },
            {
                "family_name": "Prasanna",
                "given_name": "Viktor K.",
                "clpid": "Prasanna-V-K"
            },
            {
                "family_name": "Spaanenburg",
                "given_name": "Henk A. E.",
                "clpid": "Spaanenburg-H-A-E"
            }
        ],
        "abstract": "Configurable computing offers the potential of producing powerful new computing systems. Will current research overcome the dearth of commercial applicability to make such systems a reality? Unfortunately, no system to date has yet proven attractive or competitive enough to establish a commercial presence. We believe that ample opportunity exists for work in a broad range of areas. In particular, the configurable computing community should focus on refining the emerging architectures, producing more effective software/hardware APIs, better tools for application development that incorporate the models of hardware reconfiguration, and effective benchmarking strategies.",
        "doi": "10.1109/2.642810",
        "issn": "0018-9162",
        "publisher": "Computer",
        "publication": "Computer",
        "publication_date": "1997-12-01",
        "series_number": "12",
        "volume": "30",
        "issue": "12",
        "pages": "38-43"
    }
]