[
    {
        "id": "authors:ej0t3-q1g34",
        "collection": "authors",
        "collection_id": "ej0t3-q1g34",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20130328-131459182",
        "type": "book_section",
        "title": "GROK-LAB: generating real on-chip knowledge for intra-cluster delays using timing extraction",
        "book_title": "Proceedings of the ACM/SIGDA international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "Gojman",
                "given_name": "Benjamin",
                "clpid": "Gojman-B"
            },
            {
                "family_name": "Nalmela",
                "given_name": "Sirisha",
                "clpid": "Nalmela-S"
            },
            {
                "family_name": "Mehta",
                "given_name": "Nikil",
                "clpid": "Mehta-N"
            },
            {
                "family_name": "Howarth",
                "given_name": "Nicholas",
                "clpid": "Howarth-N"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Timing Extraction identifies the delay of fine-grained components within an FPGA. From these computed delays, the\ndelay of any path can be calculated. Moreover, a comparison\nof the fine-grained delays allows a detailed understanding of\nthe amount and type of process variation that exists in the\nFPGA. To obtain these delays, Timing Extraction measures,\nusing only resources already available in the FPGA, the delay of a small subset of the total paths in the FPGA. We\napply Timing Extraction to the Logic Array Block (LAB)\non an Altera Cyclone III FPGA to obtain a view of the\ndelay down to near individual LUT granularity, characterizing components with delays on the order of a few hundred\npicoseconds with a resolution of \u00b13.2 ps. This information\nreveals that the 65 nm process used has, on average, random\nvariation of \u03c3/\u00b5 = 4.0% with components having an average maximum spread of 83 ps. Timing Extraction also shows\nthat as VDD decreases from 1.2 V to 0.9 V in a Cyclone IV\n60 nm FPGA, paths slow down and variation increases from\n\u03c3/\u00b5 = 4.3% to \u03c3/\u00b5 5.8%, a clear indication that lowering\nV_(DD) magnifies the impact of random variation.",
        "doi": "10.1145/2435264.2435281",
        "isbn": "978-1-4503-1887-7",
        "publisher": "Association for Computing Machinery (ACM)",
        "place_of_publication": "New York",
        "publication_date": "2013-02",
        "pages": "81-90"
    },
    {
        "id": "authors:kztd8-8x720",
        "collection": "authors",
        "collection_id": "kztd8-8x720",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20120521-105703473",
        "type": "book_section",
        "title": "Limit study of energy & delay benefits of component-specific routing",
        "book_title": "Proceedings of the ACM/SIGDA international symposium on Field Programmable Gate Arrays",
        "author": [
            {
                "family_name": "Mehta",
                "given_name": "Nikil",
                "clpid": "Mehta-N"
            },
            {
                "family_name": "Rubin",
                "given_name": "Raphael",
                "clpid": "Rubin-R"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "As feature sizes scale toward atomic limits, parameter variation continues to increase, leading to increased margins in both delay and energy. The possibility of very slow devices on critical paths forces designers to increase transistor sizes, reduce clock speed and operate at higher voltages than desired in order to meet timing. With post-fabrication configurability, FPGAs have the opportunity to use slow devices on non-critical paths while selecting fast devices for critical paths. To understand the potential benefit we might gain from component-specific mapping, we quantify the margins associated with parameter variation in FPGAs over a wide range of predictive technologies (45nm-12nm) and gate sizes and show how these margins can be significantly reduced by delay-aware, component-specific routing. For the Toronto 20 benchmark set, we show that component-specific routing can eliminate delay margins induced by variation and reduce energy for energy minimal designs by 1.42-1.98\u00d7. We further show that these benefits increase as technology scales.",
        "doi": "10.1145/2145694.2145710",
        "isbn": "978-1-4503-1155-7",
        "publisher": "Association for Computing Machinery (ACM)",
        "place_of_publication": "New York, NY",
        "publication_date": "2012-02",
        "pages": "97-106"
    },
    {
        "id": "authors:7jm3y-6yj13",
        "collection": "authors",
        "collection_id": "7jm3y-6yj13",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20100707-100125364",
        "type": "book_section",
        "title": "Performance comparison of single-precision SPICE Model-Evaluation on FPGA, GPU, Cell, and multi-core processors",
        "book_title": "FPL:  2009 International Conference on Field Programmable Logic and Applications",
        "author": [
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Automated code generation and performance tuning techniques for concurrent architectures such as GPUs, Cell and FPGAs can provide integer factor speedups over multi-core processor organizations for data-parallel, floating-point computation in SPICE model-evaluation. Our Verilog AMS compiler produces code for parallel evaluation of non-linear circuit models suitable for use in SPICE  simulations where the same model is evaluated several times for all the devices in the circuit. Our compiler uses architecture specific parallelization strategies (OpenMP for multi-core, PThreads for Cell, CUDA for GPU, statically scheduled VLIW for FPGA) when producing code for these different architectures. We automatically explore different implementation configurations (e.g. unroll factor, vector length) using our performance-tuner to identify the best possible configuration for each architecture. We demonstrate speedups of 3- 182times for a Xilinx Virtex5 LX 330T, 1.3-33times for an IBM Cell, and 3-131times for an NVIDIA 9600 GT GPU over a 3 GHz Intel Xeon 5160 implementation for a variety of  single-precision  device models.",
        "doi": "10.1109/FPL.2009.5272548",
        "isbn": "978-1-4244-3891-4",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2009-09-29",
        "pages": "65-72"
    },
    {
        "id": "authors:771ey-q4c65",
        "collection": "authors",
        "collection_id": "771ey-q4c65",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20100507-150521265",
        "type": "book_section",
        "title": "Accelerating SPICE Model-Evaluation using FPGAs",
        "book_title": "Field Programmable Custom Computing Machines, 2009",
        "author": [
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Single-FPGA spatial implementations can provide\nan order of magnitude speedup over sequential microprocessor\nimplementations for data-parallel, floating-point computation in\nSPICE model-evaluation. Model-evaluation is a key component\nof the SPICE circuit simulator and it is characterized by\nlarge irregular floating-point compute graphs. We show how to\nexploit the parallelism available in these graphs on single-FPGA\ndesigns with a low-overhead VLIW-scheduled architecture. Our\narchitecture uses spatial floating-point operators coupled to local\nhigh-bandwidth memories and interconnected by a time-shared\nnetwork. We retime operation inputs in the model-evaluation to\nallow independent scheduling of computation and communication.\nWith this approach, we demonstrate speedups of 2\u201318\u00d7\nover a dual-core 3GHz Intel Xeon 5160 when using a Xilinx\nVirtex 5 LX330T for a variety of SPICE device models.",
        "doi": "10.1109/FCCM.2009.14",
        "isbn": "978-0-7695-3716-0",
        "publisher": "IEEE",
        "publication_date": "2009",
        "pages": "37-44"
    },
    {
        "id": "authors:y772g-8zq28",
        "collection": "authors",
        "collection_id": "y772g-8zq28",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110817-152509198",
        "type": "book_section",
        "title": "Fault Secure Encoder and Decoder for Memory Applications",
        "book_title": "2007 22nd IEEE International Symposium on Defect and Fault-Tolerance in VLSI Systems",
        "author": [
            {
                "family_name": "Naeimi",
                "given_name": "Helia",
                "clpid": "Naeimi-H"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "We introduce a reliable memory system that can tolerate multiple transient errors in the memory words as well as transient errors in the encoder and decoder (corrector) circuitry. The key novel development is the fault-secure detector (FSD) error-correcting code (ECC) definition and associated circuitry that can detect errors in the received encoded vector despite experiencing multiple transient faults in its circuitry. The structure of the detector is general enough that it can be used for any ECC that follows our FSD-ECC definition. We prove that two known classes of Low-Density Parity-Check Codes have the FSD-ECC property:\nEuclidean Geometry and Projective Geometry codes. We identify a specific FSD-LDPC code that can tolerate up to 33 errors in each memory word or supporting logic that requires only 30% area overhead for memory blocks of 10 Kbits or larger. Larger codes can achieve even higher reliability and lower area overhead. We quantify the importance of protecting encoder and decoder (corrector) circuitry and illustrate a scenario where the system failure rate (FIT) is dominated by the failure rate of the encoder and decoder.",
        "doi": "10.1109/DFT.2007.54",
        "isbn": "978-0-7695-2885-4",
        "publisher": "IEEE",
        "place_of_publication": "Los Alamitos, CA",
        "publication_date": "2007-09",
        "pages": "409-417"
    },
    {
        "id": "authors:qrw48-kg118",
        "collection": "authors",
        "collection_id": "qrw48-kg118",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20100923-111814269",
        "type": "book_section",
        "title": "Optimistic Parallelization of Floating-Point Accumulation",
        "book_title": "18th IEEE Symposium on Computer Arithmetic",
        "author": [
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Floating-point  arithmetic is notoriously non-associative due to the limited precision representation which demands intermediate values be rounded to fit in the available precision. The resulting cyclic dependency in floating-point accumulation inhibits parallelization  of the computation, including efficient use of pipelining. In practice, however, we observe that floating-point operations are \"mostly\" associative. This observation can be exploited to parallelize floating-point accumulation using a form of optimistic  concurrency. In this scheme, we first compute an optimistic  associative approximation to the sum and then relax the computation by iteratively propagating errors until the correct sum is obtained. We map this computation to a network of 16 statically-scheduled, pipelined, double-precision floating-point adders on the Virtex-4 LX160 (-12) device where each floating-point  adder runs at 296 MHz and has a pipeline depth of  10. On this 16 PE design, we demonstrate an average speedup of 6\u00d7 with randomly generated data and 3-7\u00d7 with summations extracted from Conjugate Gradient benchmarks.",
        "doi": "10.1109/ARITH.2007.25",
        "isbn": "978-0-7695-2854-0",
        "publisher": "IEEE",
        "place_of_publication": "Los Alamitos, CA",
        "publication_date": "2007",
        "pages": "205-216"
    },
    {
        "id": "authors:80y0j-71m30",
        "collection": "authors",
        "collection_id": "80y0j-71m30",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110720-093755059",
        "type": "book_section",
        "title": "Novel Design of Three-Dimensional Crossbar for Future Network on Chip based on Post-Silicon Devices",
        "book_title": "2006 1st International Conference on Nano-Networks and Workshops",
        "author": [
            {
                "family_name": "Nomura",
                "given_name": "Kumiko",
                "clpid": "Nomura-K"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Abe",
                "given_name": "Keiko",
                "clpid": "Abe-K"
            },
            {
                "family_name": "Fujita",
                "given_name": "Shinobu",
                "clpid": "Fujita-S"
            }
        ],
        "abstract": "We present a novel 3D crossbar for future Network-on-a-Chip implementations. We introduce a routing algorithm for\nthe 3D crossbar circuit and detail two specific 3D crossbar\ntopologies. We evaluate the defect tolerance of the 3D crossbar\nand quantify the number of extra layers required to support\narbitrary permutations as a function of the defect rate. Further,\nwe estimate the circuit performance and advantages of the 3D\ncrossbar circuit based on post-silicon devices.",
        "doi": "10.1109/NANONET.2006.346226",
        "isbn": "978-1-4244-0390-5",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2006-09",
        "pages": "72-76"
    },
    {
        "id": "authors:dd5rr-vwy63",
        "collection": "authors",
        "collection_id": "dd5rr-vwy63",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20170109-150041807",
        "type": "book_section",
        "title": "SAT-based optimal hypergraph partitioning with replication",
        "book_title": "Asia and South Pacific Conference on Design Automation, 2006",
        "author": [
            {
                "family_name": "Wrighton",
                "given_name": "Michael G.",
                "clpid": "Wrighton-M-G"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9 M.",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "We propose a methodology for optimal k-way partitioning with replication of directed hypergraphs via Boolean satisfiability. We begin by leveraging the power of existing and emerging SAT solvers to attack traditional logic bipartitioning and show good scaling behavior. We continue to present the first optimal partitioning results that admit generation and assignment of replicated nodes concurrently. Our framework is general enough that we also give the first published optimal results for partitioning with respect to the maximum subdomain degree metric and the sum of external degrees metric. We show that for the bipartitioning case we can feasibly solve problems of up to 150 nodes with simultaneous replication in hundreds of seconds. For other partitioning metrics, we are able to solve problems up to 40 nodes in hundreds of seconds.",
        "doi": "10.1109/ASPDAC.2006.1594782",
        "isbn": "0-7803-9451-8",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2006-01",
        "pages": "789-795"
    },
    {
        "id": "authors:qasmy-zr353",
        "collection": "authors",
        "collection_id": "qasmy-zr353",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110225-090013910",
        "type": "book_section",
        "title": "Packet Switched vs. Time Multiplexed FPGA\n Overlay Networks",
        "book_title": "FCCM 2006: 14th Annual IEEE Symposium on Field-Programmable Custom Computing Machines",
        "author": [
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "Mehta",
                "given_name": "Nikil",
                "clpid": "Mehta-N"
            },
            {
                "family_name": "deLorimier",
                "given_name": "Michael",
                "clpid": "deLorimier-M"
            },
            {
                "family_name": "Rubin",
                "given_name": "Raphael",
                "clpid": "Rubin-R"
            },
            {
                "family_name": "Barnor",
                "given_name": "Henry",
                "clpid": "Barnor-H"
            },
            {
                "family_name": "Wilson",
                "given_name": "Michael J.",
                "clpid": "Wilson-M-J"
            },
            {
                "family_name": "Wrighton",
                "given_name": "Michael",
                "clpid": "Wrighton-"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Dedicated, spatially configured FPGA interconnect\nis efficient for applications that require high throughput connections\nbetween processing elements (PEs) but with a limited degree\nof PE interconnectivity (e.g. wiring up gates and datapaths).\nApplications which virtualize PEs may require a large number\nof distinct PE-to-PE connections (e.g. using one PE to simulate\n100s of operators, each requiring input data from thousands of\nother operators), but with each connection having low throughput\ncompared with the PE's operating cycle time. In these highly interconnected\nconditions, dedicating spatial interconnect resources\nfor all possible connections is costly and inefficient. Alternatively,\nwe can time share physical network resources by virtualizing\ninterconnect links, either by statically scheduling the sharing\nof resources prior to runtime or by dynamically negotiating\nresources at runtime. We explore the tradeoffs (e.g. area, route\nlatency, route quality) between time-multiplexed and packet-switched\nnetworks overlayed on top of commodity FPGAs. We\ndemonstrate modular and scalable networks which operate on\na Xilinx XC2V6000-4 at 166MHz. For our applications, time-multiplexed,\noffline scheduling offers up to a 63% performance\nincrease over online, packet-switched scheduling for equivalent\ntopologies. When applying designs to equivalent area, packet-switching\nis up to 2\u00d7 faster for small area designs while time-multiplexing\nis up to 5\u00d7 faster for larger area designs. When\nlimited to the capacity of a XC2V6000, if all communication is\nknown, time-multiplexed routing outperforms packet-switching;\nhowever when the active set of links drops below 40% of the\npotential links, packet-switched routing can outperform time-multiplexing.",
        "doi": "10.1109/FCCM.2006.55",
        "isbn": "0-7695-2661-6",
        "publisher": "IEEE",
        "place_of_publication": "Los Alamitos, CA",
        "publication_date": "2006",
        "pages": "205-216"
    },
    {
        "id": "authors:58gtg-gmy17",
        "collection": "authors",
        "collection_id": "58gtg-gmy17",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110630-080922757",
        "type": "book_section",
        "title": "3D Nanowire-Based Programmable Logic",
        "book_title": "2006 1st International Conference on Nano-Networks and Workshops",
        "author": [
            {
                "family_name": "Gojman",
                "given_name": "Benjamin",
                "clpid": "Gojman-B"
            },
            {
                "family_name": "Rubin",
                "given_name": "Raphael",
                "clpid": "Rubin-R"
            },
            {
                "family_name": "Pilotto",
                "given_name": "Concetta",
                "clpid": "Pilotto-C"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Tanamoto",
                "given_name": "Tefsufumi",
                "clpid": "Tanamoto-T"
            }
        ],
        "abstract": "In nanowire-based logic, the semiconducting material (e.g., Si, GaN, SiGe) is grown into individual nanowires rather than being part of the substrate. This offers us the opportunity to stack multiple layers of nanowires to create a three-dimensional logic structure which has high quality semiconductors in all vertical layers. The authors detail a feasible three-dimensional programmable logic architecture which can plausibly be realized from layers of semiconducting nanowires, making only modest assumptions about the control and placement of individual nanowires in the assembly. This shows a natural path for continuing to scale areal logic density once nanowire pitches approach fundamental limits. The authors show that the three dimensional systems are volumetrically efficient, with the surface area reducing roughly in proportion to the number of vertical layers. The authors further show that, on average, delay is reduced 18% from compact layout in three dimensions. For only a 20% area impact, the authors show how to avoid adding any manufacturing steps to physically isolate portions of nanowire layers.",
        "doi": "10.1109/NANONET.2006.346223",
        "isbn": "978-1-4244-0390-5",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2006",
        "pages": "54-58"
    },
    {
        "id": "authors:mnzr9-pqt53",
        "collection": "authors",
        "collection_id": "mnzr9-pqt53",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110223-132654232",
        "type": "book_section",
        "title": "GraphStep: A System Architecture for Sparse-Graph Algorithms",
        "book_title": "FCCM 2006: 14th Annual IEEE Symposium on Field-Programmable Custom Computing Machines",
        "author": [
            {
                "family_name": "deLorimier",
                "given_name": "Michael",
                "clpid": "deLorimier-M"
            },
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "Mehta",
                "given_name": "Nikil",
                "clpid": "Mehta-N"
            },
            {
                "family_name": "Rizzo",
                "given_name": "Dominic",
                "clpid": "Rizzo-D"
            },
            {
                "family_name": "Eslick",
                "given_name": "Ian",
                "clpid": "Eslick-I"
            },
            {
                "family_name": "Rubin",
                "given_name": "Raphael",
                "clpid": "Rubin-R"
            },
            {
                "family_name": "Uribe",
                "given_name": "Tom\u00e1s E.",
                "clpid": "Uribe-T-E"
            },
            {
                "family_name": "Knight",
                "given_name": "Thomas F., Jr.",
                "clpid": "Knight-T-F-Jr"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Many important applications are organized around\nlong-lived, irregular sparse graphs (e.g., data and knowledge\nbases, CAD optimization, numerical problems, simulations). The\ngraph structures are large, and the applications need regular\naccess to a large, data-dependent portion of the graph for each\noperation (e.g., the algorithm may need to walk the graph, visiting\nall nodes, or propagate changes through many nodes in the\ngraph). On conventional microprocessors, the graph structures\nexceed on-chip cache capacities, making main-memory bandwidth\nand latency the key performance limiters. To avoid this\n\"memory wall,\" we introduce a concurrent system architecture\nfor sparse graph algorithms that places graph nodes in small\ndistributed memories paired with specialized graph processing\nnodes interconnected by a lightweight network. This gives us a\nscalable way to map these applications so that they can exploit\nthe high-bandwidth and low-latency capabilities of embedded\nmemories (e.g., FPGA Block RAMs). On typical spreading activation\nqueries on the ConceptNet Knowledge Base, a sample\napplication, this translates into an order of magnitude speedup\nper FPGA compared to a state-of-the-art Pentium processor.",
        "doi": "10.1109/FCCM.2006.45",
        "isbn": "0-7695-2661-6",
        "publisher": "IEEE",
        "place_of_publication": "Los Alamitos, CA",
        "publication_date": "2006",
        "pages": "143-151"
    },
    {
        "id": "authors:a0n3m-tak37",
        "collection": "authors",
        "collection_id": "a0n3m-tak37",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110816-113648980",
        "type": "book_section",
        "title": "Pipelining saturated accumulation",
        "book_title": "2005 IEEE International Conference on Field-Programmable Technology",
        "author": [
            {
                "family_name": "Papadantonakis",
                "given_name": "Karl",
                "clpid": "Papadantonakis-Karl"
            },
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "Chang",
                "given_name": "Stephanie",
                "clpid": "Chang-Stephanie"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Aggressive pipelining allows FPGAs to achieve\nhigh throughput on many Digital Signal Processing\napplications. However, cyclic data dependencies\nin the computation can limit pipelining and reduce\nthe efficiency and speed of an FPGA implementation.\nSaturated accumulation is an important example\nwhere such a cycle limits the throughput of signal\nprocessing applications. We show how to reformulate\nsaturated addition as an associative operation\nso that we can use a parallel-prefix calculation to\nperform saturated accumulation at any data rate supported\nby the device. This allows us, for example, to\ndesign a 16-bit saturated accumulator which can operate\nat 280MHz on a Xilinx Spartan-3 (XC3S-5000-4), the maximum frequency supported by the component's DCM.",
        "doi": "10.1109/FPT.2005.1568519",
        "isbn": "0-7803-9407-0",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2005-12",
        "pages": "19-26"
    },
    {
        "id": "authors:61hb7-q0730",
        "collection": "authors",
        "collection_id": "61hb7-q0730",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110815-092459485",
        "type": "book_section",
        "title": "Hybrid CMOS/Nanoelectronic Digital Circuits:\n Devices, Architectures, and Design Automation",
        "book_title": "ICCAD-2005 : International Conference on Computer Aided Design",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Likharev",
                "given_name": "Konstantin K.",
                "clpid": "Likharev-K-K"
            }
        ],
        "abstract": "Physics offers several active devices with nanometerscale\nfootprint, that can be best used in combination with a\nCMOS subsystem. Such hybrid circuits offer the potential for\nhigh defect tolerance combined with unparalleled performance. In this tutorial, we highlight key issues and architectural alternatives for this promising technology and outline the challenges posed by the hybrid circuits pose for design automation.",
        "doi": "10.1109/ICCAD.2005.1560097",
        "isbn": "0-7803-9254-X",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2005-11",
        "pages": "375-382"
    },
    {
        "id": "authors:913x4-ydr36",
        "collection": "authors",
        "collection_id": "913x4-ydr36",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161006-130031306",
        "type": "book_section",
        "title": "Floating-Point Sparse Matrix-Vector Multiply for FPGAs",
        "book_title": "FPGA '05 Proceedings of the 2005 ACM/SIGDA 13th international symposium on Field-programmable gate arrays",
        "author": [
            {
                "family_name": "deLorimier",
                "given_name": "Michael",
                "clpid": "deLorimier-M"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Schmit",
                "given_name": "Herman",
                "clpid": "Schmit-H"
            },
            {
                "family_name": "Wilton",
                "given_name": "Steve",
                "clpid": "Wilton-S"
            }
        ],
        "abstract": "Large, high density FPGAs with high local distributed memory bandwidth surpass the peak floating-point performance\nof high-end, general-purpose processors. Microprocessors do\nnot deliver near their peak floating-point performance on efficient algorithms that use the Sparse Matrix-Vector Multiply (SMVM) kernel. In fact, it is not uncommon for microprocessors to yield only 10\u201320% of their peak floating-point performance when computing SMVM. We develop and analyze a scalable SMVM implementation on modern FPGAs\nand show that it can sustain high throughput, near peak,\nfloating-point performance. For benchmark matrices from\nthe Matrix Market Suite we project 1.5 double precision\nGflops/FPGA for a single Virtex II 6000-4 and 12 double\nprecision Gflops for 16 Virtex IIs (750Mflops/FPGA).",
        "doi": "10.1145/1046192.1046203",
        "isbn": "1-59593-029-9",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2005-02",
        "pages": "75-85"
    },
    {
        "id": "authors:40mkr-jd244",
        "collection": "authors",
        "collection_id": "40mkr-jd244",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20160419-162745354",
        "type": "book_section",
        "title": "Design of Programmable Interconnect for Sublithographic Programmable Logic Arrays",
        "book_title": "Proceedings of the 2005 ACM/SIGDA 13th international symposium on Field-programmable gate arrays (FPGA '05)",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Sublithographic Programmable Logic Arrays can be interconnected and restored using nanoscale wires. Building on a hybrid of bottom-up assembly techniques supported by\nconventional lithographic patterning, we show how modestsized PLA logic blocks, which are efficient for implementing logic, can be organized into a segmented, Manhattan mesh interconnection scheme. The resulting programmable architecture has a macro-scale view which is reminiscent of lithographic FPGA and CPLD designs despite the fact that the low-level, sublithographic fabrication techniques used are much more highly constrained than conventional lithography and are prone to high defect rates. Using the Toronto 20 benchmark set, we begin to explore the design space for these sublithographic architectures and show that they may allow us to exploit nanowire building blocks to reach one to\ntwo orders of magnitude greater density than 22nm CMOS\nlithography.",
        "doi": "10.1145/1046192.1046210",
        "isbn": "1-59593-029-9",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2005-02",
        "pages": "127-137"
    },
    {
        "id": "authors:t087r-77f32",
        "collection": "authors",
        "collection_id": "t087r-77f32",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110822-103525220",
        "type": "book_section",
        "title": "A greedy algorithm for tolerating defective crosspoints in nanoPLA design",
        "book_title": "2004 IEEE International Conference on Field-Programmable Technology",
        "author": [
            {
                "family_name": "Naeimi",
                "given_name": "Helia",
                "clpid": "Naeimi-H"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "abstract": "Recent developments suggest both plausible fabrication techniques and viable architectures for building sublithographic programmable logic arrays using molecular-scale wires and switches. Designs at this scale will see much higher defect rates than in conventional lithography. However, these defects need not be an impediment to programmable logic design as this scale. We introduce a strategy for tolerating defective crosspoints and develop a linear-time, greedy algorithm for mapping PLA logic around crosspoint defects. We note that P-term fanin must be bounded to guarantee low overhead mapping and develop analytical guidelines for bounding fanin. We further quantify analytical and empirical mapping overhead rates. Including fanin bounding, our greedy mapping algorithm maps a large set of benchmark designs with 13% average overhead for random junction defect rates as high as 20%.",
        "doi": "10.1109/FPT.2004.1393250",
        "isbn": "0-7803-8651-5",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2004-12",
        "pages": "49-56"
    },
    {
        "id": "authors:vyd9b-vsa67",
        "collection": "authors",
        "collection_id": "vyd9b-vsa67",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20110817-151253900",
        "type": "book_section",
        "title": "Design patterns for reconfigurable computing",
        "book_title": "12th Annual IEEE Symposium on Field-Programmable Custom Computing Machines",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Adams",
                "given_name": "Joshua",
                "clpid": "Adams-J"
            },
            {
                "family_name": "DeLorimier",
                "given_name": "Michael",
                "clpid": "DeLorimier-M"
            },
            {
                "family_name": "Kapre",
                "given_name": "Nachiket",
                "clpid": "Kapre-N"
            },
            {
                "family_name": "Matsuda",
                "given_name": "Yuki",
                "clpid": "Matsuda-Y"
            },
            {
                "family_name": "Naeimi",
                "given_name": "Helia",
                "clpid": "Naeimi-H"
            },
            {
                "family_name": "Vanier",
                "given_name": "Michael",
                "clpid": "Vanier-M"
            },
            {
                "family_name": "Wrighton",
                "given_name": "Michael",
                "clpid": "Wrighton-M"
            }
        ],
        "contributor": [
            {
                "family_name": "Pocek",
                "given_name": "Kenneth L.",
                "clpid": "Pocek-K-L"
            },
            {
                "family_name": "Arnold",
                "given_name": "Jeffrey M.",
                "clpid": "Arnold-J-M"
            }
        ],
        "abstract": "It is valuable to identify and catalog design patterns\nfor reconfigurable computing. These design patterns are\ncanonical solutions to common and recurring design challenges which arise in reconfigurable systems and applications. The catalog can form the basis for creating designs, for educating new designers, for understanding the needs of tools and languages, and for discussing reconfigurable design. Tying application and implementation lessons to the expansion and refinement of this catalog will make those lessons more relevant to the design community. In this paper, we articulate this role for design patterns in reconfigurable computing, provide a few example patterns, offer a starting point for the contents of the catalog, and discuss the potential benefits of this effort.",
        "doi": "10.1109/FCCM.2004.29",
        "isbn": "0-7695-2230-0",
        "publisher": "IEEE",
        "place_of_publication": "Los Alamitos, CA",
        "publication_date": "2004-04",
        "pages": "13-23"
    },
    {
        "id": "authors:0az87-65x26",
        "collection": "authors",
        "collection_id": "0az87-65x26",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161006-125146350",
        "type": "book_section",
        "title": "Nanowire-Based Sublithographic Programmable Logic Arrays",
        "book_title": "FPGA '04 Proceedings of the 2004 ACM/SIGDA 12th international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Wilson",
                "given_name": "Michael J.",
                "clpid": "Wilson-M-J"
            }
        ],
        "contributor": [
            {
                "family_name": "Tessier",
                "given_name": "Russ",
                "clpid": "Tessier-R"
            },
            {
                "family_name": "Schmit",
                "given_name": "Herman",
                "clpid": "Schmit-H"
            }
        ],
        "abstract": "How can Programmable Logic Arrays (PLAs) be built without relying on lithography to pattern their smallest features? In this paper, we detail designs which exploit emerging, bottom-up material synthesis techniques to build PLAs\nusing molecular-scale nanowires. Our new designs accommodate technologies where the only post-fabrication programmable element is a non-restoring diode. We introduce\nstochastic techniques which allow us to restore the diode\nlogic at the nanoscale so that it can be cascaded and interconnected for general logic evaluation. Under conservative assumptions using 10nm nanowires and 90nm lithographic support, we project yielded logic density around 500,000nm^2/or term for a 60 or-term array; a complete 60-term, two-level PLA is roughly the same size as a single 4-LUT logic block in 22nm lithography. Each\nor term is comparable in area to a 4-transistor hardwired gate at 22nm. Mapping sample datapaths and conventional programmable logic benchmarks, we estimate that each 60-\nor-term PLA plane will provide equivalent logic to 5\u201310 4-input LUTs.",
        "doi": "10.1145/968280.968299",
        "isbn": "1-58113-829-6",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2004-02",
        "pages": "123-132"
    },
    {
        "id": "authors:6kk7e-36813",
        "collection": "authors",
        "collection_id": "6kk7e-36813",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161005-174825781",
        "type": "book_section",
        "title": "What is the Right Model for Programming and Using Modern FPGAs?",
        "book_title": "FPGA '04 Proceedings of the 2004 ACM/SIGDA 12th international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Hutchings",
                "given_name": "Brad",
                "clpid": "Hutchings-B"
            },
            {
                "family_name": "Rudusky",
                "given_name": "Daryl",
                "clpid": "Rudusky-D"
            },
            {
                "family_name": "Hwang",
                "given_name": "James",
                "clpid": "Hwang-James"
            },
            {
                "family_name": "Nikhil",
                "clpid": "Nikhil-"
            },
            {
                "family_name": "Raje",
                "given_name": "Salil",
                "clpid": "Raje-S"
            },
            {
                "family_name": "Stoica",
                "given_name": "Adrian",
                "clpid": "Stoica-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Tessier",
                "given_name": "Russ",
                "clpid": "Tessier-R"
            },
            {
                "family_name": "Schmit",
                "given_name": "Herman",
                "clpid": "Schmit-H"
            }
        ],
        "abstract": "Traditionally, FPGAs have been the bastard step-brother of ASICs. They have been forced to act like ASICs and fit themselves into the ASIC development model. This has meant ignoring their unique strengths: reprogrammability, late-binding and run-time reconfiguration. Today, however, FPGAs are becoming more acceptable for their own merits. The majority of new design starts are FPGA designs. As FPGAs rise from under the shadow of their aging brother, should they continue to try to wear his hand-me-downs? Or is it time to develop more suitable models that lets them shine? At the same time, the old ASIC model is not even serving ASICs well, and new models for developing ASICs are emerging. All of this may encourage us to rethink how we should be programming FPGA-based systems. Possibilities include:\n\n\u2022 Using the traditional, ASIC model -- it's tried and true, has demonstrated success\n\n\u2022 Using the traditional, sequential processor model -- Compile programs from C down to FPGAs...perhaps evolving FPGAs to better support\n\n\u2022 Using the emerging C-level design tools\n\n\u2022 Using concurrent and/or streaming models (incl. CSP, Matlab/simulink, Ptolemy, SCORE, Stream-C) Using a biologically inspired model (neural networks, genetic programming...)",
        "doi": "10.1145/968280.968281",
        "isbn": "1-58113-829-6",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2004-02",
        "pages": "119"
    },
    {
        "id": "authors:xz519-c3423",
        "collection": "authors",
        "collection_id": "xz519-c3423",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161213-164304748",
        "type": "book_section",
        "title": "Design of FPGA interconnect for multilevel metalization",
        "book_title": "FPGA '03 Proceedings of the 2003 ACM/SIGDA eleventh international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "Rubin",
                "given_name": "Raphael",
                "clpid": "Rubin-R"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Trimberger",
                "given_name": "Steve",
                "clpid": "Trimberger-S"
            }
        ],
        "abstract": "How does multilevel metalization impact the design of FPGA interconnect? The availability of a growing number of metal layers presents the opportunity to use wiring in the third-dimension to reduce switch requirements. Unfortunately, traditional FPGA wiring schemes are not designed to exploit these additional metal layers. We introduce an alternate topology, based on Leighton's Mesh-of-Trees, which carefully exploits hierarchy to allow additional metal layers to support arbitrary device scaling. When wiring layers grow sufficiently fast with aggregate network size (N), our network requires only O(N) area; this is in stark contrast to traditional, Manhattan FPGA routing schemes where switching requirements alone grow superlinearly in N. In practice, we show that, even for the admittedly small designs in the Toronto \"FPGA Place and Route Challenge,\" the Mesh-of-Trees networks require 10% less switches than the standard, Manhattan FPGA routing scheme.",
        "doi": "10.1145/611817.611841",
        "isbn": "1-58113-651-X",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2003-02",
        "pages": "154-163"
    },
    {
        "id": "authors:jbpnj-0k057",
        "collection": "authors",
        "collection_id": "jbpnj-0k057",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20170109-150810119",
        "type": "book_section",
        "title": "Hardware-assisted simulated annealing with application for fast FPGA placement",
        "book_title": "FPGA '03 Proceedings of the 2003 ACM/SIGDA eleventh international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "Wrighton",
                "given_name": "Michael G.",
                "clpid": "Wrighton-M-G"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9 M.",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Trimberger",
                "given_name": "Steve",
                "clpid": "Trimberger-S"
            }
        ],
        "abstract": "To truly exploit FPGAs for rapid turn-around development and prototyping, placement times must be reduced to seconds; late-bound, reconfigurable computing applications may demand placement times as short as microseconds. In this paper, we show how a systolic structure can accelerate placement by assigning one processing element to each possible location for an FPGA LUT from a design netlist. We demonstrate that our technique approaches the same quality point as traditional simulated annealing as measured by a simple linear wirelength metric. Experimental results look ahead to compare quality against VPR's fast placer when considering the minimum channel width required to route as the primary optimization criteria. Preliminary results from an FPGA implementation show the feasibility of accelerating simulated annealing by three orders of magnitude using this approach. This means we can place the largest design in the University of Toronto's \"FPGA Placement and Routing Challenge\" in around 4ms.",
        "doi": "10.1145/611817.611824",
        "isbn": "1-58113-651-X",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2003-02",
        "pages": "33-42"
    },
    {
        "id": "authors:pb7bz-0jq38",
        "collection": "authors",
        "collection_id": "pb7bz-0jq38",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161107-154348196",
        "type": "book_section",
        "title": "Stochastic, spatial routing for hypergraphs, trees, and meshes",
        "book_title": "FPGA '03 Proceedings of the 2003 ACM/SIGDA eleventh international symposium on Field programmable gate arrays",
        "author": [
            {
                "family_name": "Huang",
                "given_name": "Randy",
                "clpid": "Huang-Randy"
            },
            {
                "family_name": "Wawrzynek",
                "given_name": "John",
                "clpid": "Wawrzynek-J"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Trimberger",
                "given_name": "Steve",
                "clpid": "Trimberger-S"
            }
        ],
        "abstract": "FPGA place and route is time consuming, often serving as the major obstacle inhibiting a fast edit-compile-test loop in prototyping and development and the major obstacle preventing late-bound hardware and design mapping for reconfigurable systems. Previous work showed that hardware-assisted routing can accelerate fanout-free routing on Fat-Trees by three orders of magnitude with modest modifications to the network itself. In this paper, we show how these techniques can be applied to any FPGA and how they can be implemented on top of LUT networks in cases where modification of the FPGA itself is not justified. We further show how to accommodate fanout and how to achieve comparable route quality to software-based methods. For a tree network, we estimate an FPGA implementation of our routing logic could route the Toronto Place and Route Benchmarks at least two orders of magnitude faster than a software Pathfinder while achieving within 3% of the aggregate quality. Preliminary results on small mesh benchmarks achieve within one track of vpr-fast.",
        "doi": "10.1145/611817.611830",
        "isbn": "1-58113-651-X",
        "publisher": "ACM",
        "place_of_publication": "New York, CA",
        "publication_date": "2003-02",
        "pages": "78-87"
    },
    {
        "id": "authors:bnmwd-wvz15",
        "collection": "authors",
        "collection_id": "bnmwd-wvz15",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20160812-112140221",
        "type": "book_section",
        "title": "Molecular electronics: devices, systems and tools for gigagate, gigabit chips",
        "book_title": "IEEE/ACM International Conference on Computer Aided Design, 2002. ICCAD 2002",
        "author": [
            {
                "family_name": "Butts",
                "given_name": "Michael",
                "clpid": "Butts-M"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            },
            {
                "family_name": "Goldstein",
                "given_name": "Seth Copen",
                "clpid": "Goldstein-S-C"
            }
        ],
        "abstract": "New electronics technologies are emerging which may carry us beyond the limits of lithographic processing down to molecular-scale feature sizes. Devices and interconnects can be made from a variety of molecules and materials including bistable and switchable organic molecules, carbon nanotubes, and, single-crystal semiconductor nanowires. They can be self-assembled into organized structures and attached onto lithographic substrates. This tutorial reviews emerging molecular-scale electronics technology for CAD and system designers and highlights where ICCAD research can help support this technology.",
        "doi": "10.1109/ICCAD.2002.1167569",
        "isbn": "0-7803-7607-2",
        "publisher": "IEEE",
        "place_of_publication": "Piscataway, NJ",
        "publication_date": "2002-11",
        "pages": "443-440"
    },
    {
        "id": "authors:h511y-26m36",
        "collection": "authors",
        "collection_id": "h511y-26m36",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20200127-123636117",
        "type": "book_section",
        "title": "Very Large Scale Spatial Computing",
        "book_title": "Unconventional Models of Computation",
        "author": [
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Calude",
                "given_name": "C. S.",
                "clpid": "Calude-C-S"
            }
        ],
        "abstract": "The early decades of computing were marked by limited re sources. However, as we enter the twenty-first century, silicon is offering enormous computing resources on a single die and molecular-scale devices appear plausible offering a path to even greater capacities. Exploiting the capacities of these modern and future devices demands different compu tational models and radical shifts in the way we organize, capture, and optimize computations. A key shift is toward spatially organized compu tation. A natural consequence is that the dominant effects which govern our computing space change from the total number of operations and temporal locality to interconnect complexity and spatial locality. Old computational models which hide, ignore, or obfuscate communication and emphasize temporal sequences inhibit the exploitation of these mod ern capacities, motivating the need for new models which make commu nication and spatial organization more apparent.",
        "doi": "10.1007/3-540-45833-6_3",
        "isbn": "978-3-540-44311-7",
        "publisher": "Springer",
        "place_of_publication": "Berlin",
        "publication_date": "2002-10-10",
        "pages": "27-37"
    },
    {
        "id": "authors:9nkea-5nk06",
        "collection": "authors",
        "collection_id": "9nkea-5nk06",
        "cite_using_url": "https://resolver.caltech.edu/CaltechAUTHORS:20161129-173323435",
        "type": "book_section",
        "title": "Analysis of quasi-static scheduling techniques in a virtualized reconfigurable machine",
        "book_title": "FPGA '02 Proceedings of the 2002 ACM/SIGDA tenth international symposium on Field-programmable gate arrays",
        "author": [
            {
                "family_name": "Markovskiy",
                "given_name": "Yury",
                "clpid": "Markovskiy-Y"
            },
            {
                "family_name": "Caspi",
                "given_name": "Eylon",
                "clpid": "Caspi-E"
            },
            {
                "family_name": "Huang",
                "given_name": "Randy",
                "clpid": "Huang-Randy"
            },
            {
                "family_name": "Yeh",
                "given_name": "Joseph",
                "clpid": "Yeh-Joseph"
            },
            {
                "family_name": "Chu",
                "given_name": "Michael",
                "clpid": "Chu-Michael"
            },
            {
                "family_name": "Wawrzynek",
                "given_name": "John",
                "clpid": "Wawrzynek-J"
            },
            {
                "family_name": "DeHon",
                "given_name": "Andr\u00e9",
                "clpid": "DeHon-A"
            }
        ],
        "contributor": [
            {
                "family_name": "Schlag",
                "given_name": "Martine",
                "clpid": "Schlag-"
            },
            {
                "family_name": "Trimberger",
                "given_name": "Steve",
                "clpid": "Trimberger-S"
            }
        ],
        "abstract": "he SCORE compute model uses fixed-size, virtual compute and memory pages connected by stream links to capture the definition of a computation abstracted from the detailed size of the physical hardware. When the number of physical compute pages is smaller than the number of virtual compute pages in the abstract computation graph, the design is time-multiplexed onto the available physical hardware. A key component of this strategy is an automatic scheduler that selects the temporal sequencing of virtual resources onto the physical device. We describe a quasi-static scheduling strategy that retains the full semantic power of the dynamic SCORE flow graph while taking advantage of static scheduling techniques at program load time to hoist most of the computational work out of the inner scheduling loops. This strategy reduces online scheduling work per reconfiguration epoch by an order of magnitude. In addition, a more global perspective available from offline-scheduling improves schedule quality, resulting in a net reduction of total execution time by 46-81%.",
        "doi": "10.1145/503048.503077",
        "isbn": "1-58113-452-5",
        "publisher": "ACM",
        "place_of_publication": "New York, NY",
        "publication_date": "2002-02",
        "pages": "196-205"
    }
]