[arch] update arch files

2022-08-22 18:24:37 -07:00 · 2022-08-22 18:24:37 -07:00 · bdb051f787
parent 6c44f321e5
commit bdb051f787
55 changed files with 29354 additions and 29409 deletions
--- a/openfpga_flow/vpr_arch/k4_N4_tileableIO_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileableIO_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -12,9 +12,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,63 +22,63 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="top">io.outpad</loc>
-        <loc side="right">io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!-- Perimeter of 'EMPTY' blocks -->
-      <perimeter type="EMPTY" priority="100"/>
-      <!--Fill with 'io'-->
-      <fill type="io" priority="10"/>
-      <!-- Build an inner region of clbs -->
-      <region type="clb" startx="2" endx="W-3" starty="2" endy="H-3" priority="101"/> 
-    </auto_layout>
-    <fixed_layout name="2x2" width="6" height="6">
-      <!-- Perimeter of 'EMPTY' blocks -->
-      <perimeter type="EMPTY" priority="100"/>
-      <!--Fill with 'io'-->
-      <fill type="io" priority="10"/>
-      <!-- Build an inner region of clbs -->
-      <region type="clb" startx="2" endx="W-3" starty="2" endy="H-3" priority="101"/> 
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="top">io.outpad</loc>     
+             <loc side="right">io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="10" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!-- Perimeter of 'EMPTY' blocks -->   
+         <perimeter type="EMPTY" priority="100"/>   
+         <!--Fill with 'io'-->   
+         <fill type="io" priority="10"/>   
+         <!-- Build an inner region of clbs -->   
+         <region type="clb" startx="2" endx="W-3" starty="2" endy="H-3" priority="101"/>    
+      </auto_layout>  
+      <fixed_layout name="2x2" width="6" height="6">   
+         <!-- Perimeter of 'EMPTY' blocks -->   
+         <perimeter type="EMPTY" priority="100"/>   
+         <!--Fill with 'io'-->   
+         <fill type="io" priority="10"/>   
+         <!-- Build an inner region of clbs -->   
+         <region type="clb" startx="2" endx="W-3" starty="2" endy="H-3" priority="101"/>    
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -94,20 +93,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -120,80 +119,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -202,68 +201,68 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="10" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -272,22 +271,22 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N4_tileable_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileable_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -12,9 +12,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,84 +22,84 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-    <fixed_layout name="4x4" width="6" height="6">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-    <fixed_layout name="48x48" width="50" height="50">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-    <fixed_layout name="96x96" width="98" height="98">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="10" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+      <fixed_layout name="4x4" width="6" height="6">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+      <fixed_layout name="48x48" width="50" height="50">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+      <fixed_layout name="96x96" width="98" height="98">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -115,20 +114,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -141,80 +140,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -223,68 +222,68 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="10" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -293,22 +292,22 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N4_tileable_GlobalTile4Clk_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileable_GlobalTile4Clk_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -13,9 +13,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -24,65 +23,65 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="4"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">
-        <fc_override port_name="clk" fc_type="frac" fc_val="0"/>
-      </fc>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="10" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="none"/>    
+          <clock name="clk" num_pins="4"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">     
+             <fc_override port_name="clk" fc_type="frac" fc_val="0"/>     
+          </fc>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -97,20 +96,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -123,80 +122,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -205,68 +204,68 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="4"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="10" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="none"/>   
+         <clock name="clk" num_pins="4"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -275,22 +274,22 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N4_tileable_GlobalTile8Clk_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileable_GlobalTile8Clk_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -13,9 +13,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -24,65 +23,65 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="8"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">
-        <fc_override port_name="clk" fc_type="frac" fc_val="0"/>
-      </fc>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="10" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="none"/>    
+          <clock name="clk" num_pins="8"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">     
+             <fc_override port_name="clk" fc_type="frac" fc_val="0"/>     
+          </fc>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -97,20 +96,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -123,80 +122,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -205,68 +204,68 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="8"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="10" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="none"/>   
+         <clock name="clk" num_pins="8"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -275,22 +274,22 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N4_tileable_GlobalTileClk_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileable_GlobalTileClk_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -12,9 +12,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,65 +22,65 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">
-        <fc_override port_name="clk" fc_type="frac" fc_val="0"/>
-      </fc>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="10" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">     
+             <fc_override port_name="clk" fc_type="frac" fc_val="0"/>     
+          </fc>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -96,20 +95,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -122,80 +121,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -204,68 +203,68 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="10" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -274,22 +273,22 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N4_tileable_GlobalTileClk_registerable_io_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileable_GlobalTileClk_registerable_io_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -12,9 +12,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,68 +22,68 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">
-        <fc_override port_name="clk" fc_type="frac" fc_val="0"/>
-      </fc>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad io.clk</loc>
-        <loc side="top">io.outpad io.inpad io.clk</loc>
-        <loc side="right">io.outpad io.inpad io.clk</loc>
-        <loc side="bottom">io.outpad io.inpad io.clk</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">
-        <fc_override port_name="clk" fc_type="frac" fc_val="0"/>
-      </fc>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">     
+             <fc_override port_name="clk" fc_type="frac" fc_val="0"/>     
+          </fc>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad io.clk</loc>     
+             <loc side="top">io.outpad io.inpad io.clk</loc>     
+             <loc side="right">io.outpad io.inpad io.clk</loc>     
+             <loc side="bottom">io.outpad io.inpad io.clk</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="10" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">     
+             <fc_override port_name="clk" fc_type="frac" fc_val="0"/>     
+          </fc>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -99,20 +98,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -125,111 +124,111 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-          <input name="D" num_pins="1" port_class="D"/>
-          <output name="Q" num_pins="1" port_class="Q"/>
-          <clock name="clk" num_pins="1" port_class="clock"/>
-          <T_setup value="66e-12" port="ff.D" clock="clk"/>
-          <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-        </pb_type>
-        <interconnect>
-          <direct name="clk" input="io.clk" output="ff.clk"/>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <!-- Create a selector between registered/combinational I/O -->
-          <direct name="inpad" input="iopad.inpad" output="ff.D"/>
-          <mux name="mux1" input="iopad.inpad ff.Q" output="io.inpad">
-            <delay_constant max="4.5e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-            <delay_constant max="4.243e-11" in_port="ff.Q" out_port="io.inpad"/>
-          </mux>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">     
+               <input name="D" num_pins="1" port_class="D"/>     
+               <output name="Q" num_pins="1" port_class="Q"/>     
+               <clock name="clk" num_pins="1" port_class="clock"/>     
+               <T_setup value="66e-12" port="ff.D" clock="clk"/>     
+               <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="clk" input="io.clk" output="ff.clk"/>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <!-- Create a selector between registered/combinational I/O -->     
+               <direct name="inpad" input="iopad.inpad" output="ff.D"/>     
+               <mux name="mux1" input="iopad.inpad ff.Q" output="io.inpad">      
+                  <delay_constant max="4.5e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+                  <delay_constant max="4.243e-11" in_port="ff.Q" out_port="io.inpad"/>      
+               </mux>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="inpad_registered">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-          <input name="D" num_pins="1" port_class="D"/>
-          <output name="Q" num_pins="1" port_class="Q"/>
-          <clock name="clk" num_pins="1" port_class="clock"/>
-          <T_setup value="66e-12" port="ff.D" clock="clk"/>
-          <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-        </pb_type>
-        <interconnect>
-          <direct name="clk" input="io.clk" output="ff.clk"/>
-          <direct name="inpad" input="inpad.inpad" output="ff.D">
-            <pack_pattern name="registered_io" in_port="inpad.inpad" out_port="ff.D"/>
-          </direct>
-          <direct name="ff2inpad" input="ff.Q" output="io.inpad"/>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="inpad_registered">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">     
+               <input name="D" num_pins="1" port_class="D"/>     
+               <output name="Q" num_pins="1" port_class="Q"/>     
+               <clock name="clk" num_pins="1" port_class="clock"/>     
+               <T_setup value="66e-12" port="ff.D" clock="clk"/>     
+               <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="clk" input="io.clk" output="ff.clk"/>     
+               <direct name="inpad" input="inpad.inpad" output="ff.D">      
+                  <pack_pattern name="registered_io" in_port="inpad.inpad" out_port="ff.D"/>      
+               </direct>     
+               <direct name="ff2inpad" input="ff.Q" output="io.inpad"/>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -238,68 +237,68 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="10" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -308,22 +307,22 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N4_tileable_TileOrgzBr_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileable_TileOrgzBr_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -12,9 +12,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,68 +22,68 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left"></loc>
-        <loc side="top"></loc>
-        <loc side="right">clb.I[5:9] clb.O[2:3]</loc>
-        <loc side="bottom">clb.clk clb.I[0:4] clb.O[0:1]</loc>
-      </pinlocations>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="10" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left"/>     
+             <loc side="top"/>     
+             <loc side="right">clb.I[5:9] clb.O[2:3]</loc>     
+             <loc side="bottom">clb.clk clb.I[0:4] clb.O[0:1]</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -99,20 +98,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -125,80 +124,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -207,68 +206,68 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="10" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -277,22 +276,22 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N4_tileable_TileOrgzTl_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileable_TileOrgzTl_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -12,9 +12,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,68 +22,68 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="top">clb.clk clb.I[0:4] clb.O[0:1]</loc>
-        <loc side="left">clb.I[5:9] clb.O[2:3]</loc>
-        <loc side="right"></loc>
-        <loc side="bottom"></loc>
-      </pinlocations>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="10" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="top">clb.clk clb.I[0:4] clb.O[0:1]</loc>     
+             <loc side="left">clb.I[5:9] clb.O[2:3]</loc>     
+             <loc side="right"/>     
+             <loc side="bottom"/>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -99,20 +98,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -125,80 +124,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -207,68 +206,68 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="10" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -277,22 +276,22 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N4_tileable_TileOrgzTr_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileable_TileOrgzTr_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -12,9 +12,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,68 +22,68 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left"></loc>
-        <loc side="top">clb.clk clb.I[0:4] clb.O[0:1]</loc>
-        <loc side="right">clb.I[5:9] clb.O[2:3]</loc>
-        <loc side="bottom"></loc>
-      </pinlocations>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="10" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left"/>     
+             <loc side="top">clb.clk clb.I[0:4] clb.O[0:1]</loc>     
+             <loc side="right">clb.I[5:9] clb.O[2:3]</loc>     
+             <loc side="bottom"/>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -99,20 +98,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -125,80 +124,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -207,68 +206,68 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="10" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -277,22 +276,22 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N4_tileable_dsp8reg_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileable_dsp8reg_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -15,9 +15,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -26,100 +25,100 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-    <model name="mult_8">
-      <input_ports>
-        <port name="A" combinational_sink_ports="Y"/>
-        <port name="B" combinational_sink_ports="Y"/>
-      </input_ports>
-      <output_ports>
-        <port name="Y"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-    <tile name="mult_8" height="2" area="396000">
-      <equivalent_sites>
-        <site pb_type="mult_8" pin_mapping="direct"/>
-      </equivalent_sites>
-      <input name="a" num_pins="8"/>
-      <input name="b" num_pins="8"/>
-      <output name="out" num_pins="16"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <!--  Highly recommand to customize pin location when direct connection is used!!! -->
-      <!-- pinlocations are designed to spread pin on 4 sides evenly -->
-      <pinlocations pattern="custom">
-        <loc side="left">mult_8.a[0:3] mult_8.b[0:3] mult_8.out[0:7]</loc>
-        <loc side="top">mult_8.clk</loc>
-        <loc side="right">mult_8.a[4:7] mult_8.b[4:7] mult_8.out[8:15]</loc>
-        <loc side="bottom"></loc>
-      </pinlocations>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-    <fixed_layout name="3x2" width="5" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-      <!--Column of 'memory' with 'EMPTY' blocks wherever a 'memory' does not fit. Vertical offset by 1 for perimeter.-->
-      <col type="mult_8" startx="2" starty="1" repeatx="8" priority="20"/>
-      <col type="EMPTY" startx="2" repeatx="8" starty="1" priority="19"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+      <model name="mult_8">   
+         <input_ports>    
+            <port name="A" combinational_sink_ports="Y"/>    
+            <port name="B" combinational_sink_ports="Y"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="Y"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="10" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+      <tile name="mult_8" height="2" area="396000">   <sub_tile name="mult_8">    
+          <equivalent_sites>     
+             <site pb_type="mult_8" pin_mapping="direct"/>     
+          </equivalent_sites>    
+          <input name="a" num_pins="8"/>    
+          <input name="b" num_pins="8"/>    
+          <output name="out" num_pins="16"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <!--  Highly recommand to customize pin location when direct connection is used!!! -->    
+          <!-- pinlocations are designed to spread pin on 4 sides evenly -->    
+          <pinlocations pattern="custom">     
+             <loc side="left">mult_8.a[0:3] mult_8.b[0:3] mult_8.out[0:7]</loc>     
+             <loc side="top">mult_8.clk</loc>     
+             <loc side="right">mult_8.a[4:7] mult_8.b[4:7] mult_8.out[8:15]</loc>     
+             <loc side="bottom"/>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+      <fixed_layout name="3x2" width="5" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+         <!--Column of 'memory' with 'EMPTY' blocks wherever a 'memory' does not fit. Vertical offset by 1 for perimeter.-->   
+         <col type="mult_8" startx="2" starty="1" repeatx="8" priority="20"/>   
+         <col type="EMPTY" startx="2" repeatx="8" starty="1" priority="19"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -134,20 +133,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -160,80 +159,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -242,68 +241,68 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="10" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -312,131 +311,131 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-    <!-- Define 8-bit multiplier with input and output registers begin -->
-    <pb_type name="mult_8">
-      <input name="a" num_pins="8"/>
-      <input name="b" num_pins="8"/>
-      <clock name="clk" num_pins="1"/>
-      <output name="out" num_pins="16"/>
-      <mode name="mult_8x8">
-        <pb_type name="mult_8x8_slice" num_pb="1">
-          <input name="A_cfg" num_pins="8"/>
-          <input name="B_cfg" num_pins="8"/>
-          <output name="OUT_cfg" num_pins="16"/>
-          <clock name="clk" num_pins="1"/>
-          <pb_type name="mult_8x8" blif_model=".subckt mult_8" num_pb="1">
-            <input name="A" num_pins="8"/>
-            <input name="B" num_pins="8"/>
-            <output name="Y" num_pins="16"/>
-            <delay_constant max="1.523e-9" min="0.776e-9" in_port="mult_8x8.A" out_port="mult_8x8.Y"/>
-            <delay_constant max="1.523e-9" min="0.776e-9" in_port="mult_8x8.B" out_port="mult_8x8.Y"/>
-          </pb_type>
-          <pb_type name="ff_A" blif_model=".latch" num_pb="8" class="flipflop">
-            <input name="D" num_pins="1" port_class="D"/>
-            <output name="Q" num_pins="1" port_class="Q"/>
-            <clock name="clk" num_pins="1" port_class="clock"/>
-            <T_setup value="66e-12" port="ff_A.D" clock="clk"/>
-            <T_clock_to_Q max="124e-12" port="ff_A.Q" clock="clk"/>
-          </pb_type>
-          <pb_type name="ff_B" blif_model=".latch" num_pb="8" class="flipflop">
-            <input name="D" num_pins="1" port_class="D"/>
-            <output name="Q" num_pins="1" port_class="Q"/>
-            <clock name="clk" num_pins="1" port_class="clock"/>
-            <T_setup value="66e-12" port="ff_B.D" clock="clk"/>
-            <T_clock_to_Q max="124e-12" port="ff_B.Q" clock="clk"/>
-          </pb_type>
-          <pb_type name="ff_Y" blif_model=".latch" num_pb="16" class="flipflop">
-            <input name="D" num_pins="1" port_class="D"/>
-            <output name="Q" num_pins="1" port_class="Q"/>
-            <clock name="clk" num_pins="1" port_class="clock"/>
-            <T_setup value="66e-12" port="ff_Y.D" clock="clk"/>
-            <T_clock_to_Q max="124e-12" port="ff_Y.Q" clock="clk"/>
-          </pb_type>
-          <interconnect>
-            <mux name="a2a0" input="mult_8x8_slice.A_cfg[0] ff_A[0].Q" output="mult_8x8.A[0]"/>
-            <mux name="a2a1" input="mult_8x8_slice.A_cfg[1] ff_A[1].Q" output="mult_8x8.A[1]"/>
-            <mux name="a2a2" input="mult_8x8_slice.A_cfg[2] ff_A[2].Q" output="mult_8x8.A[2]"/>
-            <mux name="a2a3" input="mult_8x8_slice.A_cfg[3] ff_A[3].Q" output="mult_8x8.A[3]"/>
-            <mux name="a2a4" input="mult_8x8_slice.A_cfg[4] ff_A[4].Q" output="mult_8x8.A[4]"/>
-            <mux name="a2a5" input="mult_8x8_slice.A_cfg[5] ff_A[5].Q" output="mult_8x8.A[5]"/>
-            <mux name="a2a6" input="mult_8x8_slice.A_cfg[6] ff_A[6].Q" output="mult_8x8.A[6]"/>
-            <mux name="a2a7" input="mult_8x8_slice.A_cfg[7] ff_A[7].Q" output="mult_8x8.A[7]"/>
-            <direct name="a2ff" input="mult_8x8_slice.A_cfg[7:0]" output="ff_A[7:0].D"/>
-            <mux name="b2b0" input="mult_8x8_slice.B_cfg[0] ff_B[0].Q" output="mult_8x8.B[0]"/>
-            <mux name="b2b1" input="mult_8x8_slice.B_cfg[1] ff_B[1].Q" output="mult_8x8.B[1]"/>
-            <mux name="b2b2" input="mult_8x8_slice.B_cfg[2] ff_B[2].Q" output="mult_8x8.B[2]"/>
-            <mux name="b2b3" input="mult_8x8_slice.B_cfg[3] ff_B[3].Q" output="mult_8x8.B[3]"/>
-            <mux name="b2b4" input="mult_8x8_slice.B_cfg[4] ff_B[4].Q" output="mult_8x8.B[4]"/>
-            <mux name="b2b5" input="mult_8x8_slice.B_cfg[5] ff_B[5].Q" output="mult_8x8.B[5]"/>
-            <mux name="b2b6" input="mult_8x8_slice.B_cfg[6] ff_B[6].Q" output="mult_8x8.B[6]"/>
-            <mux name="b2b7" input="mult_8x8_slice.B_cfg[7] ff_B[7].Q" output="mult_8x8.B[7]"/>
-            <direct name="b2ff" input="mult_8x8_slice.B_cfg[7:0]" output="ff_B[7:0].D"/>
-            <mux name="out2out0" input="mult_8x8.Y[0] ff_Y[0].Q" output="mult_8x8_slice.OUT_cfg[0]"/>
-            <mux name="out2out1" input="mult_8x8.Y[1] ff_Y[1].Q" output="mult_8x8_slice.OUT_cfg[1]"/>
-            <mux name="out2out2" input="mult_8x8.Y[2] ff_Y[2].Q" output="mult_8x8_slice.OUT_cfg[2]"/>
-            <mux name="out2out3" input="mult_8x8.Y[3] ff_Y[3].Q" output="mult_8x8_slice.OUT_cfg[3]"/>
-            <mux name="out2out4" input="mult_8x8.Y[4] ff_Y[4].Q" output="mult_8x8_slice.OUT_cfg[4]"/>
-            <mux name="out2out5" input="mult_8x8.Y[5] ff_Y[5].Q" output="mult_8x8_slice.OUT_cfg[5]"/>
-            <mux name="out2out6" input="mult_8x8.Y[6] ff_Y[6].Q" output="mult_8x8_slice.OUT_cfg[6]"/>
-            <mux name="out2out7" input="mult_8x8.Y[7] ff_Y[7].Q" output="mult_8x8_slice.OUT_cfg[7]"/>
-            <mux name="out2out8" input="mult_8x8.Y[8] ff_Y[8].Q" output="mult_8x8_slice.OUT_cfg[8]"/>
-            <mux name="out2out9" input="mult_8x8.Y[9] ff_Y[9].Q" output="mult_8x8_slice.OUT_cfg[9]"/>
-            <mux name="out2out10" input="mult_8x8.Y[10] ff_Y[10].Q" output="mult_8x8_slice.OUT_cfg[10]"/>
-            <mux name="out2out11" input="mult_8x8.Y[11] ff_Y[11].Q" output="mult_8x8_slice.OUT_cfg[11]"/>
-            <mux name="out2out12" input="mult_8x8.Y[12] ff_Y[12].Q" output="mult_8x8_slice.OUT_cfg[12]"/>
-            <mux name="out2out13" input="mult_8x8.Y[13] ff_Y[13].Q" output="mult_8x8_slice.OUT_cfg[13]"/>
-            <mux name="out2out14" input="mult_8x8.Y[14] ff_Y[14].Q" output="mult_8x8_slice.OUT_cfg[14]"/>
-            <mux name="out2out15" input="mult_8x8.Y[15] ff_Y[15].Q" output="mult_8x8_slice.OUT_cfg[15]"/>
-            <direct name="out2ff" input="mult_8x8.Y[15:0]" output="ff_Y[15:0].D"/>
-            <complete name="clk_ff_A" input="mult_8x8_slice.clk" output="ff_A.clk"/>
-            <complete name="clk_ff_B" input="mult_8x8_slice.clk" output="ff_B.clk"/>
-            <complete name="clk_ff_Y" input="mult_8x8_slice.clk" output="ff_Y.clk"/>
-          </interconnect>
-          <power method="pin-toggle">
-            <port name="A_cfg" energy_per_toggle="2.13e-12"/>
-            <port name="B_cfg" energy_per_toggle="2.13e-12"/>
-            <static_power power_per_instance="0.0"/>
-          </power>
-        </pb_type>
-        <interconnect>
-          <!-- Stratix IV input delay of 207ps is conservative for this architecture because this architecture does not have an input crossbar in the multiplier. 
+            <direct name="clbouts1" input="fle[3:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+      <!-- Define 8-bit multiplier with input and output registers begin -->  
+      <pb_type name="mult_8">   
+         <input name="a" num_pins="8"/>   
+         <input name="b" num_pins="8"/>   
+         <clock name="clk" num_pins="1"/>   
+         <output name="out" num_pins="16"/>   
+         <mode name="mult_8x8">    
+            <pb_type name="mult_8x8_slice" num_pb="1">     
+               <input name="A_cfg" num_pins="8"/>     
+               <input name="B_cfg" num_pins="8"/>     
+               <output name="OUT_cfg" num_pins="16"/>     
+               <clock name="clk" num_pins="1"/>     
+               <pb_type name="mult_8x8" blif_model=".subckt mult_8" num_pb="1">      
+                  <input name="A" num_pins="8"/>      
+                  <input name="B" num_pins="8"/>      
+                  <output name="Y" num_pins="16"/>      
+                  <delay_constant max="1.523e-9" min="0.776e-9" in_port="mult_8x8.A" out_port="mult_8x8.Y"/>      
+                  <delay_constant max="1.523e-9" min="0.776e-9" in_port="mult_8x8.B" out_port="mult_8x8.Y"/>      
+               </pb_type>     
+               <pb_type name="ff_A" blif_model=".latch" num_pb="8" class="flipflop">      
+                  <input name="D" num_pins="1" port_class="D"/>      
+                  <output name="Q" num_pins="1" port_class="Q"/>      
+                  <clock name="clk" num_pins="1" port_class="clock"/>      
+                  <T_setup value="66e-12" port="ff_A.D" clock="clk"/>      
+                  <T_clock_to_Q max="124e-12" port="ff_A.Q" clock="clk"/>      
+               </pb_type>     
+               <pb_type name="ff_B" blif_model=".latch" num_pb="8" class="flipflop">      
+                  <input name="D" num_pins="1" port_class="D"/>      
+                  <output name="Q" num_pins="1" port_class="Q"/>      
+                  <clock name="clk" num_pins="1" port_class="clock"/>      
+                  <T_setup value="66e-12" port="ff_B.D" clock="clk"/>      
+                  <T_clock_to_Q max="124e-12" port="ff_B.Q" clock="clk"/>      
+               </pb_type>     
+               <pb_type name="ff_Y" blif_model=".latch" num_pb="16" class="flipflop">      
+                  <input name="D" num_pins="1" port_class="D"/>      
+                  <output name="Q" num_pins="1" port_class="Q"/>      
+                  <clock name="clk" num_pins="1" port_class="clock"/>      
+                  <T_setup value="66e-12" port="ff_Y.D" clock="clk"/>      
+                  <T_clock_to_Q max="124e-12" port="ff_Y.Q" clock="clk"/>      
+               </pb_type>     
+               <interconnect>      
+                  <mux name="a2a0" input="mult_8x8_slice.A_cfg[0] ff_A[0].Q" output="mult_8x8.A[0]"/>      
+                  <mux name="a2a1" input="mult_8x8_slice.A_cfg[1] ff_A[1].Q" output="mult_8x8.A[1]"/>      
+                  <mux name="a2a2" input="mult_8x8_slice.A_cfg[2] ff_A[2].Q" output="mult_8x8.A[2]"/>      
+                  <mux name="a2a3" input="mult_8x8_slice.A_cfg[3] ff_A[3].Q" output="mult_8x8.A[3]"/>      
+                  <mux name="a2a4" input="mult_8x8_slice.A_cfg[4] ff_A[4].Q" output="mult_8x8.A[4]"/>      
+                  <mux name="a2a5" input="mult_8x8_slice.A_cfg[5] ff_A[5].Q" output="mult_8x8.A[5]"/>      
+                  <mux name="a2a6" input="mult_8x8_slice.A_cfg[6] ff_A[6].Q" output="mult_8x8.A[6]"/>      
+                  <mux name="a2a7" input="mult_8x8_slice.A_cfg[7] ff_A[7].Q" output="mult_8x8.A[7]"/>      
+                  <direct name="a2ff" input="mult_8x8_slice.A_cfg[7:0]" output="ff_A[7:0].D"/>      
+                  <mux name="b2b0" input="mult_8x8_slice.B_cfg[0] ff_B[0].Q" output="mult_8x8.B[0]"/>      
+                  <mux name="b2b1" input="mult_8x8_slice.B_cfg[1] ff_B[1].Q" output="mult_8x8.B[1]"/>      
+                  <mux name="b2b2" input="mult_8x8_slice.B_cfg[2] ff_B[2].Q" output="mult_8x8.B[2]"/>      
+                  <mux name="b2b3" input="mult_8x8_slice.B_cfg[3] ff_B[3].Q" output="mult_8x8.B[3]"/>      
+                  <mux name="b2b4" input="mult_8x8_slice.B_cfg[4] ff_B[4].Q" output="mult_8x8.B[4]"/>      
+                  <mux name="b2b5" input="mult_8x8_slice.B_cfg[5] ff_B[5].Q" output="mult_8x8.B[5]"/>      
+                  <mux name="b2b6" input="mult_8x8_slice.B_cfg[6] ff_B[6].Q" output="mult_8x8.B[6]"/>      
+                  <mux name="b2b7" input="mult_8x8_slice.B_cfg[7] ff_B[7].Q" output="mult_8x8.B[7]"/>      
+                  <direct name="b2ff" input="mult_8x8_slice.B_cfg[7:0]" output="ff_B[7:0].D"/>      
+                  <mux name="out2out0" input="mult_8x8.Y[0] ff_Y[0].Q" output="mult_8x8_slice.OUT_cfg[0]"/>      
+                  <mux name="out2out1" input="mult_8x8.Y[1] ff_Y[1].Q" output="mult_8x8_slice.OUT_cfg[1]"/>      
+                  <mux name="out2out2" input="mult_8x8.Y[2] ff_Y[2].Q" output="mult_8x8_slice.OUT_cfg[2]"/>      
+                  <mux name="out2out3" input="mult_8x8.Y[3] ff_Y[3].Q" output="mult_8x8_slice.OUT_cfg[3]"/>      
+                  <mux name="out2out4" input="mult_8x8.Y[4] ff_Y[4].Q" output="mult_8x8_slice.OUT_cfg[4]"/>      
+                  <mux name="out2out5" input="mult_8x8.Y[5] ff_Y[5].Q" output="mult_8x8_slice.OUT_cfg[5]"/>      
+                  <mux name="out2out6" input="mult_8x8.Y[6] ff_Y[6].Q" output="mult_8x8_slice.OUT_cfg[6]"/>      
+                  <mux name="out2out7" input="mult_8x8.Y[7] ff_Y[7].Q" output="mult_8x8_slice.OUT_cfg[7]"/>      
+                  <mux name="out2out8" input="mult_8x8.Y[8] ff_Y[8].Q" output="mult_8x8_slice.OUT_cfg[8]"/>      
+                  <mux name="out2out9" input="mult_8x8.Y[9] ff_Y[9].Q" output="mult_8x8_slice.OUT_cfg[9]"/>      
+                  <mux name="out2out10" input="mult_8x8.Y[10] ff_Y[10].Q" output="mult_8x8_slice.OUT_cfg[10]"/>      
+                  <mux name="out2out11" input="mult_8x8.Y[11] ff_Y[11].Q" output="mult_8x8_slice.OUT_cfg[11]"/>      
+                  <mux name="out2out12" input="mult_8x8.Y[12] ff_Y[12].Q" output="mult_8x8_slice.OUT_cfg[12]"/>      
+                  <mux name="out2out13" input="mult_8x8.Y[13] ff_Y[13].Q" output="mult_8x8_slice.OUT_cfg[13]"/>      
+                  <mux name="out2out14" input="mult_8x8.Y[14] ff_Y[14].Q" output="mult_8x8_slice.OUT_cfg[14]"/>      
+                  <mux name="out2out15" input="mult_8x8.Y[15] ff_Y[15].Q" output="mult_8x8_slice.OUT_cfg[15]"/>      
+                  <direct name="out2ff" input="mult_8x8.Y[15:0]" output="ff_Y[15:0].D"/>      
+                  <complete name="clk_ff_A" input="mult_8x8_slice.clk" output="ff_A.clk"/>      
+                  <complete name="clk_ff_B" input="mult_8x8_slice.clk" output="ff_B.clk"/>      
+                  <complete name="clk_ff_Y" input="mult_8x8_slice.clk" output="ff_Y.clk"/>      
+               </interconnect>     
+               <power method="pin-toggle">      
+                  <port name="A_cfg" energy_per_toggle="2.13e-12"/>      
+                  <port name="B_cfg" energy_per_toggle="2.13e-12"/>      
+                  <static_power power_per_instance="0.0"/>      
+               </power>     
+            </pb_type>    
+            <interconnect>     
+               <!-- Stratix IV input delay of 207ps is conservative for this architecture because this architecture does not have an input crossbar in the multiplier. 
 		   Subtract 72.5 ps delay, which is already in the connection block input mux, leading
 		   to a 134 ps delay.
 				The interconnect difference for DSP blocks is 0.5523, which leads to a minimum delay of 74 ps
              -->     
-          <direct name="a2a" input="mult_8.a" output="mult_8x8_slice.A_cfg">
-            <delay_constant max="134e-12" min="74e-12" in_port="mult_8.a" out_port="mult_8x8_slice.A_cfg"/>
-          </direct>
-          <direct name="b2b" input="mult_8.b" output="mult_8x8_slice.B_cfg">
-            <delay_constant max="134e-12" min="74e-12" in_port="mult_8.b" out_port="mult_8x8_slice.B_cfg"/>
-          </direct>
-          <direct name="out2out" input="mult_8x8_slice.OUT_cfg" output="mult_8.out">
-            <delay_constant max="1.93e-9" min="74e-12" in_port="mult_8x8_slice.OUT_cfg" out_port="mult_8.out"/>
-          </direct>
-          <complete name="clk" input="mult_8.clk" output="mult_8x8_slice.clk"/>
-        </interconnect>
-      </mode>
-      <!-- Place this multiplier block every 8 columns from (and including) the sixth column -->
-      <power method="sum-of-children"/>
-    </pb_type>
-    <!-- Define fracturable multiplier end -->
+               <direct name="a2a" input="mult_8.a" output="mult_8x8_slice.A_cfg">      
+                  <delay_constant max="134e-12" min="74e-12" in_port="mult_8.a" out_port="mult_8x8_slice.A_cfg"/>      
+               </direct>     
+               <direct name="b2b" input="mult_8.b" output="mult_8x8_slice.B_cfg">      
+                  <delay_constant max="134e-12" min="74e-12" in_port="mult_8.b" out_port="mult_8x8_slice.B_cfg"/>      
+               </direct>     
+               <direct name="out2out" input="mult_8x8_slice.OUT_cfg" output="mult_8.out">      
+                  <delay_constant max="1.93e-9" min="74e-12" in_port="mult_8x8_slice.OUT_cfg" out_port="mult_8.out"/>      
+               </direct>     
+               <complete name="clk" input="mult_8.clk" output="mult_8x8_slice.clk"/>     
+            </interconnect>    
+         </mode>   
+         <!-- Place this multiplier block every 8 columns from (and including) the sixth column -->   
+         <power method="sum-of-children"/>   
+      </pb_type>  
+      <!-- Define fracturable multiplier end -->  

-  </complexblocklist>
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N4_tileable_full_output_crossbar_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileable_full_output_crossbar_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -13,9 +13,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -24,63 +23,63 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="full"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="10" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="full"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -95,20 +94,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -121,80 +120,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -203,68 +202,68 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="10" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="full"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="10" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="full"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -273,24 +272,24 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <complete name="output_crossbar" input="fle[3:0].out" output="clb.O">
-          <delay_constant max="45e-12" in_port="fle[3:0].out" out_port="clb.O"/>
-        </complete>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <complete name="output_crossbar" input="fle[3:0].out" output="clb.O">     
+               <delay_constant max="45e-12" in_port="fle[3:0].out" out_port="clb.O"/>     
+            </complete>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N4_tileable_no_local_routing_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N4_tileable_no_local_routing_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -13,9 +13,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -24,66 +23,66 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I0" num_pins="4" equivalent="full"/>
-      <input name="I1" num_pins="4" equivalent="full"/>
-      <input name="I2" num_pins="4" equivalent="full"/>
-      <input name="I3" num_pins="4" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I0" num_pins="4" equivalent="full"/>    
+          <input name="I1" num_pins="4" equivalent="full"/>    
+          <input name="I2" num_pins="4" equivalent="full"/>    
+          <input name="I3" num_pins="4" equivalent="full"/>    
+          <output name="O" num_pins="4" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -98,20 +97,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -124,80 +123,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -206,90 +205,90 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <!-- FIXME These inputs should be logic equivalent
+      <pb_type name="clb">   
+         <!-- FIXME These inputs should be logic equivalent
           However, current annotation engine does not support this
           The feature should be enabled after patching
        -->   
-      <input name="I0" num_pins="4" equivalent="full"/>
-      <input name="I1" num_pins="4" equivalent="full"/>
-      <input name="I2" num_pins="4" equivalent="full"/>
-      <input name="I3" num_pins="4" equivalent="full"/>
-      <output name="O" num_pins="4" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+         <input name="I0" num_pins="4" equivalent="full"/>   
+         <input name="I1" num_pins="4" equivalent="full"/>   
+         <input name="I2" num_pins="4" equivalent="full"/>   
+         <input name="I3" num_pins="4" equivalent="full"/>   
+         <output name="O" num_pins="4" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <direct name="crossbar0" input="clb.I0" output="fle[0:0].in"/>
-        <direct name="crossbar1" input="clb.I1" output="fle[1:1].in"/>
-        <direct name="crossbar2" input="clb.I2" output="fle[2:2].in"/>
-        <direct name="crossbar3" input="clb.I3" output="fle[3:3].in"/>
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <direct name="crossbar0" input="clb.I0" output="fle[0:0].in"/>    
+            <direct name="crossbar1" input="clb.I1" output="fle[1:1].in"/>    
+            <direct name="crossbar2" input="clb.I2" output="fle[2:2].in"/>    
+            <direct name="crossbar3" input="clb.I3" output="fle[3:3].in"/>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="output_crossbar" input="fle[3:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="output_crossbar" input="fle[3:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_N5_tileable_pattern_local_routing_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_N5_tileable_pattern_local_routing_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -12,9 +12,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,63 +22,63 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="12" equivalent="full"/>
-      <output name="O" num_pins="5" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="12" equivalent="full"/>    
+          <output name="O" num_pins="5" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -94,20 +93,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -120,80 +119,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -202,72 +201,72 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="12" equivalent="full"/>
-      <output name="O" num_pins="5" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="12" equivalent="full"/>   
+         <output name="O" num_pins="5" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 4-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="5">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="1"/>
-        <output name="lut_out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
+         <pb_type name="fle" num_pb="5">    
            <input name="in" num_pins="4"/>    
-            <output name="lut_out" num_pins="1"/>
            <output name="out" num_pins="1"/>    
+            <output name="lut_out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="lut_out" num_pins="1"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="lut4.out" output="ble4.lut_out"/>
-              <direct name="direct4" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="ble4.lut_out" output="fle.lut_out[0:0]"/>
-            <direct name="direct4" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="lut4.out" output="ble4.lut_out"/>       
+                     <direct name="direct4" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="ble4.lut_out" output="fle.lut_out[0:0]"/>      
+                  <direct name="direct4" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -276,80 +275,80 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <!-- Local routing for FLE[0] inputs -->
-        <complete name="crossbar_fle0" input="clb.I fle[4:0].out" output="fle[0:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[0:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out" out_port="fle[0:0].in"/>
+            <!-- Local routing for FLE[0] inputs -->    
+            <complete name="crossbar_fle0" input="clb.I fle[4:0].out" output="fle[0:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[0:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out" out_port="fle[0:0].in"/>     
+            </complete>    
+            <!-- Local routing for FLE[1] inputs -->    
+            <complete name="crossbar_fle1_in0" input="clb.I fle[4:0].out fle[0:0].lut_out" output="fle[1:1].in[0]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[1:1].in[0]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out fle[0:0].lut_out" out_port="fle[1:1].in[0]"/>     
+            </complete>    
+            <complete name="crossbar_fle1" input="clb.I fle[4:0].out" output="fle[1:1].in[1:3]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[1:1].in[1:3]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out" out_port="fle[1:1].in[1:3]"/>     
+            </complete>    
+            <!-- Local routing for FLE[2] inputs -->    
+            <complete name="crossbar_fle2_in0" input="clb.I fle[4:0].out fle[0:0].lut_out" output="fle[2:2].in[0]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[2:2].in[0]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out fle[0:0].lut_out" out_port="fle[2:2].in[0]"/>     
+            </complete>    
+            <complete name="crossbar_fle2_in1" input="clb.I fle[4:0].out fle[1:1].lut_out" output="fle[2:2].in[1]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[2:2].in[1]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out fle[1:1].lut_out" out_port="fle[2:2].in[1]"/>     
+            </complete>    
+            <complete name="crossbar_fle2" input="clb.I fle[4:0].out" output="fle[2:2].in[2:3]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[2:2].in[2:3]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out" out_port="fle[2:2].in[2:3]"/>     
+            </complete>    
+            <!-- Local routing for FLE[3] inputs -->    
+            <complete name="crossbar_fle3_in0" input="clb.I fle[4:0].out fle[0:0].lut_out" output="fle[3:3].in[0]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:3].in[0]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out fle[0:0].lut_out" out_port="fle[3:3].in[0]"/>     
+            </complete>    
+            <complete name="crossbar_fle3_in1" input="clb.I fle[4:0].out fle[1:1].lut_out" output="fle[3:3].in[1]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:3].in[1]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out fle[1:1].lut_out" out_port="fle[3:3].in[1]"/>     
+            </complete>    
+            <complete name="crossbar_fle3_in2" input="clb.I fle[4:0].out fle[2:2].lut_out" output="fle[3:3].in[2]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:3].in[2]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out fle[2:2].lut_out" out_port="fle[3:3].in[2]"/>     
+            </complete>    
+            <complete name="crossbar_fle3" input="clb.I fle[4:0].out" output="fle[3:3].in[3:3]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:3].in[3:3]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out" out_port="fle[3:3].in[3:3]"/>     
+            </complete>    
+            <!-- Local routing for FLE[4] inputs -->    
+            <complete name="crossbar_fle4_in0" input="clb.I fle[4:0].out fle[0:0].lut_out" output="fle[4:4].in[0]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[4:4].in[0]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out fle[0:0].lut_out" out_port="fle[4:4].in[0]"/>     
+            </complete>    
+            <complete name="crossbar_fle4_in1" input="clb.I fle[4:0].out fle[1:1].lut_out" output="fle[4:4].in[1]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[4:4].in[1]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out fle[1:1].lut_out" out_port="fle[4:4].in[1]"/>     
+            </complete>    
+            <complete name="crossbar_fle4_in2" input="clb.I fle[4:0].out fle[2:2].lut_out" output="fle[4:4].in[2]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[4:4].in[2]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out fle[2:2].lut_out" out_port="fle[4:4].in[2]"/>     
+            </complete>    
+            <complete name="crossbar_fle4_in3" input="clb.I fle[4:0].out fle[3:3].lut_out" output="fle[4:4].in[3]">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[4:4].in[3]"/>     
+               <delay_constant max="75e-12" in_port="fle[4:0].out fle[3:3].lut_out" out_port="fle[4:4].in[3]"/>     
+            </complete>    
+            <!-- Local clock connections -->    
+            <complete name="clks" input="clb.clk" output="fle[4:0].clk">
        </complete>    
-        <!-- Local routing for FLE[1] inputs -->
-        <complete name="crossbar_fle1_in0" input="clb.I fle[4:0].out fle[0:0].lut_out" output="fle[1:1].in[0]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[1:1].in[0]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out fle[0:0].lut_out" out_port="fle[1:1].in[0]"/>
-        </complete>
-        <complete name="crossbar_fle1" input="clb.I fle[4:0].out" output="fle[1:1].in[1:3]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[1:1].in[1:3]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out" out_port="fle[1:1].in[1:3]"/>
-        </complete>
-        <!-- Local routing for FLE[2] inputs -->
-        <complete name="crossbar_fle2_in0" input="clb.I fle[4:0].out fle[0:0].lut_out" output="fle[2:2].in[0]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[2:2].in[0]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out fle[0:0].lut_out" out_port="fle[2:2].in[0]"/>
-        </complete>
-        <complete name="crossbar_fle2_in1" input="clb.I fle[4:0].out fle[1:1].lut_out" output="fle[2:2].in[1]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[2:2].in[1]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out fle[1:1].lut_out" out_port="fle[2:2].in[1]"/>
-        </complete>
-        <complete name="crossbar_fle2" input="clb.I fle[4:0].out" output="fle[2:2].in[2:3]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[2:2].in[2:3]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out" out_port="fle[2:2].in[2:3]"/>
-        </complete>
-        <!-- Local routing for FLE[3] inputs -->
-        <complete name="crossbar_fle3_in0" input="clb.I fle[4:0].out fle[0:0].lut_out" output="fle[3:3].in[0]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:3].in[0]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out fle[0:0].lut_out" out_port="fle[3:3].in[0]"/>
-        </complete>
-        <complete name="crossbar_fle3_in1" input="clb.I fle[4:0].out fle[1:1].lut_out" output="fle[3:3].in[1]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:3].in[1]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out fle[1:1].lut_out" out_port="fle[3:3].in[1]"/>
-        </complete>
-        <complete name="crossbar_fle3_in2" input="clb.I fle[4:0].out fle[2:2].lut_out" output="fle[3:3].in[2]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:3].in[2]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out fle[2:2].lut_out" out_port="fle[3:3].in[2]"/>
-        </complete>
-        <complete name="crossbar_fle3" input="clb.I fle[4:0].out" output="fle[3:3].in[3:3]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:3].in[3:3]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out" out_port="fle[3:3].in[3:3]"/>
-        </complete>
-        <!-- Local routing for FLE[4] inputs -->
-        <complete name="crossbar_fle4_in0" input="clb.I fle[4:0].out fle[0:0].lut_out" output="fle[4:4].in[0]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[4:4].in[0]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out fle[0:0].lut_out" out_port="fle[4:4].in[0]"/>
-        </complete>
-        <complete name="crossbar_fle4_in1" input="clb.I fle[4:0].out fle[1:1].lut_out" output="fle[4:4].in[1]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[4:4].in[1]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out fle[1:1].lut_out" out_port="fle[4:4].in[1]"/>
-        </complete>
-        <complete name="crossbar_fle4_in2" input="clb.I fle[4:0].out fle[2:2].lut_out" output="fle[4:4].in[2]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[4:4].in[2]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out fle[2:2].lut_out" out_port="fle[4:4].in[2]"/>
-        </complete>
-        <complete name="crossbar_fle4_in3" input="clb.I fle[4:0].out fle[3:3].lut_out" output="fle[4:4].in[3]">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[4:4].in[3]"/>
-          <delay_constant max="75e-12" in_port="fle[4:0].out fle[3:3].lut_out" out_port="fle[4:4].in[3]"/>
-        </complete>
-        <!-- Local clock connections -->
-        <complete name="clks" input="clb.clk" output="fle[4:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[4:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[4:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_fracNative_N4_tileable_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_fracNative_N4_tileable_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Flagship Heterogeneous Architecture (No Carry Chains) for VTR 7.0.

  - 40 nm technology
@ -12,9 +12,8 @@
  Based on flagship k4_frac_N4_mem32K_40nm.xml architecture.

  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,84 +22,84 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="frac_lut4">
-      <input_ports>
-        <port name="in"/>
-      </input_ports>
-      <output_ports>
-        <port name="lut3_out"/>
-        <port name="lut4_out"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="frac_lut4">   
+         <input_ports>    
+            <port name="in"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="lut3_out"/>    
+            <port name="lut4_out"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
         If you need to register the I/O, define clocks in the circuit models
         These clocks can be handled in back-end
     -->  
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="12" equivalent="full"/>
-      <output name="O" num_pins="8" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-    <fixed_layout name="4x4" width="6" height="6">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="12" equivalent="full"/>    
+          <output name="O" num_pins="8" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+      <fixed_layout name="4x4" width="6" height="6">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -115,20 +114,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -141,86 +140,86 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+      <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->  
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
           If you need to register the I/O, define clocks in the circuit models
           These clocks can be handled in back-end
       -->   
-      <!-- A mode denotes the physical implementation of an I/O 
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   

-      <!-- IOs can operate as either inputs or outputs.
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -229,28 +228,28 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="12" equivalent="full"/>
-      <output name="O" num_pins="8" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe fracturable logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="12" equivalent="full"/>   
+         <output name="O" num_pins="8" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe fracturable logic element.  
             Each fracturable logic element has a 6-LUT that can alternatively operate as two 5-LUTs with shared inputs. 
             The outputs of the fracturable logic element can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="2"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- Physical mode definition begin (physical implementation of the fle) -->
-        <mode name="physical" disable_packing="true">
-          <pb_type name="fabric" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="2"/>    
            <clock name="clk" num_pins="1"/>    
-            <pb_type name="frac_logic" num_pb="1">
-              <input name="in" num_pins="4"/>
-              <output name="out" num_pins="2"/>
-              <!-- Define a native (no mode switch) 4-input fracturable LUT 
+            <!-- Physical mode definition begin (physical implementation of the fle) -->    
+            <mode name="physical" disable_packing="true">     
+               <pb_type name="fabric" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="frac_logic" num_pb="1">       
+                     <input name="in" num_pins="4"/>       
+                     <output name="out" num_pins="2"/>       
+                     <!-- Define a native (no mode switch) 4-input fracturable LUT 
                   Different from standard fracturable LUT4 whose input can be tri-stated
                   to enable fracturable output,
                   The native fracturable LUT4 has a LUT3 output without any switches
@ -258,138 +257,138 @@
                   The LUT3 output is directly wired to an internal point of LUT4
                   which can be considered a spy output
                -->       
-              <pb_type name="frac_lut4" blif_model=".subckt frac_lut4" num_pb="1">
-                <input name="in" num_pins="4"/>
-                <output name="lut3_out" num_pins="1"/>
-                <output name="lut4_out" num_pins="1"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="frac_logic.in" output="frac_lut4.in"/>
-                <direct name="direct2" input="frac_lut4.lut3_out[0]" output="frac_logic.out[0]"/>
-                <direct name="direct3" input="frac_lut4.lut4_out[0]" output="frac_logic.out[1]"/>
-              </interconnect>
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="fabric.in" output="frac_logic.in"/>
-              <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>
-              <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>
-              <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>
-                <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>
-              </mux>
-              <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>
-                <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="fabric.in"/>
-            <direct name="direct2" input="fabric.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="fabric.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Physical mode definition end (physical implementation of the fle) -->
-        <!-- 3-LUT + LUT4 with shared inputs mode definition begin -->
-        <mode name="shared_lut3_lut4">
-          <pb_type name="ble3" num_pb="1">
-            <input name="in" num_pins="3"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Define LUT3 -->
-            <pb_type name="lut3" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="3" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut3.in" out_port="lut3.out">
+                     <pb_type name="frac_lut4" blif_model=".subckt frac_lut4" num_pb="1">        
+                        <input name="in" num_pins="4"/>        
+                        <output name="lut3_out" num_pins="1"/>        
+                        <output name="lut4_out" num_pins="1"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="frac_logic.in" output="frac_lut4.in"/>        
+                        <direct name="direct2" input="frac_lut4.lut3_out[0]" output="frac_logic.out[0]"/>        
+                        <direct name="direct3" input="frac_lut4.lut4_out[0]" output="frac_logic.out[1]"/>        
+                     </interconnect>       
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="fabric.in" output="frac_logic.in"/>       
+                     <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>       
+                     <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>       
+                     <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>        
+                        <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>        
+                     </mux>       
+                     <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>        
+                        <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="fabric.in"/>      
+                  <direct name="direct2" input="fabric.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="fabric.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Physical mode definition end (physical implementation of the fle) -->    
+            <!-- 3-LUT + LUT4 with shared inputs mode definition begin -->    
+            <mode name="shared_lut3_lut4">     
+               <pb_type name="ble3" num_pb="1">      
+                  <input name="in" num_pins="3"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT3 -->      
+                  <pb_type name="lut3" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="3" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut3.in" out_port="lut3.out">
                235e-12
                235e-12
                235e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define the flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-           </pb_type>
-           <interconnect>
-             <direct name="direct1" input="ble3.in[2:0]" output="lut3[0:0].in[2:0]"/>
-             <direct name="direct2" input="lut3[0:0].out" output="ff[0:0].D">
-               <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-               <pack_pattern name="ble3" in_port="lut3[0:0].out" out_port="ff[0:0].D"/>
-             </direct>
-             <direct name="direct3" input="ble3.clk" output="ff[0:0].clk"/>
-             <mux name="mux1" input="ff[0:0].Q lut3.out[0:0]" output="ble3.out[0:0]">
-               <!-- LUT to output is faster than FF to output on a Stratix IV -->
-               <delay_constant max="25e-12" in_port="lut3.out[0:0]" out_port="ble3.out[0:0]"/>
-               <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble3.out[0:0]"/>
-             </mux>
-           </interconnect>
-          </pb_type>
-          <pb_type name="ble4" num_pb="1">
-            <input name="in" num_pins="4"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Define LUT4 -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+                  </pb_type>      
+                  <!-- Define the flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                 </pb_type>      
+                 <interconnect>       
+                    <direct name="direct1" input="ble3.in[2:0]" output="lut3[0:0].in[2:0]"/>       
+                    <direct name="direct2" input="lut3[0:0].out" output="ff[0:0].D">        
+                       <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                       <pack_pattern name="ble3" in_port="lut3[0:0].out" out_port="ff[0:0].D"/>        
+                    </direct>       
+                    <direct name="direct3" input="ble3.clk" output="ff[0:0].clk"/>       
+                    <mux name="mux1" input="ff[0:0].Q lut3.out[0:0]" output="ble3.out[0:0]">        
+                       <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                       <delay_constant max="25e-12" in_port="lut3.out[0:0]" out_port="ble3.out[0:0]"/>        
+                       <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble3.out[0:0]"/>        
+                    </mux>       
+                 </interconnect>      
+               </pb_type>     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT4 -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define the flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-           </pb_type>
-           <interconnect>
-             <direct name="direct1" input="ble4.in[3:0]" output="lut4[0:0].in[3:0]"/>
-             <direct name="direct2" input="lut4[0:0].out" output="ff[0:0].D">
-               <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-               <pack_pattern name="ble4" in_port="lut4[0:0].out" out_port="ff[0:0].D"/>
-             </direct>
-             <direct name="direct3" input="ble4.clk" output="ff[0:0].clk"/>
-             <mux name="mux1" input="ff[0:0].Q lut4.out[0:0]" output="ble4.out[0:0]">
-               <!-- LUT to output is faster than FF to output on a Stratix IV -->
-               <delay_constant max="25e-12" in_port="lut4.out[0:0]" out_port="ble4.out[0:0]"/>
-               <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble4.out[0:0]"/>
-             </mux>
-           </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[2:0]" output="ble3.in"/>
-            <direct name="direct2" input="fle.in[3:0]" output="ble4.in"/>
-            <direct name="direct3" input="ble3.out" output="fle.out[0:0]"/>
-            <direct name="direct4" input="ble4.out" output="fle.out[1:1]"/>
-            <direct name="direct5" input="fle.clk" output="ble3.clk"/>
-            <direct name="direct6" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Dual 3-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define the flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                 </pb_type>      
+                 <interconnect>       
+                    <direct name="direct1" input="ble4.in[3:0]" output="lut4[0:0].in[3:0]"/>       
+                    <direct name="direct2" input="lut4[0:0].out" output="ff[0:0].D">        
+                       <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                       <pack_pattern name="ble4" in_port="lut4[0:0].out" out_port="ff[0:0].D"/>        
+                    </direct>       
+                    <direct name="direct3" input="ble4.clk" output="ff[0:0].clk"/>       
+                    <mux name="mux1" input="ff[0:0].Q lut4.out[0:0]" output="ble4.out[0:0]">        
+                       <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                       <delay_constant max="25e-12" in_port="lut4.out[0:0]" out_port="ble4.out[0:0]"/>        
+                       <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble4.out[0:0]"/>        
+                    </mux>       
+                 </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[2:0]" output="ble3.in"/>      
+                  <direct name="direct2" input="fle.in[3:0]" output="ble4.in"/>      
+                  <direct name="direct3" input="ble3.out" output="fle.out[0:0]"/>      
+                  <direct name="direct4" input="ble4.out" output="fle.out[1:1]"/>      
+                  <direct name="direct5" input="fle.clk" output="ble3.clk"/>      
+                  <direct name="direct6" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Dual 3-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -398,23 +397,23 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out[0:0]" output="clb.O[3:0]"/>
-        <direct name="clbouts2" input="fle[3:0].out[1:1]" output="clb.O[7:4]"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out[0:0]" output="clb.O[3:0]"/>    
+            <direct name="clbouts2" input="fle[3:0].out[1:1]" output="clb.O[7:4]"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_frac_N4_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N4_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Flagship Heterogeneous Architecture (No Carry Chains) for VTR 7.0.

  - 40 nm technology
@ -12,9 +12,8 @@
  Based on flagship k4_frac_N4_mem32K_40nm.xml architecture.

  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,84 +22,84 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="frac_lut4">
-      <input_ports>
-        <port name="in"/>
-      </input_ports>
-      <output_ports>
-        <port name="lut3_out"/>
-        <port name="lut4_out"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="frac_lut4">   
+         <input_ports>    
+            <port name="in"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="lut3_out"/>    
+            <port name="lut4_out"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
         If you need to register the I/O, define clocks in the circuit models
         These clocks can be handled in back-end
     -->  
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="12" equivalent="full"/>
-      <output name="O" num_pins="8" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="false">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-    <fixed_layout name="4x4" width="6" height="6">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="12" equivalent="full"/>    
+          <output name="O" num_pins="8" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="false">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+      <fixed_layout name="4x4" width="6" height="6">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -115,20 +114,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -141,86 +140,86 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+      <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->  
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
           If you need to register the I/O, define clocks in the circuit models
           These clocks can be handled in back-end
       -->   
-      <!-- A mode denotes the physical implementation of an I/O 
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   

-      <!-- IOs can operate as either inputs or outputs.
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -229,87 +228,87 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="12" equivalent="full"/>
-      <output name="O" num_pins="8" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe fracturable logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="12" equivalent="full"/>   
+         <output name="O" num_pins="8" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe fracturable logic element.  
             Each fracturable logic element has a 6-LUT that can alternatively operate as two 5-LUTs with shared inputs. 
             The outputs of the fracturable logic element can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="2"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- Physical mode definition begin (physical implementation of the fle) -->
-        <mode name="physical" disable_packing="true">
-          <pb_type name="fabric" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="2"/>    
            <clock name="clk" num_pins="1"/>    
-            <pb_type name="frac_logic" num_pb="1">
-              <input name="in" num_pins="4"/>
-              <output name="out" num_pins="2"/>
-              <!-- Define LUT -->
-              <pb_type name="frac_lut4" blif_model=".subckt frac_lut4" num_pb="1">
-                <input name="in" num_pins="4"/>
-                <output name="lut3_out" num_pins="2"/>
-                <output name="lut4_out" num_pins="1"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="frac_logic.in" output="frac_lut4.in"/>
-                <direct name="direct2" input="frac_lut4.lut3_out[1]" output="frac_logic.out[1]"/>
-                <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->
-                <mux name="mux1" input="frac_lut4.lut4_out frac_lut4.lut3_out[0]" output="frac_logic.out[0]"/>
-              </interconnect>
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="fabric.in" output="frac_logic.in"/>
-              <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>
-              <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>
-              <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>
-                <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>
-              </mux>
-              <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>
-                <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="fabric.in"/>
-            <direct name="direct2" input="fabric.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="fabric.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Physical mode definition end (physical implementation of the fle) -->
-        <!-- Dual 3-LUT mode definition begin -->
-        <mode name="n2_lut3">
-          <pb_type name="lut3inter" num_pb="1">
-            <input name="in" num_pins="3"/>
-            <output name="out" num_pins="2"/>
-            <clock name="clk" num_pins="1"/>
-            <pb_type name="ble3" num_pb="2">
-              <input name="in" num_pins="3"/>
-              <output name="out" num_pins="1"/>
-              <clock name="clk" num_pins="1"/>
-              <!-- Define the LUT -->
-              <pb_type name="lut3" blif_model=".names" num_pb="1" class="lut">
-                <input name="in" num_pins="3" port_class="lut_in"/>
-                <output name="out" num_pins="1" port_class="lut_out"/>
-                <!-- LUT timing using delay matrix -->
-                <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+            <!-- Physical mode definition begin (physical implementation of the fle) -->    
+            <mode name="physical" disable_packing="true">     
+               <pb_type name="fabric" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="frac_logic" num_pb="1">       
+                     <input name="in" num_pins="4"/>       
+                     <output name="out" num_pins="2"/>       
+                     <!-- Define LUT -->       
+                     <pb_type name="frac_lut4" blif_model=".subckt frac_lut4" num_pb="1">        
+                        <input name="in" num_pins="4"/>        
+                        <output name="lut3_out" num_pins="2"/>        
+                        <output name="lut4_out" num_pins="1"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="frac_logic.in" output="frac_lut4.in"/>        
+                        <direct name="direct2" input="frac_lut4.lut3_out[1]" output="frac_logic.out[1]"/>        
+                        <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->        
+                        <mux name="mux1" input="frac_lut4.lut4_out frac_lut4.lut3_out[0]" output="frac_logic.out[0]"/>        
+                     </interconnect>       
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="fabric.in" output="frac_logic.in"/>       
+                     <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>       
+                     <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>       
+                     <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>        
+                        <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>        
+                     </mux>       
+                     <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>        
+                        <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="fabric.in"/>      
+                  <direct name="direct2" input="fabric.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="fabric.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Physical mode definition end (physical implementation of the fle) -->    
+            <!-- Dual 3-LUT mode definition begin -->    
+            <mode name="n2_lut3">     
+               <pb_type name="lut3inter" num_pb="1">      
+                  <input name="in" num_pins="3"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="ble3" num_pb="2">       
+                     <input name="in" num_pins="3"/>       
+                     <output name="out" num_pins="1"/>       
+                     <clock name="clk" num_pins="1"/>       
+                     <!-- Define the LUT -->       
+                     <pb_type name="lut3" blif_model=".names" num_pb="1" class="lut">        
+                        <input name="in" num_pins="3" port_class="lut_in"/>        
+                        <output name="out" num_pins="1" port_class="lut_out"/>        
+                        <!-- LUT timing using delay matrix -->        
+                        <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                           we instead take the average of these numbers to get more stable results
                      82e-12
                      173e-12
@ -317,61 +316,61 @@
                      263e-12
                      398e-12
                      -->        
-                <delay_matrix type="max" in_port="lut3.in" out_port="lut3.out">
+                        <delay_matrix type="max" in_port="lut3.in" out_port="lut3.out">
                  235e-12
                  235e-12
                  235e-12
                </delay_matrix>        
-              </pb_type>
-              <!-- Define the flip-flop -->
-              <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-                <input name="D" num_pins="1" port_class="D"/>
-                <output name="Q" num_pins="1" port_class="Q"/>
-                <clock name="clk" num_pins="1" port_class="clock"/>
-                <T_setup value="66e-12" port="ff.D" clock="clk"/>
-                <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="ble3.in[2:0]" output="lut3[0:0].in[2:0]"/>
-                <direct name="direct2" input="lut3[0:0].out" output="ff[0:0].D">
-                  <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                  <pack_pattern name="ble3" in_port="lut3[0:0].out" out_port="ff[0:0].D"/>
-                </direct>
-                <direct name="direct3" input="ble3.clk" output="ff[0:0].clk"/>
-                <mux name="mux1" input="ff[0:0].Q lut3.out[0:0]" output="ble3.out[0:0]">
-                  <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                  <delay_constant max="25e-12" in_port="lut3.out[0:0]" out_port="ble3.out[0:0]"/>
-                  <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble3.out[0:0]"/>
-                </mux>
-              </interconnect>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="lut3inter.in" output="ble3[0:0].in"/>
-              <direct name="direct2" input="lut3inter.in" output="ble3[1:1].in"/>
-              <direct name="direct3" input="ble3[1:0].out" output="lut3inter.out"/>
-              <complete name="complete1" input="lut3inter.clk" output="ble3[1:0].clk"/>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[2:0]" output="lut3inter.in"/>
-            <direct name="direct2" input="lut3inter.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="lut3inter.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Dual 3-LUT mode definition end -->
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
-            <input name="in" num_pins="4"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+                     </pb_type>       
+                     <!-- Define the flip-flop -->       
+                     <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">        
+                        <input name="D" num_pins="1" port_class="D"/>        
+                        <output name="Q" num_pins="1" port_class="Q"/>        
+                        <clock name="clk" num_pins="1" port_class="clock"/>        
+                        <T_setup value="66e-12" port="ff.D" clock="clk"/>        
+                        <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="ble3.in[2:0]" output="lut3[0:0].in[2:0]"/>        
+                        <direct name="direct2" input="lut3[0:0].out" output="ff[0:0].D">         
+                           <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->         
+                           <pack_pattern name="ble3" in_port="lut3[0:0].out" out_port="ff[0:0].D"/>         
+                        </direct>        
+                        <direct name="direct3" input="ble3.clk" output="ff[0:0].clk"/>        
+                        <mux name="mux1" input="ff[0:0].Q lut3.out[0:0]" output="ble3.out[0:0]">         
+                           <!-- LUT to output is faster than FF to output on a Stratix IV -->         
+                           <delay_constant max="25e-12" in_port="lut3.out[0:0]" out_port="ble3.out[0:0]"/>         
+                           <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble3.out[0:0]"/>         
+                        </mux>        
+                     </interconnect>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="lut3inter.in" output="ble3[0:0].in"/>       
+                     <direct name="direct2" input="lut3inter.in" output="ble3[1:1].in"/>       
+                     <direct name="direct3" input="ble3[1:0].out" output="lut3inter.out"/>       
+                     <complete name="complete1" input="lut3inter.clk" output="ble3[1:0].clk"/>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[2:0]" output="lut3inter.in"/>      
+                  <direct name="direct2" input="lut3inter.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="lut3inter.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Dual 3-LUT mode definition end -->    
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
@ -380,45 +379,45 @@
                  398e-12
                  397e-12
                  -->       
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -427,23 +426,23 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out[0:0]" output="clb.O[3:0]"/>
-        <direct name="clbouts2" input="fle[3:0].out[1:1]" output="clb.O[7:4]"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out[0:0]" output="clb.O[3:0]"/>    
+            <direct name="clbouts2" input="fle[3:0].out[1:1]" output="clb.O[7:4]"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_frac_N4_tileable_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N4_tileable_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Flagship Heterogeneous Architecture (No Carry Chains) for VTR 7.0.

  - 40 nm technology
@ -12,9 +12,8 @@
  Based on flagship k4_frac_N4_mem32K_40nm.xml architecture.

  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,84 +22,84 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="frac_lut4">
-      <input_ports>
-        <port name="in"/>
-      </input_ports>
-      <output_ports>
-        <port name="lut3_out"/>
-        <port name="lut4_out"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="frac_lut4">   
+         <input_ports>    
+            <port name="in"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="lut3_out"/>    
+            <port name="lut4_out"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
         If you need to register the I/O, define clocks in the circuit models
         These clocks can be handled in back-end
     -->  
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="12" equivalent="full"/>
-      <output name="O" num_pins="8" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-    <fixed_layout name="4x4" width="6" height="6">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="12" equivalent="full"/>    
+          <output name="O" num_pins="8" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+      <fixed_layout name="4x4" width="6" height="6">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -115,20 +114,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -141,86 +140,86 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+      <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->  
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
           If you need to register the I/O, define clocks in the circuit models
           These clocks can be handled in back-end
       -->   
-      <!-- A mode denotes the physical implementation of an I/O 
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   

-      <!-- IOs can operate as either inputs or outputs.
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -229,87 +228,87 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="12" equivalent="full"/>
-      <output name="O" num_pins="8" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe fracturable logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="12" equivalent="full"/>   
+         <output name="O" num_pins="8" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe fracturable logic element.  
             Each fracturable logic element has a 6-LUT that can alternatively operate as two 5-LUTs with shared inputs. 
             The outputs of the fracturable logic element can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="2"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- Physical mode definition begin (physical implementation of the fle) -->
-        <mode name="physical" disable_packing="true">
-          <pb_type name="fabric" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="2"/>    
            <clock name="clk" num_pins="1"/>    
-            <pb_type name="frac_logic" num_pb="1">
-              <input name="in" num_pins="4"/>
-              <output name="out" num_pins="2"/>
-              <!-- Define LUT -->
-              <pb_type name="frac_lut4" blif_model=".subckt frac_lut4" num_pb="1">
-                <input name="in" num_pins="4"/>
-                <output name="lut3_out" num_pins="2"/>
-                <output name="lut4_out" num_pins="1"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="frac_logic.in" output="frac_lut4.in"/>
-                <direct name="direct2" input="frac_lut4.lut3_out[1]" output="frac_logic.out[1]"/>
-                <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->
-                <mux name="mux1" input="frac_lut4.lut4_out frac_lut4.lut3_out[0]" output="frac_logic.out[0]"/>
-              </interconnect>
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="fabric.in" output="frac_logic.in"/>
-              <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>
-              <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>
-              <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>
-                <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>
-              </mux>
-              <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>
-                <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="fabric.in"/>
-            <direct name="direct2" input="fabric.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="fabric.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Physical mode definition end (physical implementation of the fle) -->
-        <!-- Dual 3-LUT mode definition begin -->
-        <mode name="n2_lut3">
-          <pb_type name="lut3inter" num_pb="1">
-            <input name="in" num_pins="3"/>
-            <output name="out" num_pins="2"/>
-            <clock name="clk" num_pins="1"/>
-            <pb_type name="ble3" num_pb="2">
-              <input name="in" num_pins="3"/>
-              <output name="out" num_pins="1"/>
-              <clock name="clk" num_pins="1"/>
-              <!-- Define the LUT -->
-              <pb_type name="lut3" blif_model=".names" num_pb="1" class="lut">
-                <input name="in" num_pins="3" port_class="lut_in"/>
-                <output name="out" num_pins="1" port_class="lut_out"/>
-                <!-- LUT timing using delay matrix -->
-                <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+            <!-- Physical mode definition begin (physical implementation of the fle) -->    
+            <mode name="physical" disable_packing="true">     
+               <pb_type name="fabric" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="frac_logic" num_pb="1">       
+                     <input name="in" num_pins="4"/>       
+                     <output name="out" num_pins="2"/>       
+                     <!-- Define LUT -->       
+                     <pb_type name="frac_lut4" blif_model=".subckt frac_lut4" num_pb="1">        
+                        <input name="in" num_pins="4"/>        
+                        <output name="lut3_out" num_pins="2"/>        
+                        <output name="lut4_out" num_pins="1"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="frac_logic.in" output="frac_lut4.in"/>        
+                        <direct name="direct2" input="frac_lut4.lut3_out[1]" output="frac_logic.out[1]"/>        
+                        <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->        
+                        <mux name="mux1" input="frac_lut4.lut4_out frac_lut4.lut3_out[0]" output="frac_logic.out[0]"/>        
+                     </interconnect>       
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="fabric.in" output="frac_logic.in"/>       
+                     <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>       
+                     <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>       
+                     <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>        
+                        <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>        
+                     </mux>       
+                     <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>        
+                        <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="fabric.in"/>      
+                  <direct name="direct2" input="fabric.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="fabric.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Physical mode definition end (physical implementation of the fle) -->    
+            <!-- Dual 3-LUT mode definition begin -->    
+            <mode name="n2_lut3">     
+               <pb_type name="lut3inter" num_pb="1">      
+                  <input name="in" num_pins="3"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="ble3" num_pb="2">       
+                     <input name="in" num_pins="3"/>       
+                     <output name="out" num_pins="1"/>       
+                     <clock name="clk" num_pins="1"/>       
+                     <!-- Define the LUT -->       
+                     <pb_type name="lut3" blif_model=".names" num_pb="1" class="lut">        
+                        <input name="in" num_pins="3" port_class="lut_in"/>        
+                        <output name="out" num_pins="1" port_class="lut_out"/>        
+                        <!-- LUT timing using delay matrix -->        
+                        <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                           we instead take the average of these numbers to get more stable results
                      82e-12
                      173e-12
@ -317,61 +316,61 @@
                      263e-12
                      398e-12
                      -->        
-                <delay_matrix type="max" in_port="lut3.in" out_port="lut3.out">
+                        <delay_matrix type="max" in_port="lut3.in" out_port="lut3.out">
                  235e-12
                  235e-12
                  235e-12
                </delay_matrix>        
-              </pb_type>
-              <!-- Define the flip-flop -->
-              <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-                <input name="D" num_pins="1" port_class="D"/>
-                <output name="Q" num_pins="1" port_class="Q"/>
-                <clock name="clk" num_pins="1" port_class="clock"/>
-                <T_setup value="66e-12" port="ff.D" clock="clk"/>
-                <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="ble3.in[2:0]" output="lut3[0:0].in[2:0]"/>
-                <direct name="direct2" input="lut3[0:0].out" output="ff[0:0].D">
-                  <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                  <pack_pattern name="ble3" in_port="lut3[0:0].out" out_port="ff[0:0].D"/>
-                </direct>
-                <direct name="direct3" input="ble3.clk" output="ff[0:0].clk"/>
-                <mux name="mux1" input="ff[0:0].Q lut3.out[0:0]" output="ble3.out[0:0]">
-                  <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                  <delay_constant max="25e-12" in_port="lut3.out[0:0]" out_port="ble3.out[0:0]"/>
-                  <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble3.out[0:0]"/>
-                </mux>
-              </interconnect>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="lut3inter.in" output="ble3[0:0].in"/>
-              <direct name="direct2" input="lut3inter.in" output="ble3[1:1].in"/>
-              <direct name="direct3" input="ble3[1:0].out" output="lut3inter.out"/>
-              <complete name="complete1" input="lut3inter.clk" output="ble3[1:0].clk"/>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[2:0]" output="lut3inter.in"/>
-            <direct name="direct2" input="lut3inter.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="lut3inter.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Dual 3-LUT mode definition end -->
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
-            <input name="in" num_pins="4"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+                     </pb_type>       
+                     <!-- Define the flip-flop -->       
+                     <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">        
+                        <input name="D" num_pins="1" port_class="D"/>        
+                        <output name="Q" num_pins="1" port_class="Q"/>        
+                        <clock name="clk" num_pins="1" port_class="clock"/>        
+                        <T_setup value="66e-12" port="ff.D" clock="clk"/>        
+                        <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="ble3.in[2:0]" output="lut3[0:0].in[2:0]"/>        
+                        <direct name="direct2" input="lut3[0:0].out" output="ff[0:0].D">         
+                           <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->         
+                           <pack_pattern name="ble3" in_port="lut3[0:0].out" out_port="ff[0:0].D"/>         
+                        </direct>        
+                        <direct name="direct3" input="ble3.clk" output="ff[0:0].clk"/>        
+                        <mux name="mux1" input="ff[0:0].Q lut3.out[0:0]" output="ble3.out[0:0]">         
+                           <!-- LUT to output is faster than FF to output on a Stratix IV -->         
+                           <delay_constant max="25e-12" in_port="lut3.out[0:0]" out_port="ble3.out[0:0]"/>         
+                           <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble3.out[0:0]"/>         
+                        </mux>        
+                     </interconnect>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="lut3inter.in" output="ble3[0:0].in"/>       
+                     <direct name="direct2" input="lut3inter.in" output="ble3[1:1].in"/>       
+                     <direct name="direct3" input="ble3[1:0].out" output="lut3inter.out"/>       
+                     <complete name="complete1" input="lut3inter.clk" output="ble3[1:0].clk"/>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[2:0]" output="lut3inter.in"/>      
+                  <direct name="direct2" input="lut3inter.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="lut3inter.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Dual 3-LUT mode definition end -->    
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
@ -380,45 +379,45 @@
                  398e-12
                  397e-12
                  -->       
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -427,23 +426,23 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out[0:0]" output="clb.O[3:0]"/>
-        <direct name="clbouts2" input="fle[3:0].out[1:1]" output="clb.O[7:4]"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out[0:0]" output="clb.O[3:0]"/>    
+            <direct name="clbouts2" input="fle[3:0].out[1:1]" output="clb.O[7:4]"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_frac_N4_tileable_adder_chain_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N4_tileable_adder_chain_40nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N4_tileable_adder_chain_mem1K_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N4_tileable_adder_chain_mem1K_40nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N4_tileable_adder_chain_mem1K_L124_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N4_tileable_adder_chain_mem1K_L124_40nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N4_tileable_adder_chain_mem1K_frac_dsp32_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N4_tileable_adder_chain_mem1K_frac_dsp32_40nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N4_tileable_fracff2edge_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N4_tileable_fracff2edge_40nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N4_tileable_fracff_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N4_tileable_fracff_40nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N4_tileable_lutram_40nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N4_tileable_lutram_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Flagship Heterogeneous Architecture (No Carry Chains) for VTR 7.0.

  - 40 nm technology
@ -13,9 +13,8 @@
  Based on flagship k4_frac_N4_mem32K_40nm.xml architecture.

  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -24,101 +23,101 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="frac_lut4">
-      <input_ports>
-        <port name="in"/>
-      </input_ports>
-      <output_ports>
-        <port name="lut3_out"/>
-        <port name="lut4_out"/>
-      </output_ports>
-    </model>
-    <model name="spram_4x1">
-      <input_ports>
-        <port name="wr_en" clock="clk"/>
-        <port name="addr" clock="clk"/>
-        <port name="d_in" clock="clk"/>
-        <port name="clk" is_clock="1"/>
-      </input_ports>
-      <output_ports>
-        <port name="d_out" clock="clk"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="frac_lut4">   
+         <input_ports>    
+            <port name="in"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="lut3_out"/>    
+            <port name="lut4_out"/>    
+         </output_ports>   
+      </model>  
+      <model name="spram_4x1">   
+         <input_ports>    
+            <port name="wr_en" clock="clk"/>    
+            <port name="addr" clock="clk"/>    
+            <port name="d_in" clock="clk"/>    
+            <port name="clk" is_clock="1"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="d_out" clock="clk"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
         If you need to register the I/O, define clocks in the circuit models
         These clocks can be handled in back-end
     -->  
-    <tile name="io" capacity="1" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="12" equivalent="full"/>
-      <output name="O" num_pins="8" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <!--pinlocations pattern="spread"/-->
-      <pinlocations pattern="custom">
-        <loc side="left">clb.clk</loc>
-        <loc side="top"></loc>
-        <loc side="right">clb.O[3:0] clb.I[5:0]</loc>
-        <loc side="bottom">clb.O[7:4] clb.I[11:6]</loc>
-      </pinlocations>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-    <fixed_layout name="4x4" width="6" height="6">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+      <tile name="io" area="0">   <sub_tile name="io" capacity="1">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="12" equivalent="full"/>    
+          <output name="O" num_pins="8" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <!--pinlocations pattern="spread"/-->    
+          <pinlocations pattern="custom">     
+             <loc side="left">clb.clk</loc>     
+             <loc side="top"/>     
+             <loc side="right">clb.O[3:0] clb.I[5:0]</loc>     
+             <loc side="bottom">clb.O[7:4] clb.I[11:6]</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+      <fixed_layout name="4x4" width="6" height="6">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -133,20 +132,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -159,86 +158,86 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+      <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->  
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
           If you need to register the I/O, define clocks in the circuit models
           These clocks can be handled in back-end
       -->   
-      <!-- A mode denotes the physical implementation of an I/O 
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   

-      <!-- IOs can operate as either inputs or outputs.
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -247,112 +246,112 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="12" equivalent="full"/>
-      <output name="O" num_pins="8" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe fracturable logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="12" equivalent="full"/>   
+         <output name="O" num_pins="8" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe fracturable logic element.  
             Each fracturable logic element has a 6-LUT that can alternatively operate as two 5-LUTs with shared inputs. 
             The outputs of the fracturable logic element can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="4">
-        <input name="in" num_pins="4"/>
-        <output name="out" num_pins="2"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- Physical mode definition begin (physical implementation of the fle) -->
-        <mode name="physical" disable_packing="true">
-          <pb_type name="fabric" num_pb="1">
+         <pb_type name="fle" num_pb="4">    
            <input name="in" num_pins="4"/>    
            <output name="out" num_pins="2"/>    
            <clock name="clk" num_pins="1"/>    
-            <pb_type name="frac_logic" num_pb="1">
-              <input name="in" num_pins="4"/>
-              <output name="out" num_pins="2"/>
-              <!-- Define LUT -->
-              <pb_type name="frac_lut4" blif_model=".subckt frac_lut4" num_pb="1">
-                <input name="in" num_pins="4"/>
-                <output name="lut3_out" num_pins="2"/>
-                <output name="lut4_out" num_pins="1"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="frac_logic.in" output="frac_lut4.in"/>
-                <direct name="direct2" input="frac_lut4.lut3_out[1]" output="frac_logic.out[1]"/>
-                <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->
-                <mux name="mux1" input="frac_lut4.lut4_out frac_lut4.lut3_out[0]" output="frac_logic.out[0]"/>
-              </interconnect>
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <!-- Define spram_4x1 -->
-            <pb_type name="spram_4x1" blif_model=".subckt spram_4x1" num_pb="1">
-              <input name="wr_en" num_pins="1"/>
-              <input name="addr" num_pins="2"/>
-              <input name="d_in" num_pins="1"/>
-              <output name="d_out" num_pins="1"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-          		<T_setup value="509e-12" port="spram_4x1.wr_en" clock="clk"/>
-          		<T_setup value="509e-12" port="spram_4x1.addr" clock="clk"/>
-         		<T_setup value="509e-12" port="spram_4x1.d_in" clock="clk"/>
-          		<T_clock_to_Q max="1.234e-9" port="spram_4x1.d_out" clock="clk"/>
-          	  <power method="pin-toggle">
-            	<port name="clk" energy_per_toggle="17.9e-12"/>
-            	<static_power power_per_instance="0.0"/>
-          	  </power>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="fabric.in" output="frac_logic.in"/>
-              <direct name="direct_mem_en" input="fabric.in[0]" output="spram_4x1.wr_en"/>
-              <direct name="direct_mem_in" input="fabric.in[3]" output="spram_4x1.d_in"/>
-              <direct name="direct_mem_addr" input="fabric.in[2:1]" output="spram_4x1.addr"/>
-              <complete name="direct6" input="fabric.clk" output="ff[1:0].clk"/>
-              <complete name="direct_mem_clk" input="fabric.clk" output="spram_4x1.clk"/>
-              <mux name="mux1" input="frac_logic.out[0:0] spram_4x1.d_out" output="ff[0:0].D">
-                <delay_constant max="25e-12" in_port="frac_logic.out[0:0]" out_port="ff[0:0].D"/>
-                <delay_constant max="45e-12" in_port="spram_4x1.d_out" out_port="ff[0:0].D"/>
-              </mux>
-              <mux name="mux2" input="frac_logic.out[1:1] spram_4x1.d_out" output="ff[1:1].D">
-                <delay_constant max="25e-12" in_port="frac_logic.out[1:1]" out_port="ff[1:1].D"/>
-                <delay_constant max="45e-12" in_port="spram_4x1.d_out" out_port="ff[1:1].D"/>
-              </mux>
-              <mux name="mux3" input="spram_4x1.d_out ff[0].Q frac_logic.out[0]" output="fabric.out[0]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>
-              </mux>
-              <mux name="mux4" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="fabric.in"/>
-            <direct name="direct3" input="fabric.out" output="fle.out"/>
-            <direct name="direct5" input="fle.clk" output="fabric.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Physical mode definition end (physical implementation of the fle) -->
-        <!-- Dual 3-LUT mode definition begin -->
-        <mode name="n2_lut3">
-          <pb_type name="lut3inter" num_pb="1">
-            <input name="in" num_pins="3"/>
-            <output name="out" num_pins="2"/>
-            <clock name="clk" num_pins="1"/>
-            <pb_type name="ble3" num_pb="2">
-              <input name="in" num_pins="3"/>
-              <output name="out" num_pins="1"/>
-              <clock name="clk" num_pins="1"/>
-              <!-- Define the LUT -->
-              <pb_type name="lut3" blif_model=".names" num_pb="1" class="lut">
-                <input name="in" num_pins="3" port_class="lut_in"/>
-                <output name="out" num_pins="1" port_class="lut_out"/>
-                <!-- LUT timing using delay matrix -->
-                <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+            <!-- Physical mode definition begin (physical implementation of the fle) -->    
+            <mode name="physical" disable_packing="true">     
+               <pb_type name="fabric" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="frac_logic" num_pb="1">       
+                     <input name="in" num_pins="4"/>       
+                     <output name="out" num_pins="2"/>       
+                     <!-- Define LUT -->       
+                     <pb_type name="frac_lut4" blif_model=".subckt frac_lut4" num_pb="1">        
+                        <input name="in" num_pins="4"/>        
+                        <output name="lut3_out" num_pins="2"/>        
+                        <output name="lut4_out" num_pins="1"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="frac_logic.in" output="frac_lut4.in"/>        
+                        <direct name="direct2" input="frac_lut4.lut3_out[1]" output="frac_logic.out[1]"/>        
+                        <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->        
+                        <mux name="mux1" input="frac_lut4.lut4_out frac_lut4.lut3_out[0]" output="frac_logic.out[0]"/>        
+                     </interconnect>       
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <!-- Define spram_4x1 -->      
+                  <pb_type name="spram_4x1" blif_model=".subckt spram_4x1" num_pb="1">       
+                     <input name="wr_en" num_pins="1"/>       
+                     <input name="addr" num_pins="2"/>       
+                     <input name="d_in" num_pins="1"/>       
+                     <output name="d_out" num_pins="1"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+          		       <T_setup value="509e-12" port="spram_4x1.wr_en" clock="clk"/>       
+          		       <T_setup value="509e-12" port="spram_4x1.addr" clock="clk"/>       
+         		       <T_setup value="509e-12" port="spram_4x1.d_in" clock="clk"/>       
+          		       <T_clock_to_Q max="1.234e-9" port="spram_4x1.d_out" clock="clk"/>       
+          	         <power method="pin-toggle">        
+            	        <port name="clk" energy_per_toggle="17.9e-12"/>        
+            	        <static_power power_per_instance="0.0"/>        
+          	         </power>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="fabric.in" output="frac_logic.in"/>       
+                     <direct name="direct_mem_en" input="fabric.in[0]" output="spram_4x1.wr_en"/>       
+                     <direct name="direct_mem_in" input="fabric.in[3]" output="spram_4x1.d_in"/>       
+                     <direct name="direct_mem_addr" input="fabric.in[2:1]" output="spram_4x1.addr"/>       
+                     <complete name="direct6" input="fabric.clk" output="ff[1:0].clk"/>       
+                     <complete name="direct_mem_clk" input="fabric.clk" output="spram_4x1.clk"/>       
+                     <mux name="mux1" input="frac_logic.out[0:0] spram_4x1.d_out" output="ff[0:0].D">        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[0:0]" out_port="ff[0:0].D"/>        
+                        <delay_constant max="45e-12" in_port="spram_4x1.d_out" out_port="ff[0:0].D"/>        
+                     </mux>       
+                     <mux name="mux2" input="frac_logic.out[1:1] spram_4x1.d_out" output="ff[1:1].D">        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[1:1]" out_port="ff[1:1].D"/>        
+                        <delay_constant max="45e-12" in_port="spram_4x1.d_out" out_port="ff[1:1].D"/>        
+                     </mux>       
+                     <mux name="mux3" input="spram_4x1.d_out ff[0].Q frac_logic.out[0]" output="fabric.out[0]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>        
+                     </mux>       
+                     <mux name="mux4" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="fabric.in"/>      
+                  <direct name="direct3" input="fabric.out" output="fle.out"/>      
+                  <direct name="direct5" input="fle.clk" output="fabric.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Physical mode definition end (physical implementation of the fle) -->    
+            <!-- Dual 3-LUT mode definition begin -->    
+            <mode name="n2_lut3">     
+               <pb_type name="lut3inter" num_pb="1">      
+                  <input name="in" num_pins="3"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="ble3" num_pb="2">       
+                     <input name="in" num_pins="3"/>       
+                     <output name="out" num_pins="1"/>       
+                     <clock name="clk" num_pins="1"/>       
+                     <!-- Define the LUT -->       
+                     <pb_type name="lut3" blif_model=".names" num_pb="1" class="lut">        
+                        <input name="in" num_pins="3" port_class="lut_in"/>        
+                        <output name="out" num_pins="1" port_class="lut_out"/>        
+                        <!-- LUT timing using delay matrix -->        
+                        <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                           we instead take the average of these numbers to get more stable results
                      82e-12
                      173e-12
@ -360,108 +359,108 @@
                      263e-12
                      398e-12
                      -->        
-                <delay_matrix type="max" in_port="lut3.in" out_port="lut3.out">
+                        <delay_matrix type="max" in_port="lut3.in" out_port="lut3.out">
                  235e-12
                  235e-12
                  235e-12
                </delay_matrix>        
-              </pb_type>
-              <!-- Define the flip-flop -->
-              <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-                <input name="D" num_pins="1" port_class="D"/>
-                <output name="Q" num_pins="1" port_class="Q"/>
-                <clock name="clk" num_pins="1" port_class="clock"/>
-                <T_setup value="66e-12" port="ff.D" clock="clk"/>
-                <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="ble3.in[2:0]" output="lut3[0:0].in[2:0]"/>
-                <direct name="direct2" input="lut3[0:0].out" output="ff[0:0].D">
-                  <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                  <pack_pattern name="ble3" in_port="lut3[0:0].out" out_port="ff[0:0].D"/>
-                </direct>
-                <direct name="direct3" input="ble3.clk" output="ff[0:0].clk"/>
-                <mux name="mux1" input="ff[0:0].Q lut3.out[0:0]" output="ble3.out[0:0]">
-                  <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                  <delay_constant max="25e-12" in_port="lut3.out[0:0]" out_port="ble3.out[0:0]"/>
-                  <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble3.out[0:0]"/>
-                </mux>
-              </interconnect>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="lut3inter.in" output="ble3[0:0].in"/>
-              <direct name="direct2" input="lut3inter.in" output="ble3[1:1].in"/>
-              <direct name="direct3" input="ble3[1:0].out" output="lut3inter.out"/>
-              <complete name="complete1" input="lut3inter.clk" output="ble3[1:0].clk"/>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[2:0]" output="lut3inter.in"/>
-            <direct name="direct2" input="lut3inter.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="lut3inter.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Dual 3-LUT mode definition end -->
-        <!-- BEGIN lutram mode -->
-        <mode name="lutram">
-          <pb_type name="lutram" num_pb="1">
-            <input name="in" num_pins="4"/>
-            <output name="out" num_pins="2"/>
-            <clock name="clk" num_pins="1"/>
-            <pb_type name="spram_4x1" blif_model=".subckt spram_4x1" num_pb="1">
-              <input name="wr_en" num_pins="1"/>
-              <input name="addr" num_pins="2"/>
-              <input name="d_in" num_pins="1"/>
-              <output name="d_out" num_pins="1"/>
-			  <clock name="clk" num_pins="1" port_class="clock"/>
-          		<T_setup value="509e-12" port="spram_4x1.wr_en" clock="clk"/>
-          		<T_setup value="509e-12" port="spram_4x1.addr" clock="clk"/>
-         		<T_setup value="509e-12" port="spram_4x1.d_in" clock="clk"/>
-          		<T_clock_to_Q max="1.234e-9" port="spram_4x1.d_out" clock="clk"/>
-            </pb_type>
-            <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="clock_mem" input="lutram.clk" output="spram_4x1.clk"/>
-              <direct name="mem_wr_en" input="lutram.in[0]" output="spram_4x1.wr_en"/>
-              <direct name="mem_addr" input="lutram.in[2:1]" output="spram_4x1.addr"/>
-              <direct name="mem_d_in" input="lutram.in[3]" output="spram_4x1.d_in"/>
-              <complete name="ff_d" input="spram_4x1.d_out" output="ff.D"/>
-              <complete name="clock_ff" input="lutram.clk" output="ff.clk"/>
-              <mux name="mem_out_0" input="ff[0].Q spram_4x1.d_out" output="lutram.out[0]">
-                <delay_constant max="25e-12" in_port="spram_4x1.d_out" out_port="lutram.out[0]"/>
-                <delay_constant max="45e-12" in_port="ff[0].Q" out_port="lutram.out"/>
-              </mux>
-              <mux name="mem_out_1" input="ff[1].Q spram_4x1.d_out" output="lutram.out[1]">
-                <delay_constant max="25e-12" in_port="spram_4x1.d_out" out_port="lutram.out[1]"/>
-                <delay_constant max="45e-12" in_port="ff[1].Q" out_port="lutram.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[3:0]" output="lutram.in"/>
-            <direct name="direct3" input="fle.clk" output="lutram.clk"/>
-            <direct name="direct4" input="lutram.out" output="fle.out"/>
-          </interconnect>
-        </mode>
-        <!-- 4-LUT mode definition begin -->
-        <mode name="n1_lut4">
-          <!-- Define 4-LUT mode -->
-          <pb_type name="ble4" num_pb="1">
-            <input name="in" num_pins="4"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Define LUT -->
-            <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+                     </pb_type>       
+                     <!-- Define the flip-flop -->       
+                     <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">        
+                        <input name="D" num_pins="1" port_class="D"/>        
+                        <output name="Q" num_pins="1" port_class="Q"/>        
+                        <clock name="clk" num_pins="1" port_class="clock"/>        
+                        <T_setup value="66e-12" port="ff.D" clock="clk"/>        
+                        <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="ble3.in[2:0]" output="lut3[0:0].in[2:0]"/>        
+                        <direct name="direct2" input="lut3[0:0].out" output="ff[0:0].D">         
+                           <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->         
+                           <pack_pattern name="ble3" in_port="lut3[0:0].out" out_port="ff[0:0].D"/>         
+                        </direct>        
+                        <direct name="direct3" input="ble3.clk" output="ff[0:0].clk"/>        
+                        <mux name="mux1" input="ff[0:0].Q lut3.out[0:0]" output="ble3.out[0:0]">         
+                           <!-- LUT to output is faster than FF to output on a Stratix IV -->         
+                           <delay_constant max="25e-12" in_port="lut3.out[0:0]" out_port="ble3.out[0:0]"/>         
+                           <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble3.out[0:0]"/>         
+                        </mux>        
+                     </interconnect>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="lut3inter.in" output="ble3[0:0].in"/>       
+                     <direct name="direct2" input="lut3inter.in" output="ble3[1:1].in"/>       
+                     <direct name="direct3" input="ble3[1:0].out" output="lut3inter.out"/>       
+                     <complete name="complete1" input="lut3inter.clk" output="ble3[1:0].clk"/>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[2:0]" output="lut3inter.in"/>      
+                  <direct name="direct2" input="lut3inter.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="lut3inter.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Dual 3-LUT mode definition end -->    
+            <!-- BEGIN lutram mode -->    
+            <mode name="lutram">     
+               <pb_type name="lutram" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="spram_4x1" blif_model=".subckt spram_4x1" num_pb="1">       
+                     <input name="wr_en" num_pins="1"/>       
+                     <input name="addr" num_pins="2"/>       
+                     <input name="d_in" num_pins="1"/>       
+                     <output name="d_out" num_pins="1"/>       
+			         <clock name="clk" num_pins="1" port_class="clock"/>       
+          		       <T_setup value="509e-12" port="spram_4x1.wr_en" clock="clk"/>       
+          		       <T_setup value="509e-12" port="spram_4x1.addr" clock="clk"/>       
+         		       <T_setup value="509e-12" port="spram_4x1.d_in" clock="clk"/>       
+          		       <T_clock_to_Q max="1.234e-9" port="spram_4x1.d_out" clock="clk"/>       
+                  </pb_type>      
+                  <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="clock_mem" input="lutram.clk" output="spram_4x1.clk"/>       
+                     <direct name="mem_wr_en" input="lutram.in[0]" output="spram_4x1.wr_en"/>       
+                     <direct name="mem_addr" input="lutram.in[2:1]" output="spram_4x1.addr"/>       
+                     <direct name="mem_d_in" input="lutram.in[3]" output="spram_4x1.d_in"/>       
+                     <complete name="ff_d" input="spram_4x1.d_out" output="ff.D"/>       
+                     <complete name="clock_ff" input="lutram.clk" output="ff.clk"/>       
+                     <mux name="mem_out_0" input="ff[0].Q spram_4x1.d_out" output="lutram.out[0]">        
+                        <delay_constant max="25e-12" in_port="spram_4x1.d_out" out_port="lutram.out[0]"/>        
+                        <delay_constant max="45e-12" in_port="ff[0].Q" out_port="lutram.out"/>        
+                     </mux>       
+                     <mux name="mem_out_1" input="ff[1].Q spram_4x1.d_out" output="lutram.out[1]">        
+                        <delay_constant max="25e-12" in_port="spram_4x1.d_out" out_port="lutram.out[1]"/>        
+                        <delay_constant max="45e-12" in_port="ff[1].Q" out_port="lutram.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[3:0]" output="lutram.in"/>      
+                  <direct name="direct3" input="fle.clk" output="lutram.clk"/>      
+                  <direct name="direct4" input="lutram.out" output="fle.out"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 4-LUT mode definition begin -->    
+            <mode name="n1_lut4">     
+               <!-- Define 4-LUT mode -->     
+               <pb_type name="ble4" num_pb="1">      
+                  <input name="in" num_pins="4"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
@ -470,45 +469,45 @@
                  398e-12
                  397e-12
                  -->       
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                261e-12
                261e-12
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>
-              <direct name="direct2" input="lut4.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble4.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble4.in"/>
-            <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble4.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 4-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble4.in" output="lut4[0:0].in"/>       
+                     <direct name="direct2" input="lut4.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble4" in_port="lut4.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble4.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut4.out" output="ble4.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut4.out" out_port="ble4.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble4.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble4.in"/>      
+                  <direct name="direct2" input="ble4.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble4.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 4-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -517,23 +516,23 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>
+            <complete name="crossbar" input="clb.I fle[3:0].out" output="fle[3:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[3:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[3:0].out" out_port="fle[3:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[3:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[3:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[3:0].out[0:0]" output="clb.O[3:0]"/>
-        <direct name="clbouts2" input="fle[3:0].out[1:1]" output="clb.O[7:4]"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[3:0].out[0:0]" output="clb.O[3:0]"/>    
+            <direct name="clbouts2" input="fle[3:0].out[1:1]" output="clb.O[7:4]"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k4_frac_N8_tileable_register_scan_chain_nonLR_caravel_io_skywater130nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N8_tileable_register_scan_chain_nonLR_caravel_io_skywater130nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N8_tileable_register_scan_chain_nonLR_embedded_io_skywater130nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N8_tileable_register_scan_chain_nonLR_embedded_io_skywater130nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_register_scan_chain_nonLR_caravel_io_skywater130nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_register_scan_chain_nonLR_caravel_io_skywater130nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_softadderSuperLUT_register_scan_chain_nonLR_caravel_io_skywater130nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_softadderSuperLUT_register_scan_chain_nonLR_caravel_io_skywater130nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_softadder_register_scan_chain_dsp8_nonLR_caravel_io_skywater130nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_softadder_register_scan_chain_dsp8_nonLR_caravel_io_skywater130nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_softadder_register_scan_chain_frac_dsp16_nonLR_caravel_io_skywater130nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_softadder_register_scan_chain_frac_dsp16_nonLR_caravel_io_skywater130nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_softadder_register_scan_chain_nonLR_caravel_io_skywater130nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_softadder_register_scan_chain_nonLR_caravel_io_skywater130nm.xml
--- a/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_softadder_register_scan_chain_wide_frac_dsp16_nonLR_caravel_io_skywater130nm.xml
+++ b/openfpga_flow/vpr_arch/k4_frac_N8_tileable_reset_softadder_register_scan_chain_wide_frac_dsp16_nonLR_caravel_io_skywater130nm.xml
--- a/openfpga_flow/vpr_arch/k6_N10_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_N10_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -12,9 +12,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,56 +22,56 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="40" equivalent="full"/>
-      <output name="O" num_pins="10" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="false">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="40" equivalent="full"/>    
+          <output name="O" num_pins="10" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="false">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -87,20 +86,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -113,80 +112,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -195,30 +194,30 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="40" equivalent="full"/>
-      <output name="O" num_pins="10" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="40" equivalent="full"/>   
+         <output name="O" num_pins="10" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 6-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="10">
-        <input name="in" num_pins="6"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 6-LUT mode definition begin -->
-        <mode name="n1_lut6">
-          <!-- Define 6-LUT mode -->
-          <pb_type name="ble6" num_pb="1">
+         <pb_type name="fle" num_pb="10">    
            <input name="in" num_pins="6"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="6" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+            <!-- 6-LUT mode definition begin -->    
+            <mode name="n1_lut6">     
+               <!-- Define 6-LUT mode -->     
+               <pb_type name="ble6" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="6" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
@ -227,7 +226,7 @@
                  398e-12
                  397e-12
                  -->       
-              <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
+                     <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
                261e-12
                261e-12
                261e-12
@ -235,39 +234,39 @@
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>
-              <direct name="direct2" input="lut6.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble6.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble6.in"/>
-            <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble6.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>       
+                     <direct name="direct2" input="lut6.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble6.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble6.in"/>      
+                  <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble6.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -276,22 +275,22 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>
+            <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[9:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[9:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[9:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[9:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k6_N10_tileable_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_N10_tileable_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Architecture with no fracturable LUTs

  - 40 nm technology
@ -12,9 +12,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,56 +22,56 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="40" equivalent="full"/>
-      <output name="O" num_pins="10" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="40" equivalent="full"/>    
+          <output name="O" num_pins="10" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -87,20 +86,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -113,80 +112,80 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- A mode denotes the physical implementation of an I/O 
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- IOs can operate as either inputs or outputs.
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -195,30 +194,30 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="40" equivalent="full"/>
-      <output name="O" num_pins="10" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe basic logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="40" equivalent="full"/>   
+         <output name="O" num_pins="10" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe basic logic element.  
             Each basic logic element has a 6-LUT that can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="10">
-        <input name="in" num_pins="6"/>
-        <output name="out" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- 6-LUT mode definition begin -->
-        <mode name="n1_lut6">
-          <!-- Define 6-LUT mode -->
-          <pb_type name="ble6" num_pb="1">
+         <pb_type name="fle" num_pb="10">    
            <input name="in" num_pins="6"/>    
            <output name="out" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <!-- Define LUT -->
-            <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="6" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+            <!-- 6-LUT mode definition begin -->    
+            <mode name="n1_lut6">     
+               <!-- Define 6-LUT mode -->     
+               <pb_type name="ble6" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="6" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
@ -227,7 +226,7 @@
                  398e-12
                  397e-12
                  -->       
-              <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
+                     <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
                261e-12
                261e-12
                261e-12
@ -235,39 +234,39 @@
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>
-              <direct name="direct2" input="lut6.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble6.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble6.in"/>
-            <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble6.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>       
+                     <direct name="direct2" input="lut6.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble6.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble6.in"/>      
+                  <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble6.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -276,22 +275,22 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>
+            <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[9:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[9:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[9:0].out" output="clb.O"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[9:0].out" output="clb.O"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k6_frac_N10_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_40nm.xml
@ -1,4 +1,4 @@
-<!--
+<?xml version="1.0" ?><!--
  Flagship Heterogeneous Architecture (No Carry Chains) for VTR 7.0.

  - 40 nm technology
@ -12,9 +12,8 @@
  Based on flagship k6_frac_N10_mem32K_40nm.xml architecture.

  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!--
+--><architecture> 
+   <!--
       ODIN II specific config begins
       Describes the types of user-specified netlist blocks (in blif, this corresponds to
       ".model [type_of_block]") that this architecture supports.
@ -23,77 +22,77 @@
       already special structures in blif (.names, .input, .output, and .latch)
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="frac_lut6">
-      <input_ports>
-        <port name="in"/>
-      </input_ports>
-      <output_ports>
-        <port name="lut5_out"/>
-        <port name="lut6_out"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="frac_lut6">   
+         <input_ports>    
+            <port name="in"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="lut5_out"/>    
+            <port name="lut6_out"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
         If you need to register the I/O, define clocks in the circuit models
         These clocks can be handled in back-end
     -->  
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="40" equivalent="full"/>
-      <output name="O" num_pins="20" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="false">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="40" equivalent="full"/>    
+          <output name="O" num_pins="20" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="false">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of
@ -108,20 +107,20 @@
 			     proposed FPGA, and which is also 40 nm
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -134,86 +133,86 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+      <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->  
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
           If you need to register the I/O, define clocks in the circuit models
           These clocks can be handled in back-end
       -->   
-      <!-- A mode denotes the physical implementation of an I/O
+         <!-- A mode denotes the physical implementation of an I/O
           This mode will be not packable but is mainly used for fabric verilog generation
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   

-      <!-- IOs can operate as either inputs or outputs.
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency,
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency,
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -222,87 +221,87 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range.
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="40" equivalent="full"/>
-      <output name="O" num_pins="20" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe fracturable logic element.
+      <pb_type name="clb">   
+         <input name="I" num_pins="40" equivalent="full"/>   
+         <output name="O" num_pins="20" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe fracturable logic element.
             Each fracturable logic element has a 6-LUT that can alternatively operate as two 5-LUTs with shared inputs.
             The outputs of the fracturable logic element can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="10">
-        <input name="in" num_pins="6"/>
-        <output name="out" num_pins="2"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- Physical mode definition begin (physical implementation of the fle) -->
-        <mode name="physical" disable_packing="true">
-          <pb_type name="fabric" num_pb="1">
+         <pb_type name="fle" num_pb="10">    
            <input name="in" num_pins="6"/>    
            <output name="out" num_pins="2"/>    
            <clock name="clk" num_pins="1"/>    
-            <pb_type name="frac_logic" num_pb="1">
-              <input name="in" num_pins="6"/>
-              <output name="out" num_pins="2"/>
-              <!-- Define LUT -->
-              <pb_type name="frac_lut6" blif_model=".subckt frac_lut6" num_pb="1">
-                <input name="in" num_pins="6"/>
-                <output name="lut5_out" num_pins="2"/>
-                <output name="lut6_out" num_pins="1"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="frac_logic.in" output="frac_lut6.in"/>
-                <direct name="direct2" input="frac_lut6.lut5_out[1]" output="frac_logic.out[1]"/>
-                <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->
-                <mux name="mux1" input="frac_lut6.lut6_out frac_lut6.lut5_out[0]" output="frac_logic.out[0]"/>
-              </interconnect>
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="fabric.in" output="frac_logic.in"/>
-              <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>
-              <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>
-              <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>
-                <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>
-              </mux>
-              <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>
-                <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="fabric.in"/>
-            <direct name="direct2" input="fabric.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="fabric.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Physical mode definition end (physical implementation of the fle) -->
-        <!-- Dual 5-LUT mode definition begin -->
-        <mode name="n2_lut5">
-          <pb_type name="lut5inter" num_pb="1">
-            <input name="in" num_pins="5"/>
-            <output name="out" num_pins="2"/>
-            <clock name="clk" num_pins="1"/>
-            <pb_type name="ble5" num_pb="2">
-              <input name="in" num_pins="5"/>
-              <output name="out" num_pins="1"/>
-              <clock name="clk" num_pins="1"/>
-              <!-- Define the LUT -->
-              <pb_type name="lut5" blif_model=".names" num_pb="1" class="lut">
-                <input name="in" num_pins="5" port_class="lut_in"/>
-                <output name="out" num_pins="1" port_class="lut_out"/>
-                <!-- LUT timing using delay matrix -->
-                <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+            <!-- Physical mode definition begin (physical implementation of the fle) -->    
+            <mode name="physical" disable_packing="true">     
+               <pb_type name="fabric" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="frac_logic" num_pb="1">       
+                     <input name="in" num_pins="6"/>       
+                     <output name="out" num_pins="2"/>       
+                     <!-- Define LUT -->       
+                     <pb_type name="frac_lut6" blif_model=".subckt frac_lut6" num_pb="1">        
+                        <input name="in" num_pins="6"/>        
+                        <output name="lut5_out" num_pins="2"/>        
+                        <output name="lut6_out" num_pins="1"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="frac_logic.in" output="frac_lut6.in"/>        
+                        <direct name="direct2" input="frac_lut6.lut5_out[1]" output="frac_logic.out[1]"/>        
+                        <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->        
+                        <mux name="mux1" input="frac_lut6.lut6_out frac_lut6.lut5_out[0]" output="frac_logic.out[0]"/>        
+                     </interconnect>       
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="fabric.in" output="frac_logic.in"/>       
+                     <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>       
+                     <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>       
+                     <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>        
+                        <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>        
+                     </mux>       
+                     <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>        
+                        <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="fabric.in"/>      
+                  <direct name="direct2" input="fabric.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="fabric.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Physical mode definition end (physical implementation of the fle) -->    
+            <!-- Dual 5-LUT mode definition begin -->    
+            <mode name="n2_lut5">     
+               <pb_type name="lut5inter" num_pb="1">      
+                  <input name="in" num_pins="5"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="ble5" num_pb="2">       
+                     <input name="in" num_pins="5"/>       
+                     <output name="out" num_pins="1"/>       
+                     <clock name="clk" num_pins="1"/>       
+                     <!-- Define the LUT -->       
+                     <pb_type name="lut5" blif_model=".names" num_pb="1" class="lut">        
+                        <input name="in" num_pins="5" port_class="lut_in"/>        
+                        <output name="out" num_pins="1" port_class="lut_out"/>        
+                        <!-- LUT timing using delay matrix -->        
+                        <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                           we instead take the average of these numbers to get more stable results
                      82e-12
                      173e-12
@ -310,63 +309,63 @@
                      263e-12
                      398e-12
                      -->        
-                <delay_matrix type="max" in_port="lut5.in" out_port="lut5.out">
+                        <delay_matrix type="max" in_port="lut5.in" out_port="lut5.out">
                  235e-12
                  235e-12
                  235e-12
                  235e-12
                  235e-12
                </delay_matrix>        
-              </pb_type>
-              <!-- Define the flip-flop -->
-              <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-                <input name="D" num_pins="1" port_class="D"/>
-                <output name="Q" num_pins="1" port_class="Q"/>
-                <clock name="clk" num_pins="1" port_class="clock"/>
-                <T_setup value="66e-12" port="ff.D" clock="clk"/>
-                <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="ble5.in[4:0]" output="lut5[0:0].in[4:0]"/>
-                <direct name="direct2" input="lut5[0:0].out" output="ff[0:0].D">
-                  <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                  <pack_pattern name="ble5" in_port="lut5[0:0].out" out_port="ff[0:0].D"/>
-                </direct>
-                <direct name="direct3" input="ble5.clk" output="ff[0:0].clk"/>
-                <mux name="mux1" input="ff[0:0].Q lut5.out[0:0]" output="ble5.out[0:0]">
-                  <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                  <delay_constant max="25e-12" in_port="lut5.out[0:0]" out_port="ble5.out[0:0]"/>
-                  <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble5.out[0:0]"/>
-                </mux>
-              </interconnect>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="lut5inter.in" output="ble5[0:0].in"/>
-              <direct name="direct2" input="lut5inter.in" output="ble5[1:1].in"/>
-              <direct name="direct3" input="ble5[1:0].out" output="lut5inter.out"/>
-              <complete name="complete1" input="lut5inter.clk" output="ble5[1:0].clk"/>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[4:0]" output="lut5inter.in"/>
-            <direct name="direct2" input="lut5inter.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="lut5inter.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Dual 5-LUT mode definition end -->
-        <!-- 6-LUT mode definition begin -->
-        <mode name="n1_lut6">
-          <!-- Define 6-LUT mode -->
-          <pb_type name="ble6" num_pb="1">
-            <input name="in" num_pins="6"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Define LUT -->
-            <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="6" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+                     </pb_type>       
+                     <!-- Define the flip-flop -->       
+                     <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">        
+                        <input name="D" num_pins="1" port_class="D"/>        
+                        <output name="Q" num_pins="1" port_class="Q"/>        
+                        <clock name="clk" num_pins="1" port_class="clock"/>        
+                        <T_setup value="66e-12" port="ff.D" clock="clk"/>        
+                        <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="ble5.in[4:0]" output="lut5[0:0].in[4:0]"/>        
+                        <direct name="direct2" input="lut5[0:0].out" output="ff[0:0].D">         
+                           <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->         
+                           <pack_pattern name="ble5" in_port="lut5[0:0].out" out_port="ff[0:0].D"/>         
+                        </direct>        
+                        <direct name="direct3" input="ble5.clk" output="ff[0:0].clk"/>        
+                        <mux name="mux1" input="ff[0:0].Q lut5.out[0:0]" output="ble5.out[0:0]">         
+                           <!-- LUT to output is faster than FF to output on a Stratix IV -->         
+                           <delay_constant max="25e-12" in_port="lut5.out[0:0]" out_port="ble5.out[0:0]"/>         
+                           <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble5.out[0:0]"/>         
+                        </mux>        
+                     </interconnect>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="lut5inter.in" output="ble5[0:0].in"/>       
+                     <direct name="direct2" input="lut5inter.in" output="ble5[1:1].in"/>       
+                     <direct name="direct3" input="ble5[1:0].out" output="lut5inter.out"/>       
+                     <complete name="complete1" input="lut5inter.clk" output="ble5[1:0].clk"/>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[4:0]" output="lut5inter.in"/>      
+                  <direct name="direct2" input="lut5inter.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="lut5inter.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Dual 5-LUT mode definition end -->    
+            <!-- 6-LUT mode definition begin -->    
+            <mode name="n1_lut6">     
+               <!-- Define 6-LUT mode -->     
+               <pb_type name="ble6" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="6" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
@ -375,7 +374,7 @@
                  398e-12
                  397e-12
                  -->       
-              <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
+                     <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
                261e-12
                261e-12
                261e-12
@ -383,39 +382,39 @@
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>
-              <direct name="direct2" input="lut6.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble6.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble6.in"/>
-            <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble6.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>       
+                     <direct name="direct2" input="lut6.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble6.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble6.in"/>      
+                  <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble6.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -424,23 +423,23 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>
+            <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[9:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[9:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs,
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[9:0].out[0:0]" output="clb.O[9:0]"/>
-        <direct name="clbouts2" input="fle[9:0].out[1:1]" output="clb.O[19:10]"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[9:0].out[0:0]" output="clb.O[9:0]"/>    
+            <direct name="clbouts2" input="fle[9:0].out[1:1]" output="clb.O[19:10]"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k6_frac_N10_adder_chain_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_adder_chain_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Flagship Heterogeneous Architecture with Carry Chains for VTR 7.0.

  - 40 nm technology
@ -75,9 +75,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -86,97 +85,97 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <model name="adder">
-      <input_ports>
-        <port name="a" combinational_sink_ports="sumout cout"/>
-        <port name="b" combinational_sink_ports="sumout cout"/>
-        <port name="cin" combinational_sink_ports="sumout cout"/>
-      </input_ports>
-      <output_ports>
-        <port name="cout"/>
-        <port name="sumout"/>
-      </output_ports>
-    </model>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="frac_lut6">
-      <input_ports>
-        <port name="in"/>
-      </input_ports>
-      <output_ports>
-        <port name="lut4_out"/>
-        <port name="lut5_out"/>
-        <port name="lut6_out"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="40" equivalent="full"/>
-      <input name="cin" num_pins="1"/>
-      <output name="O" num_pins="20" equivalent="none"/>
-      <output name="cout" num_pins="1"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">
-        <fc_override port_name="cin" fc_type="frac" fc_val="0"/>
-        <fc_override port_name="cout" fc_type="frac" fc_val="0"/>
-      </fc>
-      <!--  Highly recommand to customize pin location when direct connection is used!!! -->
-      <!--pinlocations pattern="spread"/-->
-      <pinlocations pattern="custom">
-        <loc side="left">clb.clk</loc>
-        <loc side="top">clb.cin</loc>
-        <loc side="right">clb.O[9:0] clb.I[19:0]</loc>
-        <loc side="bottom">clb.cout clb.O[19:10] clb.I[39:20]</loc>
-      </pinlocations>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="false">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="4x4" width="6" height="6">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <model name="adder">   
+         <input_ports>    
+            <port name="a" combinational_sink_ports="sumout cout"/>    
+            <port name="b" combinational_sink_ports="sumout cout"/>    
+            <port name="cin" combinational_sink_ports="sumout cout"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="cout"/>    
+            <port name="sumout"/>    
+         </output_ports>   
+      </model>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="frac_lut6">   
+         <input_ports>    
+            <port name="in"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="lut4_out"/>    
+            <port name="lut5_out"/>    
+            <port name="lut6_out"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="40" equivalent="full"/>    
+          <input name="cin" num_pins="1"/>    
+          <output name="O" num_pins="20" equivalent="none"/>    
+          <output name="cout" num_pins="1"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">     
+             <fc_override port_name="cin" fc_type="frac" fc_val="0"/>     
+             <fc_override port_name="cout" fc_type="frac" fc_val="0"/>     
+          </fc>    
+          <!--  Highly recommand to customize pin location when direct connection is used!!! -->    
+          <!--pinlocations pattern="spread"/-->    
+          <pinlocations pattern="custom">     
+             <loc side="left">clb.clk</loc>     
+             <loc side="top">clb.cin</loc>     
+             <loc side="right">clb.O[9:0] clb.I[19:0]</loc>     
+             <loc side="bottom">clb.cout clb.O[19:10] clb.I[39:20]</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="false">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="4x4" width="6" height="6">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -191,20 +190,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	    -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
           book area formula. This means the mux transistors are about 5x minimum drive strength.
           We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
           mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -217,90 +216,90 @@
           2.5x when looking up in Jeff's tables.
           Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
           This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
             With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
             reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <directlist>
-    <direct name="adder_carry" from_pin="clb.cout" to_pin="clb.cin" x_offset="0" y_offset="-1" z_offset="0"/>
-  </directlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
+      <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->  
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <directlist>  
+      <direct name="adder_carry" from_pin="clb.cout" to_pin="clb.cin" x_offset="0" y_offset="-1" z_offset="0"/>  
+   </directlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   

-      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+         <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
           If you need to register the I/O, define clocks in the circuit models
           These clocks can be handled in back-end
       -->   
-      <!-- A mode denotes the physical implementation of an I/O 
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   

-      <!-- IOs can operate as either inputs or outputs.
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -309,115 +308,115 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="40" equivalent="full"/>
-      <input name="cin" num_pins="1"/>
-      <output name="O" num_pins="20" equivalent="none"/>
-      <output name="cout" num_pins="1"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe fracturable logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="40" equivalent="full"/>   
+         <input name="cin" num_pins="1"/>   
+         <output name="O" num_pins="20" equivalent="none"/>   
+         <output name="cout" num_pins="1"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe fracturable logic element.  
             Each fracturable logic element has a 6-LUT that can alternatively operate as two 5-LUTs with shared inputs. 
             The outputs of the fracturable logic element can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="10">
-        <input name="in" num_pins="6"/>
-        <input name="cin" num_pins="1"/>
-        <output name="out" num_pins="2"/>
-        <output name="cout" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- Physical mode definition begin (physical implementation of the fle) -->
-        <mode name="physical" disable_packing="true">
-          <pb_type name="fabric" num_pb="1">
+         <pb_type name="fle" num_pb="10">    
            <input name="in" num_pins="6"/>    
            <input name="cin" num_pins="1"/>    
            <output name="out" num_pins="2"/>    
            <output name="cout" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <pb_type name="frac_logic" num_pb="1">
-              <input name="in" num_pins="6"/>
-              <output name="lut4_out" num_pins="4"/>
-              <output name="out" num_pins="2"/>
-              <!-- Define LUT -->
-              <pb_type name="frac_lut6" blif_model=".subckt frac_lut6" num_pb="1">
-                <input name="in" num_pins="6"/>
-                <output name="lut4_out" num_pins="4"/>
-                <output name="lut5_out" num_pins="2"/>
-                <output name="lut6_out" num_pins="1"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="frac_logic.in" output="frac_lut6.in"/>
-                <direct name="direct2" input="frac_lut6.lut4_out" output="frac_logic.lut4_out"/>
-                <direct name="direct3" input="frac_lut6.lut5_out[1]" output="frac_logic.out[1]"/>
-                <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->
-                <mux name="mux1" input="frac_lut6.lut6_out frac_lut6.lut5_out[0]" output="frac_logic.out[0]"/>
-              </interconnect>
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>         
-            <!-- Define adders -->
-            <pb_type name="adder" blif_model=".subckt adder" num_pb="2">
-              <input name="a" num_pins="1"/>
-              <input name="b" num_pins="1"/>
-              <input name="cin" num_pins="1"/>
-              <output name="cout" num_pins="1"/>
-              <output name="sumout" num_pins="1"/>
-              <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.cin" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.cout"/>
-              <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.cout"/>
-              <delay_constant max="0.01e-9" in_port="adder.cin" out_port="adder.cout"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="fabric.in" output="frac_logic.in"/>
-              <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>
-              <direct name="direct3" input="fabric.cin" output="adder[0:0].cin"/>
-              <direct name="direct4" input="adder[0:0].cout" output="adder[1:1].cin"/>
-              <direct name="direct5" input="adder[1:1].cout" output="fabric.cout"/>
-              <direct name="direct6" input="frac_logic.lut4_out[0:0]" output="adder[0:0].a"/>
-              <direct name="direct7" input="frac_logic.lut4_out[1:1]" output="adder[0:0].b"/>
-              <direct name="direct8" input="frac_logic.lut4_out[2:2]" output="adder[1:1].a"/>
-              <direct name="direct9" input="frac_logic.lut4_out[3:3]" output="adder[1:1].b"/>
-              <complete name="direct10" input="fabric.clk" output="ff[1:0].clk"/>
-              <mux name="mux1" input="adder[0].sumout ff[0].Q frac_logic.out[0]" output="fabric.out[0]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>
-                <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>
-              </mux>
-              <mux name="mux2" input="adder[1].sumout ff[1].Q frac_logic.out[1]" output="fabric.out[1]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>
-                <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="fabric.in"/>
-            <direct name="direct2" input="fle.cin" output="fabric.cin"/>
-            <direct name="direct3" input="fabric.out" output="fle.out"/>
-            <direct name="direct4" input="fabric.cout" output="fle.cout"/>
-            <direct name="direct5" input="fle.clk" output="fabric.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Physical mode definition end (physical implementation of the fle) -->
-        <!-- BEGIN fle mode of dual lut5 -->
-        <mode name="n2_lut5">
-          <pb_type name="ble5" num_pb="2">
-            <input name="in" num_pins="5"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Regular LUT mode -->
-            <pb_type name="lut5" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="5" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+            <!-- Physical mode definition begin (physical implementation of the fle) -->    
+            <mode name="physical" disable_packing="true">     
+               <pb_type name="fabric" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <input name="cin" num_pins="1"/>      
+                  <output name="out" num_pins="2"/>      
+                  <output name="cout" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="frac_logic" num_pb="1">       
+                     <input name="in" num_pins="6"/>       
+                     <output name="lut4_out" num_pins="4"/>       
+                     <output name="out" num_pins="2"/>       
+                     <!-- Define LUT -->       
+                     <pb_type name="frac_lut6" blif_model=".subckt frac_lut6" num_pb="1">        
+                        <input name="in" num_pins="6"/>        
+                        <output name="lut4_out" num_pins="4"/>        
+                        <output name="lut5_out" num_pins="2"/>        
+                        <output name="lut6_out" num_pins="1"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="frac_logic.in" output="frac_lut6.in"/>        
+                        <direct name="direct2" input="frac_lut6.lut4_out" output="frac_logic.lut4_out"/>        
+                        <direct name="direct3" input="frac_lut6.lut5_out[1]" output="frac_logic.out[1]"/>        
+                        <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->        
+                        <mux name="mux1" input="frac_lut6.lut6_out frac_lut6.lut5_out[0]" output="frac_logic.out[0]"/>        
+                     </interconnect>       
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>               
+                  <!-- Define adders -->      
+                  <pb_type name="adder" blif_model=".subckt adder" num_pb="2">       
+                     <input name="a" num_pins="1"/>       
+                     <input name="b" num_pins="1"/>       
+                     <input name="cin" num_pins="1"/>       
+                     <output name="cout" num_pins="1"/>       
+                     <output name="sumout" num_pins="1"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.cin" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.cout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.cout"/>       
+                     <delay_constant max="0.01e-9" in_port="adder.cin" out_port="adder.cout"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="fabric.in" output="frac_logic.in"/>       
+                     <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>       
+                     <direct name="direct3" input="fabric.cin" output="adder[0:0].cin"/>       
+                     <direct name="direct4" input="adder[0:0].cout" output="adder[1:1].cin"/>       
+                     <direct name="direct5" input="adder[1:1].cout" output="fabric.cout"/>       
+                     <direct name="direct6" input="frac_logic.lut4_out[0:0]" output="adder[0:0].a"/>       
+                     <direct name="direct7" input="frac_logic.lut4_out[1:1]" output="adder[0:0].b"/>       
+                     <direct name="direct8" input="frac_logic.lut4_out[2:2]" output="adder[1:1].a"/>       
+                     <direct name="direct9" input="frac_logic.lut4_out[3:3]" output="adder[1:1].b"/>       
+                     <complete name="direct10" input="fabric.clk" output="ff[1:0].clk"/>       
+                     <mux name="mux1" input="adder[0].sumout ff[0].Q frac_logic.out[0]" output="fabric.out[0]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>        
+                        <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>        
+                     </mux>       
+                     <mux name="mux2" input="adder[1].sumout ff[1].Q frac_logic.out[1]" output="fabric.out[1]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>        
+                        <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="fabric.in"/>      
+                  <direct name="direct2" input="fle.cin" output="fabric.cin"/>      
+                  <direct name="direct3" input="fabric.out" output="fle.out"/>      
+                  <direct name="direct4" input="fabric.cout" output="fle.cout"/>      
+                  <direct name="direct5" input="fle.clk" output="fabric.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Physical mode definition end (physical implementation of the fle) -->    
+            <!-- BEGIN fle mode of dual lut5 -->    
+            <mode name="n2_lut5">     
+               <pb_type name="ble5" num_pb="2">      
+                  <input name="in" num_pins="5"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Regular LUT mode -->      
+                  <pb_type name="lut5" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="5" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                  we instead take the average of these numbers to get more stable results
                82e-12
                173e-12
@ -425,138 +424,138 @@
                263e-12
                398e-12
                -->       
-              <delay_matrix type="max" in_port="lut5.in" out_port="lut5.out">
+                     <delay_matrix type="max" in_port="lut5.in" out_port="lut5.out">
                235e-12
                235e-12
                235e-12
                235e-12
                235e-12
              </delay_matrix>       
-            </pb_type>
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble5.in" output="lut5.in"/>
-              <direct name="direct2" input="lut5.out" output="ff.D">
-                <pack_pattern name="ble5" in_port="lut5.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble5.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut5.out" output="ble5.out">
-                <delay_constant max="25e-12" in_port="lut5.out" out_port="ble5.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble5.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[4:0]" output="ble5[0:0].in"/>
-            <direct name="direct2" input="fle.in[4:0]" output="ble5[1:1].in"/>
-            <complete name="direct3" input="fle.clk" output="ble5.clk"/>
-            <direct name="direct4" input="ble5.out" output="fle.out"/>
-          </interconnect>
-        </mode>
-        <!-- END fle mode of dual lut5 -->
-        <!-- BEGIN arithmetic mode of dual lut4 + adders -->
-        <mode name="arithmetic">
-          <pb_type name="arithmetic" num_pb="2">
-            <input name="in" num_pins="4"/>
-            <input name="cin" num_pins="1"/>
-            <output name="out" num_pins="1"/>
-            <output name="cout" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Special dual-LUT mode that drives adder only -->
-            <pb_type name="lut4" blif_model=".names" num_pb="2" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+                  </pb_type>      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble5.in" output="lut5.in"/>       
+                     <direct name="direct2" input="lut5.out" output="ff.D">        
+                        <pack_pattern name="ble5" in_port="lut5.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble5.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut5.out" output="ble5.out">        
+                        <delay_constant max="25e-12" in_port="lut5.out" out_port="ble5.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble5.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[4:0]" output="ble5[0:0].in"/>      
+                  <direct name="direct2" input="fle.in[4:0]" output="ble5[1:1].in"/>      
+                  <complete name="direct3" input="fle.clk" output="ble5.clk"/>      
+                  <direct name="direct4" input="ble5.out" output="fle.out"/>      
+               </interconnect>     
+            </mode>    
+            <!-- END fle mode of dual lut5 -->    
+            <!-- BEGIN arithmetic mode of dual lut4 + adders -->    
+            <mode name="arithmetic">     
+               <pb_type name="arithmetic" num_pb="2">      
+                  <input name="in" num_pins="4"/>      
+                  <input name="cin" num_pins="1"/>      
+                  <output name="out" num_pins="1"/>      
+                  <output name="cout" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Special dual-LUT mode that drives adder only -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="2" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
                  261e-12
                  263e-12
                  -->       
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                  195e-12
                  195e-12
                  195e-12
                  195e-12
                </delay_matrix>       
-            </pb_type>
-            <pb_type name="adder" blif_model=".subckt adder" num_pb="1">
-              <input name="a" num_pins="1"/>
-              <input name="b" num_pins="1"/>
-              <input name="cin" num_pins="1"/>
-              <output name="cout" num_pins="1"/>
-              <output name="sumout" num_pins="1"/>
-              <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.cin" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.cout"/>
-              <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.cout"/>
-              <delay_constant max="0.01e-9" in_port="adder.cin" out_port="adder.cout"/>
-            </pb_type>
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="clock" input="arithmetic.clk" output="ff.clk"/>
-              <direct name="lut_in1" input="arithmetic.in[3:0]" output="lut4[0:0].in[3:0]"/>
-              <direct name="lut_in2" input="arithmetic.in[3:0]" output="lut4[1:1].in[3:0]"/>
-              <direct name="lut_to_add1" input="lut4[0:0].out" output="adder.a">
+                  </pb_type>      
+                  <pb_type name="adder" blif_model=".subckt adder" num_pb="1">       
+                     <input name="a" num_pins="1"/>       
+                     <input name="b" num_pins="1"/>       
+                     <input name="cin" num_pins="1"/>       
+                     <output name="cout" num_pins="1"/>       
+                     <output name="sumout" num_pins="1"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.cin" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.cout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.cout"/>       
+                     <delay_constant max="0.01e-9" in_port="adder.cin" out_port="adder.cout"/>       
+                  </pb_type>      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="clock" input="arithmetic.clk" output="ff.clk"/>       
+                     <direct name="lut_in1" input="arithmetic.in[3:0]" output="lut4[0:0].in[3:0]"/>       
+                     <direct name="lut_in2" input="arithmetic.in[3:0]" output="lut4[1:1].in[3:0]"/>       
+                     <direct name="lut_to_add1" input="lut4[0:0].out" output="adder.a">
                </direct>       
-              <direct name="lut_to_add2" input="lut4[1:1].out" output="adder.b">
+                     <direct name="lut_to_add2" input="lut4[1:1].out" output="adder.b">
                </direct>       
-              <direct name="add_to_ff" input="adder.sumout" output="ff.D">
-                <pack_pattern name="chain" in_port="adder.sumout" out_port="ff.D"/>
-              </direct>
-              <direct name="carry_in" input="arithmetic.cin" output="adder.cin">
-                <pack_pattern name="chain" in_port="arithmetic.cin" out_port="adder.cin"/>
-              </direct>
-              <direct name="carry_out" input="adder.cout" output="arithmetic.cout">
-                <pack_pattern name="chain" in_port="adder.cout" out_port="arithmetic.cout"/>
-              </direct>
-              <mux name="sumout" input="ff.Q adder.sumout" output="arithmetic.out">
-                <delay_constant max="25e-12" in_port="adder.sumout" out_port="arithmetic.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="arithmetic.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[3:0]" output="arithmetic[0:0].in"/>
-            <direct name="direct2" input="fle.in[3:0]" output="arithmetic[1:1].in"/>
-            <direct name="carry_in" input="fle.cin" output="arithmetic[0:0].cin">
-              <pack_pattern name="chain" in_port="fle.cin" out_port="arithmetic[0:0].cin"/>
-            </direct>
-            <direct name="carry_inter" input="arithmetic[0:0].cout" output="arithmetic[1:1].cin">
-              <pack_pattern name="chain" in_port="arithmetic[0:0].cout" out_port="arithmetic[1:1].cin"/>
-            </direct>
-            <direct name="carry_out" input="arithmetic[1:1].cout" output="fle.cout">
-              <pack_pattern name="chain" in_port="arithmetic.cout" out_port="fle.cout"/>
-            </direct>
-            <complete name="direct3" input="fle.clk" output="arithmetic.clk"/>
-            <direct name="direct4" input="arithmetic.out" output="fle.out"/>
-          </interconnect>
-        </mode>
-        <!-- n2_lut5 -->
-        <mode name="n1_lut6">
-          <pb_type name="ble6" num_pb="1">
-            <input name="in" num_pins="6"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="6" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+                     <direct name="add_to_ff" input="adder.sumout" output="ff.D">        
+                        <pack_pattern name="chain" in_port="adder.sumout" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="carry_in" input="arithmetic.cin" output="adder.cin">        
+                        <pack_pattern name="chain" in_port="arithmetic.cin" out_port="adder.cin"/>        
+                     </direct>       
+                     <direct name="carry_out" input="adder.cout" output="arithmetic.cout">        
+                        <pack_pattern name="chain" in_port="adder.cout" out_port="arithmetic.cout"/>        
+                     </direct>       
+                     <mux name="sumout" input="ff.Q adder.sumout" output="arithmetic.out">        
+                        <delay_constant max="25e-12" in_port="adder.sumout" out_port="arithmetic.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="arithmetic.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[3:0]" output="arithmetic[0:0].in"/>      
+                  <direct name="direct2" input="fle.in[3:0]" output="arithmetic[1:1].in"/>      
+                  <direct name="carry_in" input="fle.cin" output="arithmetic[0:0].cin">       
+                     <pack_pattern name="chain" in_port="fle.cin" out_port="arithmetic[0:0].cin"/>       
+                  </direct>      
+                  <direct name="carry_inter" input="arithmetic[0:0].cout" output="arithmetic[1:1].cin">       
+                     <pack_pattern name="chain" in_port="arithmetic[0:0].cout" out_port="arithmetic[1:1].cin"/>       
+                  </direct>      
+                  <direct name="carry_out" input="arithmetic[1:1].cout" output="fle.cout">       
+                     <pack_pattern name="chain" in_port="arithmetic.cout" out_port="fle.cout"/>       
+                  </direct>      
+                  <complete name="direct3" input="fle.clk" output="arithmetic.clk"/>      
+                  <direct name="direct4" input="arithmetic.out" output="fle.out"/>      
+               </interconnect>     
+            </mode>    
+            <!-- n2_lut5 -->    
+            <mode name="n1_lut6">     
+               <pb_type name="ble6" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="6" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
@ -565,7 +564,7 @@
                  398e-12
                  397e-12
                  -->       
-              <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
+                     <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
                  261e-12
                  261e-12
                  261e-12
@ -573,36 +572,36 @@
                  261e-12
                  261e-12
                </delay_matrix>       
-            </pb_type>
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>
-              <direct name="direct2" input="lut6.out" output="ff.D">
-                <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble6.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">
-                <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[5:0]" output="ble6.in"/>
-            <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble6.clk"/>
-          </interconnect>
-        </mode>
-        <!-- n1_lut6 -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a 50% depop crossbar built using small full xbars to get sets of logically equivalent pins at inputs of CLB 
+                  </pb_type>      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>       
+                     <direct name="direct2" input="lut6.out" output="ff.D">        
+                        <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble6.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">        
+                        <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[5:0]" output="ble6.in"/>      
+                  <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble6.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- n1_lut6 -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a 50% depop crossbar built using small full xbars to get sets of logically equivalent pins at inputs of CLB 
           The delays below come from Stratix IV. the delay through a connection block
           input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
           delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -611,34 +610,34 @@
           Since all our outputs LUT outputs go to a BLE output, and have a delay of 
           25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
           to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>
-        </complete>
+            <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>     
+            </complete>    

-        <complete name="clks" input="clb.clk" output="fle[9:0].clk">
+            <complete name="clks" input="clb.clk" output="fle[9:0].clk">
        </complete>    
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
                 By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
                 then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
                 naive specification).
          -->    
-        <direct name="clbouts1" input="fle[9:0].out[0:0]" output="clb.O[9:0]"/>
-        <direct name="clbouts2" input="fle[9:0].out[1:1]" output="clb.O[19:10]"/>
-        <!-- Carry chain links -->
-        <direct name="carry_in" input="clb.cin" output="fle[0:0].cin">
-          <!-- Put all inter-block carry chain delay on this one edge -->
-          <delay_constant max="0.16e-9" in_port="clb.cin" out_port="fle[0:0].cin"/>
-          <pack_pattern name="chain" in_port="clb.cin" out_port="fle[0:0].cin"/>
-        </direct>
-        <direct name="carry_out" input="fle[9:9].cout" output="clb.cout">
-          <pack_pattern name="chain" in_port="fle[9:9].cout" out_port="clb.cout"/>
-        </direct>
-        <direct name="carry_link" input="fle[8:0].cout" output="fle[9:1].cin">
-          <pack_pattern name="chain" in_port="fle[8:0].cout" out_port="fle[9:1].cin"/>
-        </direct>
-      </interconnect>
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[9:0].out[0:0]" output="clb.O[9:0]"/>    
+            <direct name="clbouts2" input="fle[9:0].out[1:1]" output="clb.O[19:10]"/>    
+            <!-- Carry chain links -->    
+            <direct name="carry_in" input="clb.cin" output="fle[0:0].cin">     
+               <!-- Put all inter-block carry chain delay on this one edge -->     
+               <delay_constant max="0.16e-9" in_port="clb.cin" out_port="fle[0:0].cin"/>     
+               <pack_pattern name="chain" in_port="clb.cin" out_port="fle[0:0].cin"/>     
+            </direct>    
+            <direct name="carry_out" input="fle[9:9].cout" output="clb.cout">     
+               <pack_pattern name="chain" in_port="fle[9:9].cout" out_port="clb.cout"/>     
+            </direct>    
+            <direct name="carry_link" input="fle[8:0].cout" output="fle[9:1].cin">     
+               <pack_pattern name="chain" in_port="fle[8:0].cout" out_port="fle[9:1].cin"/>     
+            </direct>    
+         </interconnect>   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k6_frac_N10_adder_chain_mem16K_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_adder_chain_mem16K_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Flagship Heterogeneous Architecture (No Carry Chains) for VTR 7.0.

  - 40 nm technology
@ -12,9 +12,8 @@
  Based on flagship k6_frac_N10_mem32K_40nm.xml architecture.

  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,77 +22,77 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="frac_lut6">
-      <input_ports>
-        <port name="in"/>
-      </input_ports>
-      <output_ports>
-        <port name="lut5_out"/>
-        <port name="lut6_out"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="frac_lut6">   
+         <input_ports>    
+            <port name="in"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="lut5_out"/>    
+            <port name="lut6_out"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
         If you need to register the I/O, define clocks in the circuit models
         These clocks can be handled in back-end
     -->  
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="40" equivalent="full"/>
-      <output name="O" num_pins="20" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="40" equivalent="full"/>    
+          <output name="O" num_pins="20" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -108,20 +107,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -134,86 +133,86 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+      <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->  
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
           If you need to register the I/O, define clocks in the circuit models
           These clocks can be handled in back-end
       -->   
-      <!-- A mode denotes the physical implementation of an I/O 
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   

-      <!-- IOs can operate as either inputs or outputs.
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -222,87 +221,87 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="40" equivalent="full"/>
-      <output name="O" num_pins="20" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe fracturable logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="40" equivalent="full"/>   
+         <output name="O" num_pins="20" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe fracturable logic element.  
             Each fracturable logic element has a 6-LUT that can alternatively operate as two 5-LUTs with shared inputs. 
             The outputs of the fracturable logic element can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="10">
-        <input name="in" num_pins="6"/>
-        <output name="out" num_pins="2"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- Physical mode definition begin (physical implementation of the fle) -->
-        <mode name="physical" disable_packing="true">
-          <pb_type name="fabric" num_pb="1">
+         <pb_type name="fle" num_pb="10">    
            <input name="in" num_pins="6"/>    
            <output name="out" num_pins="2"/>    
            <clock name="clk" num_pins="1"/>    
-            <pb_type name="frac_logic" num_pb="1">
-              <input name="in" num_pins="6"/>
-              <output name="out" num_pins="2"/>
-              <!-- Define LUT -->
-              <pb_type name="frac_lut6" blif_model=".subckt frac_lut6" num_pb="1">
-                <input name="in" num_pins="6"/>
-                <output name="lut5_out" num_pins="2"/>
-                <output name="lut6_out" num_pins="1"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="frac_logic.in" output="frac_lut6.in"/>
-                <direct name="direct2" input="frac_lut6.lut5_out[1]" output="frac_logic.out[1]"/>
-                <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->
-                <mux name="mux1" input="frac_lut6.lut6_out frac_lut6.lut5_out[0]" output="frac_logic.out[0]"/>
-              </interconnect>
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="fabric.in" output="frac_logic.in"/>
-              <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>
-              <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>
-              <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>
-                <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>
-              </mux>
-              <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>
-                <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="fabric.in"/>
-            <direct name="direct2" input="fabric.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="fabric.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Physical mode definition end (physical implementation of the fle) -->
-        <!-- Dual 5-LUT mode definition begin -->
-        <mode name="n2_lut5">
-          <pb_type name="lut5inter" num_pb="1">
-            <input name="in" num_pins="5"/>
-            <output name="out" num_pins="2"/>
-            <clock name="clk" num_pins="1"/>
-            <pb_type name="ble5" num_pb="2">
-              <input name="in" num_pins="5"/>
-              <output name="out" num_pins="1"/>
-              <clock name="clk" num_pins="1"/>
-              <!-- Define the LUT -->
-              <pb_type name="lut5" blif_model=".names" num_pb="1" class="lut">
-                <input name="in" num_pins="5" port_class="lut_in"/>
-                <output name="out" num_pins="1" port_class="lut_out"/>
-                <!-- LUT timing using delay matrix -->
-                <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+            <!-- Physical mode definition begin (physical implementation of the fle) -->    
+            <mode name="physical" disable_packing="true">     
+               <pb_type name="fabric" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="frac_logic" num_pb="1">       
+                     <input name="in" num_pins="6"/>       
+                     <output name="out" num_pins="2"/>       
+                     <!-- Define LUT -->       
+                     <pb_type name="frac_lut6" blif_model=".subckt frac_lut6" num_pb="1">        
+                        <input name="in" num_pins="6"/>        
+                        <output name="lut5_out" num_pins="2"/>        
+                        <output name="lut6_out" num_pins="1"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="frac_logic.in" output="frac_lut6.in"/>        
+                        <direct name="direct2" input="frac_lut6.lut5_out[1]" output="frac_logic.out[1]"/>        
+                        <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->        
+                        <mux name="mux1" input="frac_lut6.lut6_out frac_lut6.lut5_out[0]" output="frac_logic.out[0]"/>        
+                     </interconnect>       
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="fabric.in" output="frac_logic.in"/>       
+                     <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>       
+                     <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>       
+                     <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>        
+                        <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>        
+                     </mux>       
+                     <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>        
+                        <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="fabric.in"/>      
+                  <direct name="direct2" input="fabric.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="fabric.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Physical mode definition end (physical implementation of the fle) -->    
+            <!-- Dual 5-LUT mode definition begin -->    
+            <mode name="n2_lut5">     
+               <pb_type name="lut5inter" num_pb="1">      
+                  <input name="in" num_pins="5"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="ble5" num_pb="2">       
+                     <input name="in" num_pins="5"/>       
+                     <output name="out" num_pins="1"/>       
+                     <clock name="clk" num_pins="1"/>       
+                     <!-- Define the LUT -->       
+                     <pb_type name="lut5" blif_model=".names" num_pb="1" class="lut">        
+                        <input name="in" num_pins="5" port_class="lut_in"/>        
+                        <output name="out" num_pins="1" port_class="lut_out"/>        
+                        <!-- LUT timing using delay matrix -->        
+                        <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                           we instead take the average of these numbers to get more stable results
                      82e-12
                      173e-12
@ -310,63 +309,63 @@
                      263e-12
                      398e-12
                      -->        
-                <delay_matrix type="max" in_port="lut5.in" out_port="lut5.out">
+                        <delay_matrix type="max" in_port="lut5.in" out_port="lut5.out">
                  235e-12
                  235e-12
                  235e-12
                  235e-12
                  235e-12
                </delay_matrix>        
-              </pb_type>
-              <!-- Define the flip-flop -->
-              <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-                <input name="D" num_pins="1" port_class="D"/>
-                <output name="Q" num_pins="1" port_class="Q"/>
-                <clock name="clk" num_pins="1" port_class="clock"/>
-                <T_setup value="66e-12" port="ff.D" clock="clk"/>
-                <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="ble5.in[4:0]" output="lut5[0:0].in[4:0]"/>
-                <direct name="direct2" input="lut5[0:0].out" output="ff[0:0].D">
-                  <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                  <pack_pattern name="ble5" in_port="lut5[0:0].out" out_port="ff[0:0].D"/>
-                </direct>
-                <direct name="direct3" input="ble5.clk" output="ff[0:0].clk"/>
-                <mux name="mux1" input="ff[0:0].Q lut5.out[0:0]" output="ble5.out[0:0]">
-                  <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                  <delay_constant max="25e-12" in_port="lut5.out[0:0]" out_port="ble5.out[0:0]"/>
-                  <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble5.out[0:0]"/>
-                </mux>
-              </interconnect>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="lut5inter.in" output="ble5[0:0].in"/>
-              <direct name="direct2" input="lut5inter.in" output="ble5[1:1].in"/>
-              <direct name="direct3" input="ble5[1:0].out" output="lut5inter.out"/>
-              <complete name="complete1" input="lut5inter.clk" output="ble5[1:0].clk"/>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[4:0]" output="lut5inter.in"/>
-            <direct name="direct2" input="lut5inter.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="lut5inter.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Dual 5-LUT mode definition end -->
-        <!-- 6-LUT mode definition begin -->
-        <mode name="n1_lut6">
-          <!-- Define 6-LUT mode -->
-          <pb_type name="ble6" num_pb="1">
-            <input name="in" num_pins="6"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Define LUT -->
-            <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="6" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+                     </pb_type>       
+                     <!-- Define the flip-flop -->       
+                     <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">        
+                        <input name="D" num_pins="1" port_class="D"/>        
+                        <output name="Q" num_pins="1" port_class="Q"/>        
+                        <clock name="clk" num_pins="1" port_class="clock"/>        
+                        <T_setup value="66e-12" port="ff.D" clock="clk"/>        
+                        <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="ble5.in[4:0]" output="lut5[0:0].in[4:0]"/>        
+                        <direct name="direct2" input="lut5[0:0].out" output="ff[0:0].D">         
+                           <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->         
+                           <pack_pattern name="ble5" in_port="lut5[0:0].out" out_port="ff[0:0].D"/>         
+                        </direct>        
+                        <direct name="direct3" input="ble5.clk" output="ff[0:0].clk"/>        
+                        <mux name="mux1" input="ff[0:0].Q lut5.out[0:0]" output="ble5.out[0:0]">         
+                           <!-- LUT to output is faster than FF to output on a Stratix IV -->         
+                           <delay_constant max="25e-12" in_port="lut5.out[0:0]" out_port="ble5.out[0:0]"/>         
+                           <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble5.out[0:0]"/>         
+                        </mux>        
+                     </interconnect>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="lut5inter.in" output="ble5[0:0].in"/>       
+                     <direct name="direct2" input="lut5inter.in" output="ble5[1:1].in"/>       
+                     <direct name="direct3" input="ble5[1:0].out" output="lut5inter.out"/>       
+                     <complete name="complete1" input="lut5inter.clk" output="ble5[1:0].clk"/>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[4:0]" output="lut5inter.in"/>      
+                  <direct name="direct2" input="lut5inter.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="lut5inter.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Dual 5-LUT mode definition end -->    
+            <!-- 6-LUT mode definition begin -->    
+            <mode name="n1_lut6">     
+               <!-- Define 6-LUT mode -->     
+               <pb_type name="ble6" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="6" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
@ -375,7 +374,7 @@
                  398e-12
                  397e-12
                  -->       
-              <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
+                     <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
                261e-12
                261e-12
                261e-12
@ -383,39 +382,39 @@
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>
-              <direct name="direct2" input="lut6.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble6.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble6.in"/>
-            <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble6.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>       
+                     <direct name="direct2" input="lut6.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble6.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble6.in"/>      
+                  <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble6.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -424,23 +423,23 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>
+            <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[9:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[9:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[9:0].out[0:0]" output="clb.O[9:0]"/>
-        <direct name="clbouts2" input="fle[9:0].out[1:1]" output="clb.O[19:10]"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[9:0].out[0:0]" output="clb.O[9:0]"/>    
+            <direct name="clbouts2" input="fle[9:0].out[1:1]" output="clb.O[19:10]"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Flagship Heterogeneous Architecture with Carry Chains for VTR 7.0.

  - 40 nm technology
@ -75,9 +75,8 @@


  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -86,97 +85,97 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <model name="adder">
-      <input_ports>
-        <port name="a" combinational_sink_ports="sumout cout"/>
-        <port name="b" combinational_sink_ports="sumout cout"/>
-        <port name="cin" combinational_sink_ports="sumout cout"/>
-      </input_ports>
-      <output_ports>
-        <port name="cout"/>
-        <port name="sumout"/>
-      </output_ports>
-    </model>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="frac_lut6">
-      <input_ports>
-        <port name="in"/>
-      </input_ports>
-      <output_ports>
-        <port name="lut4_out"/>
-        <port name="lut5_out"/>
-        <port name="lut6_out"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="40" equivalent="full"/>
-      <input name="cin" num_pins="1"/>
-      <output name="O" num_pins="20" equivalent="none"/>
-      <output name="cout" num_pins="1"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">
-        <fc_override port_name="cin" fc_type="frac" fc_val="0"/>
-        <fc_override port_name="cout" fc_type="frac" fc_val="0"/>
-      </fc>
-      <!--  Highly recommand to customize pin location when direct connection is used!!! -->
-      <!--pinlocations pattern="spread"/-->
-      <pinlocations pattern="custom">
-        <loc side="left">clb.clk</loc>
-        <loc side="top">clb.cin</loc>
-        <loc side="right">clb.O[9:0] clb.I[19:0]</loc>
-        <loc side="bottom">clb.cout clb.O[19:10] clb.I[39:20]</loc>
-      </pinlocations>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+   <models>  
+      <model name="adder">   
+         <input_ports>    
+            <port name="a" combinational_sink_ports="sumout cout"/>    
+            <port name="b" combinational_sink_ports="sumout cout"/>    
+            <port name="cin" combinational_sink_ports="sumout cout"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="cout"/>    
+            <port name="sumout"/>    
+         </output_ports>   
+      </model>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="frac_lut6">   
+         <input_ports>    
+            <port name="in"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="lut4_out"/>    
+            <port name="lut5_out"/>    
+            <port name="lut6_out"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="40" equivalent="full"/>    
+          <input name="cin" num_pins="1"/>    
+          <output name="O" num_pins="20" equivalent="none"/>    
+          <output name="cout" num_pins="1"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10">     
+             <fc_override port_name="cin" fc_type="frac" fc_val="0"/>     
+             <fc_override port_name="cout" fc_type="frac" fc_val="0"/>     
+          </fc>    
+          <!--  Highly recommand to customize pin location when direct connection is used!!! -->    
+          <!--pinlocations pattern="spread"/-->    
+          <pinlocations pattern="custom">     
+             <loc side="left">clb.clk</loc>     
+             <loc side="top">clb.cin</loc>     
+             <loc side="right">clb.O[9:0] clb.I[19:0]</loc>     
+             <loc side="bottom">clb.cout clb.O[19:10] clb.I[39:20]</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -191,20 +190,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	    -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
           book area formula. This means the mux transistors are about 5x minimum drive strength.
           We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
           mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -217,90 +216,90 @@
           2.5x when looking up in Jeff's tables.
           Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
           This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
             With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
             reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <directlist>
-    <direct name="adder_carry" from_pin="clb.cout" to_pin="clb.cin" x_offset="0" y_offset="-1" z_offset="0"/>
-  </directlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
+      <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->  
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <directlist>  
+      <direct name="adder_carry" from_pin="clb.cout" to_pin="clb.cin" x_offset="0" y_offset="-1" z_offset="0"/>  
+   </directlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   

-      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+         <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
           If you need to register the I/O, define clocks in the circuit models
           These clocks can be handled in back-end
       -->   
-      <!-- A mode denotes the physical implementation of an I/O 
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   

-      <!-- IOs can operate as either inputs or outputs.
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -309,115 +308,115 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="40" equivalent="full"/>
-      <input name="cin" num_pins="1"/>
-      <output name="O" num_pins="20" equivalent="none"/>
-      <output name="cout" num_pins="1"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe fracturable logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="40" equivalent="full"/>   
+         <input name="cin" num_pins="1"/>   
+         <output name="O" num_pins="20" equivalent="none"/>   
+         <output name="cout" num_pins="1"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe fracturable logic element.  
             Each fracturable logic element has a 6-LUT that can alternatively operate as two 5-LUTs with shared inputs. 
             The outputs of the fracturable logic element can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="10">
-        <input name="in" num_pins="6"/>
-        <input name="cin" num_pins="1"/>
-        <output name="out" num_pins="2"/>
-        <output name="cout" num_pins="1"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- Physical mode definition begin (physical implementation of the fle) -->
-        <mode name="physical" disable_packing="true">
-          <pb_type name="fabric" num_pb="1">
+         <pb_type name="fle" num_pb="10">    
            <input name="in" num_pins="6"/>    
            <input name="cin" num_pins="1"/>    
            <output name="out" num_pins="2"/>    
            <output name="cout" num_pins="1"/>    
            <clock name="clk" num_pins="1"/>    
-            <pb_type name="frac_logic" num_pb="1">
-              <input name="in" num_pins="6"/>
-              <output name="lut4_out" num_pins="4"/>
-              <output name="out" num_pins="2"/>
-              <!-- Define LUT -->
-              <pb_type name="frac_lut6" blif_model=".subckt frac_lut6" num_pb="1">
-                <input name="in" num_pins="6"/>
-                <output name="lut4_out" num_pins="4"/>
-                <output name="lut5_out" num_pins="2"/>
-                <output name="lut6_out" num_pins="1"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="frac_logic.in" output="frac_lut6.in"/>
-                <direct name="direct2" input="frac_lut6.lut4_out" output="frac_logic.lut4_out"/>
-                <direct name="direct3" input="frac_lut6.lut5_out[1]" output="frac_logic.out[1]"/>
-                <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->
-                <mux name="mux1" input="frac_lut6.lut6_out frac_lut6.lut5_out[0]" output="frac_logic.out[0]"/>
-              </interconnect>
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>         
-            <!-- Define adders -->
-            <pb_type name="adder" blif_model=".subckt adder" num_pb="2">
-              <input name="a" num_pins="1"/>
-              <input name="b" num_pins="1"/>
-              <input name="cin" num_pins="1"/>
-              <output name="cout" num_pins="1"/>
-              <output name="sumout" num_pins="1"/>
-              <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.cin" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.cout"/>
-              <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.cout"/>
-              <delay_constant max="0.01e-9" in_port="adder.cin" out_port="adder.cout"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="fabric.in" output="frac_logic.in"/>
-              <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>
-              <direct name="direct3" input="fabric.cin" output="adder[0:0].cin"/>
-              <direct name="direct4" input="adder[0:0].cout" output="adder[1:1].cin"/>
-              <direct name="direct5" input="adder[1:1].cout" output="fabric.cout"/>
-              <direct name="direct6" input="frac_logic.lut4_out[0:0]" output="adder[0:0].a"/>
-              <direct name="direct7" input="frac_logic.lut4_out[1:1]" output="adder[0:0].b"/>
-              <direct name="direct8" input="frac_logic.lut4_out[2:2]" output="adder[1:1].a"/>
-              <direct name="direct9" input="frac_logic.lut4_out[3:3]" output="adder[1:1].b"/>
-              <complete name="direct10" input="fabric.clk" output="ff[1:0].clk"/>
-              <mux name="mux1" input="adder[0].sumout ff[0].Q frac_logic.out[0]" output="fabric.out[0]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>
-                <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>
-              </mux>
-              <mux name="mux2" input="adder[1].sumout ff[1].Q frac_logic.out[1]" output="fabric.out[1]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>
-                <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="fabric.in"/>
-            <direct name="direct2" input="fle.cin" output="fabric.cin"/>
-            <direct name="direct3" input="fabric.out" output="fle.out"/>
-            <direct name="direct4" input="fabric.cout" output="fle.cout"/>
-            <direct name="direct5" input="fle.clk" output="fabric.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Physical mode definition end (physical implementation of the fle) -->
-        <!-- BEGIN fle mode of dual lut5 -->
-        <mode name="n2_lut5">
-          <pb_type name="ble5" num_pb="2">
-            <input name="in" num_pins="5"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Regular LUT mode -->
-            <pb_type name="lut5" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="5" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+            <!-- Physical mode definition begin (physical implementation of the fle) -->    
+            <mode name="physical" disable_packing="true">     
+               <pb_type name="fabric" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <input name="cin" num_pins="1"/>      
+                  <output name="out" num_pins="2"/>      
+                  <output name="cout" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="frac_logic" num_pb="1">       
+                     <input name="in" num_pins="6"/>       
+                     <output name="lut4_out" num_pins="4"/>       
+                     <output name="out" num_pins="2"/>       
+                     <!-- Define LUT -->       
+                     <pb_type name="frac_lut6" blif_model=".subckt frac_lut6" num_pb="1">        
+                        <input name="in" num_pins="6"/>        
+                        <output name="lut4_out" num_pins="4"/>        
+                        <output name="lut5_out" num_pins="2"/>        
+                        <output name="lut6_out" num_pins="1"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="frac_logic.in" output="frac_lut6.in"/>        
+                        <direct name="direct2" input="frac_lut6.lut4_out" output="frac_logic.lut4_out"/>        
+                        <direct name="direct3" input="frac_lut6.lut5_out[1]" output="frac_logic.out[1]"/>        
+                        <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->        
+                        <mux name="mux1" input="frac_lut6.lut6_out frac_lut6.lut5_out[0]" output="frac_logic.out[0]"/>        
+                     </interconnect>       
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>               
+                  <!-- Define adders -->      
+                  <pb_type name="adder" blif_model=".subckt adder" num_pb="2">       
+                     <input name="a" num_pins="1"/>       
+                     <input name="b" num_pins="1"/>       
+                     <input name="cin" num_pins="1"/>       
+                     <output name="cout" num_pins="1"/>       
+                     <output name="sumout" num_pins="1"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.cin" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.cout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.cout"/>       
+                     <delay_constant max="0.01e-9" in_port="adder.cin" out_port="adder.cout"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="fabric.in" output="frac_logic.in"/>       
+                     <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>       
+                     <direct name="direct3" input="fabric.cin" output="adder[0:0].cin"/>       
+                     <direct name="direct4" input="adder[0:0].cout" output="adder[1:1].cin"/>       
+                     <direct name="direct5" input="adder[1:1].cout" output="fabric.cout"/>       
+                     <direct name="direct6" input="frac_logic.lut4_out[0:0]" output="adder[0:0].a"/>       
+                     <direct name="direct7" input="frac_logic.lut4_out[1:1]" output="adder[0:0].b"/>       
+                     <direct name="direct8" input="frac_logic.lut4_out[2:2]" output="adder[1:1].a"/>       
+                     <direct name="direct9" input="frac_logic.lut4_out[3:3]" output="adder[1:1].b"/>       
+                     <complete name="direct10" input="fabric.clk" output="ff[1:0].clk"/>       
+                     <mux name="mux1" input="adder[0].sumout ff[0].Q frac_logic.out[0]" output="fabric.out[0]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>        
+                        <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>        
+                     </mux>       
+                     <mux name="mux2" input="adder[1].sumout ff[1].Q frac_logic.out[1]" output="fabric.out[1]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>        
+                        <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="fabric.in"/>      
+                  <direct name="direct2" input="fle.cin" output="fabric.cin"/>      
+                  <direct name="direct3" input="fabric.out" output="fle.out"/>      
+                  <direct name="direct4" input="fabric.cout" output="fle.cout"/>      
+                  <direct name="direct5" input="fle.clk" output="fabric.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Physical mode definition end (physical implementation of the fle) -->    
+            <!-- BEGIN fle mode of dual lut5 -->    
+            <mode name="n2_lut5">     
+               <pb_type name="ble5" num_pb="2">      
+                  <input name="in" num_pins="5"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Regular LUT mode -->      
+                  <pb_type name="lut5" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="5" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                  we instead take the average of these numbers to get more stable results
                82e-12
                173e-12
@ -425,138 +424,138 @@
                263e-12
                398e-12
                -->       
-              <delay_matrix type="max" in_port="lut5.in" out_port="lut5.out">
+                     <delay_matrix type="max" in_port="lut5.in" out_port="lut5.out">
                235e-12
                235e-12
                235e-12
                235e-12
                235e-12
              </delay_matrix>       
-            </pb_type>
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble5.in" output="lut5.in"/>
-              <direct name="direct2" input="lut5.out" output="ff.D">
-                <pack_pattern name="ble5" in_port="lut5.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble5.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut5.out" output="ble5.out">
-                <delay_constant max="25e-12" in_port="lut5.out" out_port="ble5.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble5.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[4:0]" output="ble5[0:0].in"/>
-            <direct name="direct2" input="fle.in[4:0]" output="ble5[1:1].in"/>
-            <complete name="direct3" input="fle.clk" output="ble5.clk"/>
-            <direct name="direct4" input="ble5.out" output="fle.out"/>
-          </interconnect>
-        </mode>
-        <!-- END fle mode of dual lut5 -->
-        <!-- BEGIN arithmetic mode of dual lut4 + adders -->
-        <mode name="arithmetic">
-          <pb_type name="arithmetic" num_pb="2">
-            <input name="in" num_pins="4"/>
-            <input name="cin" num_pins="1"/>
-            <output name="out" num_pins="1"/>
-            <output name="cout" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Special dual-LUT mode that drives adder only -->
-            <pb_type name="lut4" blif_model=".names" num_pb="2" class="lut">
-              <input name="in" num_pins="4" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+                  </pb_type>      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble5.in" output="lut5.in"/>       
+                     <direct name="direct2" input="lut5.out" output="ff.D">        
+                        <pack_pattern name="ble5" in_port="lut5.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble5.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut5.out" output="ble5.out">        
+                        <delay_constant max="25e-12" in_port="lut5.out" out_port="ble5.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble5.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[4:0]" output="ble5[0:0].in"/>      
+                  <direct name="direct2" input="fle.in[4:0]" output="ble5[1:1].in"/>      
+                  <complete name="direct3" input="fle.clk" output="ble5.clk"/>      
+                  <direct name="direct4" input="ble5.out" output="fle.out"/>      
+               </interconnect>     
+            </mode>    
+            <!-- END fle mode of dual lut5 -->    
+            <!-- BEGIN arithmetic mode of dual lut4 + adders -->    
+            <mode name="arithmetic">     
+               <pb_type name="arithmetic" num_pb="2">      
+                  <input name="in" num_pins="4"/>      
+                  <input name="cin" num_pins="1"/>      
+                  <output name="out" num_pins="1"/>      
+                  <output name="cout" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Special dual-LUT mode that drives adder only -->      
+                  <pb_type name="lut4" blif_model=".names" num_pb="2" class="lut">       
+                     <input name="in" num_pins="4" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
                  261e-12
                  263e-12
                  -->       
-              <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
+                     <delay_matrix type="max" in_port="lut4.in" out_port="lut4.out">
                  195e-12
                  195e-12
                  195e-12
                  195e-12
                </delay_matrix>       
-            </pb_type>
-            <pb_type name="adder" blif_model=".subckt adder" num_pb="1">
-              <input name="a" num_pins="1"/>
-              <input name="b" num_pins="1"/>
-              <input name="cin" num_pins="1"/>
-              <output name="cout" num_pins="1"/>
-              <output name="sumout" num_pins="1"/>
-              <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.cin" out_port="adder.sumout"/>
-              <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.cout"/>
-              <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.cout"/>
-              <delay_constant max="0.01e-9" in_port="adder.cin" out_port="adder.cout"/>
-            </pb_type>
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="clock" input="arithmetic.clk" output="ff.clk"/>
-              <direct name="lut_in1" input="arithmetic.in[3:0]" output="lut4[0:0].in[3:0]"/>
-              <direct name="lut_in2" input="arithmetic.in[3:0]" output="lut4[1:1].in[3:0]"/>
-              <direct name="lut_to_add1" input="lut4[0:0].out" output="adder.a">
+                  </pb_type>      
+                  <pb_type name="adder" blif_model=".subckt adder" num_pb="1">       
+                     <input name="a" num_pins="1"/>       
+                     <input name="b" num_pins="1"/>       
+                     <input name="cin" num_pins="1"/>       
+                     <output name="cout" num_pins="1"/>       
+                     <output name="sumout" num_pins="1"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.cin" out_port="adder.sumout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.a" out_port="adder.cout"/>       
+                     <delay_constant max="0.3e-9" in_port="adder.b" out_port="adder.cout"/>       
+                     <delay_constant max="0.01e-9" in_port="adder.cin" out_port="adder.cout"/>       
+                  </pb_type>      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="clock" input="arithmetic.clk" output="ff.clk"/>       
+                     <direct name="lut_in1" input="arithmetic.in[3:0]" output="lut4[0:0].in[3:0]"/>       
+                     <direct name="lut_in2" input="arithmetic.in[3:0]" output="lut4[1:1].in[3:0]"/>       
+                     <direct name="lut_to_add1" input="lut4[0:0].out" output="adder.a">
                </direct>       
-              <direct name="lut_to_add2" input="lut4[1:1].out" output="adder.b">
+                     <direct name="lut_to_add2" input="lut4[1:1].out" output="adder.b">
                </direct>       
-              <direct name="add_to_ff" input="adder.sumout" output="ff.D">
-                <pack_pattern name="chain" in_port="adder.sumout" out_port="ff.D"/>
-              </direct>
-              <direct name="carry_in" input="arithmetic.cin" output="adder.cin">
-                <pack_pattern name="chain" in_port="arithmetic.cin" out_port="adder.cin"/>
-              </direct>
-              <direct name="carry_out" input="adder.cout" output="arithmetic.cout">
-                <pack_pattern name="chain" in_port="adder.cout" out_port="arithmetic.cout"/>
-              </direct>
-              <mux name="sumout" input="ff.Q adder.sumout" output="arithmetic.out">
-                <delay_constant max="25e-12" in_port="adder.sumout" out_port="arithmetic.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="arithmetic.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[3:0]" output="arithmetic[0:0].in"/>
-            <direct name="direct2" input="fle.in[3:0]" output="arithmetic[1:1].in"/>
-            <direct name="carry_in" input="fle.cin" output="arithmetic[0:0].cin">
-              <pack_pattern name="chain" in_port="fle.cin" out_port="arithmetic[0:0].cin"/>
-            </direct>
-            <direct name="carry_inter" input="arithmetic[0:0].cout" output="arithmetic[1:1].cin">
-              <pack_pattern name="chain" in_port="arithmetic[0:0].cout" out_port="arithmetic[1:1].cin"/>
-            </direct>
-            <direct name="carry_out" input="arithmetic[1:1].cout" output="fle.cout">
-              <pack_pattern name="chain" in_port="arithmetic.cout" out_port="fle.cout"/>
-            </direct>
-            <complete name="direct3" input="fle.clk" output="arithmetic.clk"/>
-            <direct name="direct4" input="arithmetic.out" output="fle.out"/>
-          </interconnect>
-        </mode>
-        <!-- n2_lut5 -->
-        <mode name="n1_lut6">
-          <pb_type name="ble6" num_pb="1">
-            <input name="in" num_pins="6"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="6" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+                     <direct name="add_to_ff" input="adder.sumout" output="ff.D">        
+                        <pack_pattern name="chain" in_port="adder.sumout" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="carry_in" input="arithmetic.cin" output="adder.cin">        
+                        <pack_pattern name="chain" in_port="arithmetic.cin" out_port="adder.cin"/>        
+                     </direct>       
+                     <direct name="carry_out" input="adder.cout" output="arithmetic.cout">        
+                        <pack_pattern name="chain" in_port="adder.cout" out_port="arithmetic.cout"/>        
+                     </direct>       
+                     <mux name="sumout" input="ff.Q adder.sumout" output="arithmetic.out">        
+                        <delay_constant max="25e-12" in_port="adder.sumout" out_port="arithmetic.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="arithmetic.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[3:0]" output="arithmetic[0:0].in"/>      
+                  <direct name="direct2" input="fle.in[3:0]" output="arithmetic[1:1].in"/>      
+                  <direct name="carry_in" input="fle.cin" output="arithmetic[0:0].cin">       
+                     <pack_pattern name="chain" in_port="fle.cin" out_port="arithmetic[0:0].cin"/>       
+                  </direct>      
+                  <direct name="carry_inter" input="arithmetic[0:0].cout" output="arithmetic[1:1].cin">       
+                     <pack_pattern name="chain" in_port="arithmetic[0:0].cout" out_port="arithmetic[1:1].cin"/>       
+                  </direct>      
+                  <direct name="carry_out" input="arithmetic[1:1].cout" output="fle.cout">       
+                     <pack_pattern name="chain" in_port="arithmetic.cout" out_port="fle.cout"/>       
+                  </direct>      
+                  <complete name="direct3" input="fle.clk" output="arithmetic.clk"/>      
+                  <direct name="direct4" input="arithmetic.out" output="fle.out"/>      
+               </interconnect>     
+            </mode>    
+            <!-- n2_lut5 -->    
+            <mode name="n1_lut6">     
+               <pb_type name="ble6" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="6" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
@ -565,7 +564,7 @@
                  398e-12
                  397e-12
                  -->       
-              <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
+                     <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
                  261e-12
                  261e-12
                  261e-12
@ -573,36 +572,36 @@
                  261e-12
                  261e-12
                </delay_matrix>       
-            </pb_type>
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>
-              <direct name="direct2" input="lut6.out" output="ff.D">
-                <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble6.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">
-                <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[5:0]" output="ble6.in"/>
-            <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble6.clk"/>
-          </interconnect>
-        </mode>
-        <!-- n1_lut6 -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a 50% depop crossbar built using small full xbars to get sets of logically equivalent pins at inputs of CLB 
+                  </pb_type>      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>       
+                     <direct name="direct2" input="lut6.out" output="ff.D">        
+                        <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble6.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">        
+                        <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[5:0]" output="ble6.in"/>      
+                  <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble6.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- n1_lut6 -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a 50% depop crossbar built using small full xbars to get sets of logically equivalent pins at inputs of CLB 
           The delays below come from Stratix IV. the delay through a connection block
           input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
           delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -611,34 +610,34 @@
           Since all our outputs LUT outputs go to a BLE output, and have a delay of 
           25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
           to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>
-        </complete>
+            <complete name="crossbar" input="clb.I fle[9:0].out" output="fle[9:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[9:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[9:0].out" out_port="fle[9:0].in"/>     
+            </complete>    

-        <complete name="clks" input="clb.clk" output="fle[9:0].clk">
+            <complete name="clks" input="clb.clk" output="fle[9:0].clk">
        </complete>    
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
                 By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
                 then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
                 naive specification).
          -->    
-        <direct name="clbouts1" input="fle[9:0].out[0:0]" output="clb.O[9:0]"/>
-        <direct name="clbouts2" input="fle[9:0].out[1:1]" output="clb.O[19:10]"/>
-        <!-- Carry chain links -->
-        <direct name="carry_in" input="clb.cin" output="fle[0:0].cin">
-          <!-- Put all inter-block carry chain delay on this one edge -->
-          <delay_constant max="0.16e-9" in_port="clb.cin" out_port="fle[0:0].cin"/>
-          <pack_pattern name="chain" in_port="clb.cin" out_port="fle[0:0].cin"/>
-        </direct>
-        <direct name="carry_out" input="fle[9:9].cout" output="clb.cout">
-          <pack_pattern name="chain" in_port="fle[9:9].cout" out_port="clb.cout"/>
-        </direct>
-        <direct name="carry_link" input="fle[8:0].cout" output="fle[9:1].cin">
-          <pack_pattern name="chain" in_port="fle[8:0].cout" out_port="fle[9:1].cin"/>
-        </direct>
-      </interconnect>
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[9:0].out[0:0]" output="clb.O[9:0]"/>    
+            <direct name="clbouts2" input="fle[9:0].out[1:1]" output="clb.O[19:10]"/>    
+            <!-- Carry chain links -->    
+            <direct name="carry_in" input="clb.cin" output="fle[0:0].cin">     
+               <!-- Put all inter-block carry chain delay on this one edge -->     
+               <delay_constant max="0.16e-9" in_port="clb.cin" out_port="fle[0:0].cin"/>     
+               <pack_pattern name="chain" in_port="clb.cin" out_port="fle[0:0].cin"/>     
+            </direct>    
+            <direct name="carry_out" input="fle[9:9].cout" output="clb.cout">     
+               <pack_pattern name="chain" in_port="fle[9:9].cout" out_port="clb.cout"/>     
+            </direct>    
+            <direct name="carry_link" input="fle[8:0].cout" output="fle[9:1].cin">     
+               <pack_pattern name="chain" in_port="fle[8:0].cout" out_port="fle[9:1].cin"/>     
+            </direct>    
+         </interconnect>   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_dpram8K_dsp36_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_dpram8K_dsp36_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_dpram8K_dsp36_fracff_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_dpram8K_dsp36_fracff_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_frac_mem32K_frac_dsp36_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_frac_mem32K_frac_dsp36_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_frac_mem32K_frac_dsp36_GlobalTile8Clk_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_frac_mem32K_frac_dsp36_GlobalTile8Clk_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_mem16K_aib_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_mem16K_aib_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_mem16K_multi_io_capacity_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_mem16K_multi_io_capacity_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_mem16K_reduced_io_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_mem16K_reduced_io_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_mem1K_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_mem1K_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_wide_mem1K_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_chain_wide_mem1K_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_register_chain_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_register_chain_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_register_scan_chain_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_register_scan_chain_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_register_scan_chain_depop50_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_register_scan_chain_depop50_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_register_scan_chain_depop50_spypad_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_register_scan_chain_depop50_spypad_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_register_scan_chain_mem16K_depop50_12nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_adder_register_scan_chain_mem16K_depop50_12nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_thru_channel_adder_chain_mem16K_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_thru_channel_adder_chain_mem16K_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N10_tileable_thru_channel_adder_chain_wide_mem16K_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N10_tileable_thru_channel_adder_chain_wide_mem16K_40nm.xml
--- a/openfpga_flow/vpr_arch/k6_frac_N8_tileable_40nm.xml
+++ b/openfpga_flow/vpr_arch/k6_frac_N8_tileable_40nm.xml
@ -1,4 +1,4 @@
-<!-- 
+<?xml version="1.0" ?><!-- 
  Flagship Heterogeneous Architecture (No Carry Chains) for VTR 7.0.

  - 40 nm technology
@ -12,9 +12,8 @@
  Based on flagship k6_frac_N10_mem32K_40nm.xml architecture.

  Authors: Jason Luu, Jeff Goeders, Vaughn Betz
-->
-<architecture>
-  <!-- 
+--><architecture> 
+   <!-- 
       ODIN II specific config begins 
       Describes the types of user-specified netlist blocks (in blif, this corresponds to 
       ".model [type_of_block]") that this architecture supports.
@ -23,77 +22,77 @@
       already special structures in blif (.names, .input, .output, and .latch) 
       that describe them.
  --> 
-  <models>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="io">
-      <input_ports>
-        <port name="outpad"/>
-      </input_ports>
-      <output_ports>
-        <port name="inpad"/>
-      </output_ports>
-    </model>
-    <!-- A virtual model for I/O to be used in the physical mode of io block -->
-    <model name="frac_lut6">
-      <input_ports>
-        <port name="in"/>
-      </input_ports>
-      <output_ports>
-        <port name="lut5_out"/>
-        <port name="lut6_out"/>
-      </output_ports>
-    </model>
-  </models>
-  <tiles>
-    <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+   <models>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="io">   
+         <input_ports>    
+            <port name="outpad"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="inpad"/>    
+         </output_ports>   
+      </model>  
+      <!-- A virtual model for I/O to be used in the physical mode of io block -->  
+      <model name="frac_lut6">   
+         <input_ports>    
+            <port name="in"/>    
+         </input_ports>   
+         <output_ports>    
+            <port name="lut5_out"/>    
+            <port name="lut6_out"/>    
+         </output_ports>   
+      </model>  
+   </models> 
+   <tiles>  
+      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
         If you need to register the I/O, define clocks in the circuit models
         These clocks can be handled in back-end
     -->  
-    <tile name="io" capacity="8" area="0">
-      <equivalent_sites>
-        <site pb_type="io"/>
-      </equivalent_sites>
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="custom">
-        <loc side="left">io.outpad io.inpad</loc>
-        <loc side="top">io.outpad io.inpad</loc>
-        <loc side="right">io.outpad io.inpad</loc>
-        <loc side="bottom">io.outpad io.inpad</loc>
-      </pinlocations>
-    </tile>
-    <tile name="clb" area="53894">
-      <equivalent_sites>
-        <site pb_type="clb"/>
-      </equivalent_sites>
-      <input name="I" num_pins="32" equivalent="full"/>
-      <output name="O" num_pins="16" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>
-      <pinlocations pattern="spread"/>
-    </tile>
-  </tiles>
-  <!-- ODIN II specific config ends -->
-  <!-- Physical descriptions begin -->
-  <layout tileable="true">
-    <auto_layout aspect_ratio="1.0">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </auto_layout>
-    <fixed_layout name="2x2" width="4" height="4">
-      <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->
-      <perimeter type="io" priority="100"/>
-      <corners type="EMPTY" priority="101"/>
-      <!--Fill with 'clb'-->
-      <fill type="clb" priority="10"/>
-    </fixed_layout>
-  </layout>
-  <device>
-    <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
+      <tile name="io" area="0">   <sub_tile name="io" capacity="8">    
+          <equivalent_sites>     
+             <site pb_type="io"/>     
+          </equivalent_sites>    
+          <input name="outpad" num_pins="1"/>    
+          <output name="inpad" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="custom">     
+             <loc side="left">io.outpad io.inpad</loc>     
+             <loc side="top">io.outpad io.inpad</loc>     
+             <loc side="right">io.outpad io.inpad</loc>     
+             <loc side="bottom">io.outpad io.inpad</loc>     
+          </pinlocations>    
+       </sub_tile>  </tile>  
+      <tile name="clb" area="53894">   <sub_tile name="clb">    
+          <equivalent_sites>     
+             <site pb_type="clb"/>     
+          </equivalent_sites>    
+          <input name="I" num_pins="32" equivalent="full"/>    
+          <output name="O" num_pins="16" equivalent="none"/>    
+          <clock name="clk" num_pins="1"/>    
+          <fc in_type="frac" in_val="0.15" out_type="frac" out_val="0.10"/>    
+          <pinlocations pattern="spread"/>    
+       </sub_tile>  </tile>  
+   </tiles> 
+   <!-- ODIN II specific config ends --> 
+   <!-- Physical descriptions begin --> 
+   <layout tileable="true">  
+      <auto_layout aspect_ratio="1.0">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </auto_layout>  
+      <fixed_layout name="2x2" width="4" height="4">   
+         <!--Perimeter of 'io' blocks with 'EMPTY' blocks at corners-->   
+         <perimeter type="io" priority="100"/>   
+         <corners type="EMPTY" priority="101"/>   
+         <!--Fill with 'clb'-->   
+         <fill type="clb" priority="10"/>   
+      </fixed_layout>  
+   </layout> 
+   <device>  
+      <!-- VB & JL: Using Ian Kuon's transistor sizing and drive strength data for routing, at 40 nm. Ian used BPTM 
 			     models. We are modifying the delay values however, to include metal C and R, which allows more architecture
 			     experimentation. We are also modifying the relative resistance of PMOS to be 1.8x that of NMOS
 			     (vs. Ian's 3x) as 1.8x lines up with Jeff G's data from a 45 nm process (and is more typical of 
@ -108,20 +107,20 @@
 			     proposed FPGA, and which is also 40 nm 
 			     C_ipin_cblock: input capacitance of a track buffer, which VPR assumes is a single-stage
 			     4x minimum drive strength buffer. -->  
-    <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>
-    <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
+      <sizing R_minW_nmos="8926" R_minW_pmos="16067"/>  
+      <!-- The grid_logic_tile_area below will be used for all blocks that do not explicitly set their own (non-routing)
     	  area; set to 0 since we explicitly set the area of all blocks currently in this architecture file.
 	  -->  
-    <area grid_logic_tile_area="0"/>
-    <chan_width_distr>
-      <x distr="uniform" peak="1.000000"/>
-      <y distr="uniform" peak="1.000000"/>
-    </chan_width_distr>
-    <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>
-    <connection_block input_switch_name="ipin_cblock"/>
-  </device>
-  <switchlist>
-    <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
+      <area grid_logic_tile_area="0"/>  
+      <chan_width_distr>   
+         <x distr="uniform" peak="1.000000"/>   
+         <y distr="uniform" peak="1.000000"/>   
+      </chan_width_distr>  
+      <switch_block type="wilton" fs="3" sub_type="subset" sub_fs="3"/>  
+      <connection_block input_switch_name="ipin_cblock"/>  
+   </device> 
+   <switchlist>  
+      <!-- VB: the mux_trans_size and buf_size data below is in minimum width transistor *areas*, assuming the purple
 	       book area formula. This means the mux transistors are about 5x minimum drive strength.
 	       We assume the first stage of the buffer is 3x min drive strength to be reasonable given the large 
 	       mux transistors, and this gives a reasonable stage ratio of a bit over 5x to the second stage. We assume
@ -134,86 +133,86 @@
 	       2.5x when looking up in Jeff's tables.
 	       Finally, we choose a switch delay (58 ps) that leads to length 4 wires having a delay equal to that of SIV of 126 ps.
 	       This also leads to the switch being 46% of the total wire delay, which is reasonable. -->  
-    <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>
-    <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->
-    <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>
-  </switchlist>
-  <segmentlist>
-    <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
+      <switch type="mux" name="0" R="551" Cin=".77e-15" Cout="4e-15" Tdel="58e-12" mux_trans_size="2.630740" buf_size="27.645901"/>  
+      <!--switch ipin_cblock resistance set to yeild for 4x minimum drive strength buffer-->  
+      <switch type="mux" name="ipin_cblock" R="2231.5" Cout="0." Cin="1.47e-15" Tdel="7.247000e-11" mux_trans_size="1.222260" buf_size="auto"/>  
+   </switchlist> 
+   <segmentlist>  
+      <!--- VB & JL: using ITRS metal stack data, 96 nm half pitch wires, which are intermediate metal width/space.  
 			     With the 96 nm half pitch, such wires would take 60 um of height, vs. a 90 nm high (approximated as square) Stratix IV tile so this seems
 			     reasonable. Using a tile length of 90 nm, corresponding to the length of a Stratix IV tile if it were square. -->  
-    <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->
-    <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">
-      <mux name="0"/>
-      <sb type="pattern">1 1 1 1 1</sb>
-      <cb type="pattern">1 1 1 1</cb>
-    </segment>
-  </segmentlist>
-  <complexblocklist>
-    <!-- Define I/O pads begin -->
-    <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->
-    <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->
-    <pb_type name="io">
-      <input name="outpad" num_pins="1"/>
-      <output name="inpad" num_pins="1"/>
-      <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
+      <!-- GIVE a specific name for the segment! OpenFPGA appreciate that! -->  
+      <segment name="L4" freq="1.000000" length="4" type="unidir" Rmetal="101" Cmetal="22.5e-15">   
+         <mux name="0"/>   
+         <sb type="pattern">1 1 1 1 1</sb>   
+         <cb type="pattern">1 1 1 1</cb>   
+      </segment>  
+   </segmentlist> 
+   <complexblocklist>  
+      <!-- Define I/O pads begin -->  
+      <!-- Capacity is a unique property of I/Os, it is the maximum number of I/Os that can be placed at the same (X,Y) location on the FPGA -->  
+      <!-- Not sure of the area of an I/O (varies widely), and it's not relevant to the design of the FPGA core, so we're setting it to 0. -->  
+      <pb_type name="io">   
+         <input name="outpad" num_pins="1"/>   
+         <output name="inpad" num_pins="1"/>   
+         <!-- Do NOT add clock pins to I/O here!!! VPR does not build clock network in the way that OpenFPGA can support
           If you need to register the I/O, define clocks in the circuit models
           These clocks can be handled in back-end
       -->   
-      <!-- A mode denotes the physical implementation of an I/O 
+         <!-- A mode denotes the physical implementation of an I/O 
           This mode will be not packable but is mainly used for fabric verilog generation   
        -->   
-      <mode name="physical" disable_packing="true">
-        <pb_type name="iopad" blif_model=".subckt io" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="iopad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>
-          </direct>
-          <direct name="inpad" input="iopad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
+         <mode name="physical" disable_packing="true">    
+            <pb_type name="iopad" blif_model=".subckt io" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="iopad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="iopad.outpad"/>      
+               </direct>     
+               <direct name="inpad" input="iopad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="iopad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   

-      <!-- IOs can operate as either inputs or outputs.
+         <!-- IOs can operate as either inputs or outputs.
 	     Delays below come from Ian Kuon. They are small, so they should be interpreted as
 	     the delays to and from registers in the I/O (and generally I/Os are registered 
 	     today and that is when you timing analyze them.
 	     -->   
-      <mode name="inpad">
-        <pb_type name="inpad" blif_model=".input" num_pb="1">
-          <output name="inpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="inpad" input="inpad.inpad" output="io.inpad">
-            <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <mode name="outpad">
-        <pb_type name="outpad" blif_model=".output" num_pb="1">
-          <input name="outpad" num_pins="1"/>
-        </pb_type>
-        <interconnect>
-          <direct name="outpad" input="io.outpad" output="outpad.outpad">
-            <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>
-          </direct>
-        </interconnect>
-      </mode>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- IOs go on the periphery of the FPGA, for consistency, 
+         <mode name="inpad">    
+            <pb_type name="inpad" blif_model=".input" num_pb="1">     
+               <output name="inpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="inpad" input="inpad.inpad" output="io.inpad">      
+                  <delay_constant max="4.243e-11" in_port="inpad.inpad" out_port="io.inpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <mode name="outpad">    
+            <pb_type name="outpad" blif_model=".output" num_pb="1">     
+               <input name="outpad" num_pins="1"/>     
+            </pb_type>    
+            <interconnect>     
+               <direct name="outpad" input="io.outpad" output="outpad.outpad">      
+                  <delay_constant max="1.394e-11" in_port="io.outpad" out_port="outpad.outpad"/>      
+               </direct>     
+            </interconnect>    
+         </mode>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- IOs go on the periphery of the FPGA, for consistency, 
          make it physically equivalent on all sides so that only one definition of I/Os is needed.
          If I do not make a physically equivalent definition, then I need to define 4 different I/Os, one for each side of the FPGA
        -->   
-      <!-- Place I/Os on the sides of the FPGA -->
-      <power method="ignore"/>
-    </pb_type>
-    <!-- Define I/O pads ends -->
-    <!-- Define general purpose logic block (CLB) begin -->
-    <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
+         <!-- Place I/Os on the sides of the FPGA -->   
+         <power method="ignore"/>   
+      </pb_type>  
+      <!-- Define I/O pads ends -->  
+      <!-- Define general purpose logic block (CLB) begin -->  
+      <!--- Area calculation: Total Stratix IV tile area is about 8100 um^2, and a minimum width transistor 
 	   area is 60 L^2 yields a tile area of 84375 MWTAs.
 	   Routing at W=300 is 30481 MWTAs, leaving us with a total of 53000 MWTAs for logic block area 
 	   This means that only 37% of our area is in the general routing, and 63% is inside the logic
@ -222,87 +221,87 @@
 	   assume, but note that the total routing area really includes the crossbar, which would push
 	   routing area up significantly, we estimate into the ~70% range. 
 	   -->  
-    <pb_type name="clb">
-      <input name="I" num_pins="32" equivalent="full"/>
-      <output name="O" num_pins="16" equivalent="none"/>
-      <clock name="clk" num_pins="1"/>
-      <!-- Describe fracturable logic element.  
+      <pb_type name="clb">   
+         <input name="I" num_pins="32" equivalent="full"/>   
+         <output name="O" num_pins="16" equivalent="none"/>   
+         <clock name="clk" num_pins="1"/>   
+         <!-- Describe fracturable logic element.  
             Each fracturable logic element has a 6-LUT that can alternatively operate as two 5-LUTs with shared inputs. 
             The outputs of the fracturable logic element can be optionally registered
        -->   
-      <pb_type name="fle" num_pb="8">
-        <input name="in" num_pins="6"/>
-        <output name="out" num_pins="2"/>
-        <clock name="clk" num_pins="1"/>
-        <!-- Physical mode definition begin (physical implementation of the fle) -->
-        <mode name="physical" disable_packing="true">
-          <pb_type name="fabric" num_pb="1">
+         <pb_type name="fle" num_pb="8">    
            <input name="in" num_pins="6"/>    
            <output name="out" num_pins="2"/>    
            <clock name="clk" num_pins="1"/>    
-            <pb_type name="frac_logic" num_pb="1">
-              <input name="in" num_pins="6"/>
-              <output name="out" num_pins="2"/>
-              <!-- Define LUT -->
-              <pb_type name="frac_lut6" blif_model=".subckt frac_lut6" num_pb="1">
-                <input name="in" num_pins="6"/>
-                <output name="lut5_out" num_pins="2"/>
-                <output name="lut6_out" num_pins="1"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="frac_logic.in" output="frac_lut6.in"/>
-                <direct name="direct2" input="frac_lut6.lut5_out[1]" output="frac_logic.out[1]"/>
-                <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->
-                <mux name="mux1" input="frac_lut6.lut6_out frac_lut6.lut5_out[0]" output="frac_logic.out[0]"/>
-              </interconnect>
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="fabric.in" output="frac_logic.in"/>
-              <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>
-              <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>
-              <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>
-                <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>
-              </mux>
-              <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>
-                <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="fabric.in"/>
-            <direct name="direct2" input="fabric.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="fabric.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Physical mode definition end (physical implementation of the fle) -->
-        <!-- Dual 5-LUT mode definition begin -->
-        <mode name="n2_lut5">
-          <pb_type name="lut5inter" num_pb="1">
-            <input name="in" num_pins="5"/>
-            <output name="out" num_pins="2"/>
-            <clock name="clk" num_pins="1"/>
-            <pb_type name="ble5" num_pb="2">
-              <input name="in" num_pins="5"/>
-              <output name="out" num_pins="1"/>
-              <clock name="clk" num_pins="1"/>
-              <!-- Define the LUT -->
-              <pb_type name="lut5" blif_model=".names" num_pb="1" class="lut">
-                <input name="in" num_pins="5" port_class="lut_in"/>
-                <output name="out" num_pins="1" port_class="lut_out"/>
-                <!-- LUT timing using delay matrix -->
-                <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+            <!-- Physical mode definition begin (physical implementation of the fle) -->    
+            <mode name="physical" disable_packing="true">     
+               <pb_type name="fabric" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="frac_logic" num_pb="1">       
+                     <input name="in" num_pins="6"/>       
+                     <output name="out" num_pins="2"/>       
+                     <!-- Define LUT -->       
+                     <pb_type name="frac_lut6" blif_model=".subckt frac_lut6" num_pb="1">        
+                        <input name="in" num_pins="6"/>        
+                        <output name="lut5_out" num_pins="2"/>        
+                        <output name="lut6_out" num_pins="1"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="frac_logic.in" output="frac_lut6.in"/>        
+                        <direct name="direct2" input="frac_lut6.lut5_out[1]" output="frac_logic.out[1]"/>        
+                        <!-- Xifan Tang: I use out[0] because the output of lut6 in lut6 mode is wired to the out[0] -->        
+                        <mux name="mux1" input="frac_lut6.lut6_out frac_lut6.lut5_out[0]" output="frac_logic.out[0]"/>        
+                     </interconnect>       
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="2" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="fabric.in" output="frac_logic.in"/>       
+                     <direct name="direct2" input="frac_logic.out[1:0]" output="ff[1:0].D"/>       
+                     <complete name="direct3" input="fabric.clk" output="ff[1:0].clk"/>       
+                     <mux name="mux1" input="ff[0].Q frac_logic.out[0]" output="fabric.out[0]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[0]" out_port="fabric.out[0]"/>        
+                        <delay_constant max="45e-12" in_port="ff[0].Q" out_port="fabric.out[0]"/>        
+                     </mux>       
+                     <mux name="mux2" input="ff[1].Q frac_logic.out[1]" output="fabric.out[1]">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="frac_logic.out[1]" out_port="fabric.out[1]"/>        
+                        <delay_constant max="45e-12" in_port="ff[1].Q" out_port="fabric.out[1]"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="fabric.in"/>      
+                  <direct name="direct2" input="fabric.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="fabric.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Physical mode definition end (physical implementation of the fle) -->    
+            <!-- Dual 5-LUT mode definition begin -->    
+            <mode name="n2_lut5">     
+               <pb_type name="lut5inter" num_pb="1">      
+                  <input name="in" num_pins="5"/>      
+                  <output name="out" num_pins="2"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <pb_type name="ble5" num_pb="2">       
+                     <input name="in" num_pins="5"/>       
+                     <output name="out" num_pins="1"/>       
+                     <clock name="clk" num_pins="1"/>       
+                     <!-- Define the LUT -->       
+                     <pb_type name="lut5" blif_model=".names" num_pb="1" class="lut">        
+                        <input name="in" num_pins="5" port_class="lut_in"/>        
+                        <output name="out" num_pins="1" port_class="lut_out"/>        
+                        <!-- LUT timing using delay matrix -->        
+                        <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                           we instead take the average of these numbers to get more stable results
                      82e-12
                      173e-12
@ -310,63 +309,63 @@
                      263e-12
                      398e-12
                      -->        
-                <delay_matrix type="max" in_port="lut5.in" out_port="lut5.out">
+                        <delay_matrix type="max" in_port="lut5.in" out_port="lut5.out">
                  235e-12
                  235e-12
                  235e-12
                  235e-12
                  235e-12
                </delay_matrix>        
-              </pb_type>
-              <!-- Define the flip-flop -->
-              <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-                <input name="D" num_pins="1" port_class="D"/>
-                <output name="Q" num_pins="1" port_class="Q"/>
-                <clock name="clk" num_pins="1" port_class="clock"/>
-                <T_setup value="66e-12" port="ff.D" clock="clk"/>
-                <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-              </pb_type>
-              <interconnect>
-                <direct name="direct1" input="ble5.in[4:0]" output="lut5[0:0].in[4:0]"/>
-                <direct name="direct2" input="lut5[0:0].out" output="ff[0:0].D">
-                  <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                  <pack_pattern name="ble5" in_port="lut5[0:0].out" out_port="ff[0:0].D"/>
-                </direct>
-                <direct name="direct3" input="ble5.clk" output="ff[0:0].clk"/>
-                <mux name="mux1" input="ff[0:0].Q lut5.out[0:0]" output="ble5.out[0:0]">
-                  <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                  <delay_constant max="25e-12" in_port="lut5.out[0:0]" out_port="ble5.out[0:0]"/>
-                  <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble5.out[0:0]"/>
-                </mux>
-              </interconnect>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="lut5inter.in" output="ble5[0:0].in"/>
-              <direct name="direct2" input="lut5inter.in" output="ble5[1:1].in"/>
-              <direct name="direct3" input="ble5[1:0].out" output="lut5inter.out"/>
-              <complete name="complete1" input="lut5inter.clk" output="ble5[1:0].clk"/>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in[4:0]" output="lut5inter.in"/>
-            <direct name="direct2" input="lut5inter.out" output="fle.out"/>
-            <direct name="direct3" input="fle.clk" output="lut5inter.clk"/>
-          </interconnect>
-        </mode>
-        <!-- Dual 5-LUT mode definition end -->
-        <!-- 6-LUT mode definition begin -->
-        <mode name="n1_lut6">
-          <!-- Define 6-LUT mode -->
-          <pb_type name="ble6" num_pb="1">
-            <input name="in" num_pins="6"/>
-            <output name="out" num_pins="1"/>
-            <clock name="clk" num_pins="1"/>
-            <!-- Define LUT -->
-            <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">
-              <input name="in" num_pins="6" port_class="lut_in"/>
-              <output name="out" num_pins="1" port_class="lut_out"/>
-              <!-- LUT timing using delay matrix -->
-              <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
+                     </pb_type>       
+                     <!-- Define the flip-flop -->       
+                     <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">        
+                        <input name="D" num_pins="1" port_class="D"/>        
+                        <output name="Q" num_pins="1" port_class="Q"/>        
+                        <clock name="clk" num_pins="1" port_class="clock"/>        
+                        <T_setup value="66e-12" port="ff.D" clock="clk"/>        
+                        <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>        
+                     </pb_type>       
+                     <interconnect>        
+                        <direct name="direct1" input="ble5.in[4:0]" output="lut5[0:0].in[4:0]"/>        
+                        <direct name="direct2" input="lut5[0:0].out" output="ff[0:0].D">         
+                           <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->         
+                           <pack_pattern name="ble5" in_port="lut5[0:0].out" out_port="ff[0:0].D"/>         
+                        </direct>        
+                        <direct name="direct3" input="ble5.clk" output="ff[0:0].clk"/>        
+                        <mux name="mux1" input="ff[0:0].Q lut5.out[0:0]" output="ble5.out[0:0]">         
+                           <!-- LUT to output is faster than FF to output on a Stratix IV -->         
+                           <delay_constant max="25e-12" in_port="lut5.out[0:0]" out_port="ble5.out[0:0]"/>         
+                           <delay_constant max="45e-12" in_port="ff[0:0].Q" out_port="ble5.out[0:0]"/>         
+                        </mux>        
+                     </interconnect>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="lut5inter.in" output="ble5[0:0].in"/>       
+                     <direct name="direct2" input="lut5inter.in" output="ble5[1:1].in"/>       
+                     <direct name="direct3" input="ble5[1:0].out" output="lut5inter.out"/>       
+                     <complete name="complete1" input="lut5inter.clk" output="ble5[1:0].clk"/>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in[4:0]" output="lut5inter.in"/>      
+                  <direct name="direct2" input="lut5inter.out" output="fle.out"/>      
+                  <direct name="direct3" input="fle.clk" output="lut5inter.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- Dual 5-LUT mode definition end -->    
+            <!-- 6-LUT mode definition begin -->    
+            <mode name="n1_lut6">     
+               <!-- Define 6-LUT mode -->     
+               <pb_type name="ble6" num_pb="1">      
+                  <input name="in" num_pins="6"/>      
+                  <output name="out" num_pins="1"/>      
+                  <clock name="clk" num_pins="1"/>      
+                  <!-- Define LUT -->      
+                  <pb_type name="lut6" blif_model=".names" num_pb="1" class="lut">       
+                     <input name="in" num_pins="6" port_class="lut_in"/>       
+                     <output name="out" num_pins="1" port_class="lut_out"/>       
+                     <!-- LUT timing using delay matrix -->       
+                     <!-- These are the physical delay inputs on a Stratix IV LUT but because VPR cannot do LUT rebalancing,
                       we instead take the average of these numbers to get more stable results
                  82e-12
                  173e-12
@ -375,7 +374,7 @@
                  398e-12
                  397e-12
                  -->       
-              <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
+                     <delay_matrix type="max" in_port="lut6.in" out_port="lut6.out">
                261e-12
                261e-12
                261e-12
@ -383,39 +382,39 @@
                261e-12
                261e-12
              </delay_matrix>       
-            </pb_type>
-            <!-- Define flip-flop -->
-            <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">
-              <input name="D" num_pins="1" port_class="D"/>
-              <output name="Q" num_pins="1" port_class="Q"/>
-              <clock name="clk" num_pins="1" port_class="clock"/>
-              <T_setup value="66e-12" port="ff.D" clock="clk"/>
-              <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>
-            </pb_type>
-            <interconnect>
-              <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>
-              <direct name="direct2" input="lut6.out" output="ff.D">
-                <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->
-                <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>
-              </direct>
-              <direct name="direct3" input="ble6.clk" output="ff.clk"/>
-              <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">
-                <!-- LUT to output is faster than FF to output on a Stratix IV -->
-                <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>
-                <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>
-              </mux>
-            </interconnect>
-          </pb_type>
-          <interconnect>
-            <direct name="direct1" input="fle.in" output="ble6.in"/>
-            <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>
-            <direct name="direct3" input="fle.clk" output="ble6.clk"/>
-          </interconnect>
-        </mode>
-        <!-- 6-LUT mode definition end -->
-      </pb_type>
-      <interconnect>
-        <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
+                  </pb_type>      
+                  <!-- Define flip-flop -->      
+                  <pb_type name="ff" blif_model=".latch" num_pb="1" class="flipflop">       
+                     <input name="D" num_pins="1" port_class="D"/>       
+                     <output name="Q" num_pins="1" port_class="Q"/>       
+                     <clock name="clk" num_pins="1" port_class="clock"/>       
+                     <T_setup value="66e-12" port="ff.D" clock="clk"/>       
+                     <T_clock_to_Q max="124e-12" port="ff.Q" clock="clk"/>       
+                  </pb_type>      
+                  <interconnect>       
+                     <direct name="direct1" input="ble6.in" output="lut6[0:0].in"/>       
+                     <direct name="direct2" input="lut6.out" output="ff.D">        
+                        <!-- Advanced user option that tells CAD tool to find LUT+FF pairs in netlist -->        
+                        <pack_pattern name="ble6" in_port="lut6.out" out_port="ff.D"/>        
+                     </direct>       
+                     <direct name="direct3" input="ble6.clk" output="ff.clk"/>       
+                     <mux name="mux1" input="ff.Q lut6.out" output="ble6.out">        
+                        <!-- LUT to output is faster than FF to output on a Stratix IV -->        
+                        <delay_constant max="25e-12" in_port="lut6.out" out_port="ble6.out"/>        
+                        <delay_constant max="45e-12" in_port="ff.Q" out_port="ble6.out"/>        
+                     </mux>       
+                  </interconnect>      
+               </pb_type>     
+               <interconnect>      
+                  <direct name="direct1" input="fle.in" output="ble6.in"/>      
+                  <direct name="direct2" input="ble6.out" output="fle.out[0:0]"/>      
+                  <direct name="direct3" input="fle.clk" output="ble6.clk"/>      
+               </interconnect>     
+            </mode>    
+            <!-- 6-LUT mode definition end -->    
+         </pb_type>   
+         <interconnect>    
+            <!-- We use a full crossbar to get logical equivalence at inputs of CLB 
 		     The delays below come from Stratix IV. the delay through a connection block
 		     input mux + the crossbar in Stratix IV is 167 ps. We already have a 72 ps 
 		     delay on the connection block input mux (modeled by Ian Kuon), so the remaining
@ -424,23 +423,23 @@
 		     Since all our outputs LUT outputs go to a BLE output, and have a delay of 
 		     25 ps to do so, we subtract 25 ps from the 100 ps delay of a feedback
 		     to get the part that should be marked on the crossbar.	 -->    
-        <complete name="crossbar" input="clb.I fle[7:0].out" output="fle[7:0].in">
-          <delay_constant max="95e-12" in_port="clb.I" out_port="fle[7:0].in"/>
-          <delay_constant max="75e-12" in_port="fle[7:0].out" out_port="fle[7:0].in"/>
+            <complete name="crossbar" input="clb.I fle[7:0].out" output="fle[7:0].in">     
+               <delay_constant max="95e-12" in_port="clb.I" out_port="fle[7:0].in"/>     
+               <delay_constant max="75e-12" in_port="fle[7:0].out" out_port="fle[7:0].in"/>     
+            </complete>    
+            <complete name="clks" input="clb.clk" output="fle[7:0].clk">
        </complete>    
-        <complete name="clks" input="clb.clk" output="fle[7:0].clk">
-        </complete>
-        <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
+            <!-- This way of specifying direct connection to clb outputs is important because this architecture uses automatic spreading of opins.  
               By grouping to output pins in this fashion, if a logic block is completely filled by 6-LUTs, 
               then the outputs those 6-LUTs take get evenly distributed across all four sides of the CLB instead of clumped on two sides (which is what happens with a more
               naive specification).
          -->    
-        <direct name="clbouts1" input="fle[7:0].out[0:0]" output="clb.O[7:0]"/>
-        <direct name="clbouts2" input="fle[7:0].out[1:1]" output="clb.O[15:8]"/>
-      </interconnect>
-      <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->
-      <!-- Place this general purpose logic block in any unspecified column -->
-    </pb_type>
-    <!-- Define general purpose logic block (CLB) ends -->
-  </complexblocklist>
+            <direct name="clbouts1" input="fle[7:0].out[0:0]" output="clb.O[7:0]"/>    
+            <direct name="clbouts2" input="fle[7:0].out[1:1]" output="clb.O[15:8]"/>    
+         </interconnect>   
+         <!-- Every input pin is driven by 15% of the tracks in a channel, every output pin is driven by 10% of the tracks in a channel -->   
+         <!-- Place this general purpose logic block in any unspecified column -->   
+      </pb_type>  
+      <!-- Define general purpose logic block (CLB) ends -->  
+   </complexblocklist> 
 </architecture>