////////////////////////////////////////////////////////////////////////////////
//
/// LSU EE 4755 Fall 2026 Homework 2
//

 /// Assignment  https://www.ece.lsu.edu/koppel/v/2026/hw02.pdf

 /// Instructions:
  //
  // (1) Find the undergraduate workstation laboratory, room 2241 Patrick
  //     F. Taylor Hall.
  //
  // (2) Locate your account.  If you did not get an account please
  //     E-mail: koppel@ece.lsu.edu
  //
  // (3) Log in to a Linux workstation.
  //
  // (4) If you haven't already, follow the account setup instructions here:
  //     https://www.ece.lsu.edu/koppel/v/proc.html
  //
  // (5) Copy this assignment, local path name
  //     /home/faculty/koppel/pub/ee4755/hw/2026/hw02
  //     to a directory ~/hw02 in your class account. (~ is your home
  //     directory.) Use this file for your solution.
  ///      BE SURE THAT YOUR FILE IS CORRECTLY NAMED AND IN THE RIGHT PLACE.
  //
  // (6) Find the problems in this file and solve them.
  //
  //     Your entire solution should be in this file.
  //
  //     Do not change module names.
  //
  // (7) Your solution will automatically be copied from your account by
  //     the TA-bot.
  //

 /// Additional Resources
  //
  // Verilog Documentation
  //    The Verilog Standard
  //      https://ieeexplore.ieee.org/document/10458102
  //    Introductory Treatment (Warning: Does not include SystemVerilog)
  //      Brown & Vranesic, Fundamentals of Digital Logic with Verilog, 3rd Ed.
  //
  // Account Setup and Emacs (Text Editor) Instructions
  //      https://www.ece.lsu.edu/koppel/v/proc.html
  //      To learn Emacs look for Emacs tutorial.
  //


//////////////////////////////////////////////////////////////////////////////
/// Common Settings
 //
 // Do not modify the Verilog in this section.

`default_nettype none

package hw2;
typedef enum { M_lj_parts, M_lj_custom, M_lj_procedural } M_LJ_Method;
typedef enum { M_i_unsigned, M_i_twos_comp, M_i_sign_mag } M_I_Format;
typedef enum logic [2:0]
  { Rnd_to_even = 0, Rnd_to_0 = 1, Rnd_to_plus_if = 2,
    Rnd_to_minus_inf = 3, Rnd_to_plus_inf = 4, Rnd_from_0 = 5 }
    Rnd;

endpackage
import hw2::*;


//////////////////////////////////////////////////////////////////////////////
///  Problem 1
//
  /// Complete tiger_itof.
  //  
//
//     [ ] Complete tiger_itof so that it passes tests using left_justify_parts.
//     [ ] Your solution must use the outputs of left_justify_wrapper.
//     [ ] Your solution should modify an input to left_justify_wrapper.
//
//     [ ] Make sure that the testbench does not report errors for ljp.
//     [ ] The module must be synthesizable.
//
//     [ ] Code must be written clearly.
//     [ ] Pay attention to cost and performance.


module tiger_itof
  #( M_LJ_Method lj_way = M_lj_procedural, M_I_Format i_fmt = M_i_twos_comp,
    int w_i = 10, w_exp = 4, w_sig = 8, w_fp = w_exp + w_sig + 1 )
   ( output uwire [w_fp-1:0] f,
     input uwire [w_i-1:0] i );

   localparam bit fmt_unsigned = i_fmt == M_i_unsigned;
   localparam bit fmt_twos_comp = i_fmt == M_i_twos_comp;
   localparam bit fmt_sign_mag = i_fmt == M_i_sign_mag;

   localparam int bias = ( 1 << w_exp - 1 ) - 1;

   uwire [w_i-1:0] i_left;
   localparam int w_nlz = $clog2(w_i);
   uwire [w_nlz-1:0] nlz;  // Number of leading zeros.

   // [ ] Convert integer i to a floating point representation.
   // [ ] The module must use the outputs of left_justify_wrapper.
   // [ ] Pay attention to i_fmt (unsigned, signed, or sign-magnitude.)
   // [ ] Round values toward zero. (Important when w_i > w_sig.)
   // [ ] Value of lj_way does not affect code in this module.

   left_justify_wrapper #(lj_way, w_i) lj( i_left, nlz, i );

   uwire sign = i[w_i-1];
   uwire [w_exp-1:0] exp = nlz;
   uwire [w_sig-1:0] sig = i_left;

   assign f = { sign, exp, i_left };

endmodule


//////////////////////////////////////////////////////////////////////////////
///  Problem 2
//
  /// Complete left_justify_custom.
  //  
//
//     [ ] Complete left_justify_custom so that no ljc errors are reported.
//     [ ] The module must not use procedural code.
//     [ ] Use a generate loop, don't use recursion.
//
//     [ ] Make sure that the testbench does not report errors.
//     [ ] The module must be synthesizable.
//
//     [ ] Design the module to minimize cost rather than performance.
//     [ ] Code must be written clearly.

module left_justify_custom
  #( int w = 16, lgw = $clog2(w) )
   ( output uwire [w-1:0] justified,
     output uwire [lgw-1:0] nlz,
     input uwire [w-1:0] un );


   // [ ] Set nlz to the number of leading zeros if un is non-zero.
   // [ ] Shift un to the left by the number of leading zeros.
   // [ ] Base design on shift_left_logarithmic.
   // [ ] Reduce cost over clz by looking at already shifted values.
   // [ ] To get started, look at shift_left_logarithmic, further below.

   assign justified = un;
   assign nlz = w/2;


endmodule


//////////////////////////////////////////////////////////////////////////////
/// Reference Modules


module left_justify_procedural
  #( int w = 16, lgw = $clog2(w) )
   ( output logic [w-1:0] justified,
     output logic [lgw-1:0] nlz,
     input uwire [w-1:0] un );

   always_comb begin
      nlz = 0;
      justified = un;
      for ( int i=0; i<w; i++ ) begin
         if ( justified[w-1] == 1 ) break;
         nlz++;
         justified = justified << 1;
      end
   end

endmodule

module left_justify_parts
  #( int w = 16, w_nlz = $clog2(w) )
   ( output uwire [w-1:0] justified,
     output uwire [w_nlz-1:0] nlz,
     input uwire [w-1:0] un );

   clz_tree #(w,w_nlz) cl( nlz, un );
   shift_left_logarithmic #(w) ls( justified, un, nlz );

endmodule

module shift_left_logarithmic
  #( int w = 16, lgw = $clog2(w) )
   ( output uwire [w-1:0] shifted,
     input uwire [w-1:0] un,
     input uwire [lgw-1:0] amt );

   uwire [w-1:0] s[lgw:-1];  // Array of wires to interconnect mux2 instances.
   assign s[-1] = un;

   for ( genvar i=0; i<lgw; i++ )
     mux2 #(w) st( s[i],  amt[i],  s[i-1],  s[i-1] << ( 1 << i )  );

   assign shifted = s[lgw-1];

endmodule

module mux2 #( int w = 5 )
   ( output uwire [w-1:0] x, input uwire select, input uwire [w-1:0] a0, a1 );
   assign x = select ? a1 : a0;
endmodule

module left_justify_wrapper
  #( M_LJ_Method lj_way, int w = 16, w_nlz = $clog2(w) )
   ( output uwire [w-1:0] justified,
     output uwire [w_nlz-1:0] nlz,
     input uwire [w-1:0] i );

   case ( lj_way )
     M_lj_parts:
       left_justify_parts #(w,w_nlz) lj( justified, nlz, i );
     M_lj_procedural:
       left_justify_procedural #(w,w_nlz) lj( justified, nlz, i );
     M_lj_custom:
       left_justify_custom #(w,w_nlz) lj( justified, nlz, i );
   endcase

endmodule

module clz_tree
  #( int w = 19,
     int ww = $clog2(w+1) )
   ( output uwire [ww-1:0] nlz_rv,
     input uwire [w-1:0] a );

   // Handle case where ww < $clog2(w+1), perhaps because a=0 is special case.
   localparam int w_nlz = $clog2(w+1);
   uwire [w_nlz-1:0] nlz;
   assign nlz_rv = nlz;

   if ( w == 1 ) begin

      assign nlz = !a[0];

   end else if ( w == 2 ) begin

      assign nlz = { !a[0] && !a[1], !a[1] && a[0] };

   end else begin

      localparam int lhi = $clog2(w) - 1;
      localparam int whi = 1 << lhi;
      localparam int wwhi = lhi + 1;

      localparam int wlo = w - whi;
      localparam int wwlo = $clog2(wlo+1);

      uwire [wwlo-1:0] nlz_lo;
      uwire [wwhi-1:0] nlz_hi;

      // Declare recursive modules.
      //
      clz_tree #(wlo) clo( nlz_lo, a[wlo-1:0] );
      clz_tree #(whi) chi( nlz_hi, a[w-1:wlo] );

      uwire ov_lo, ov_hi;
      uwire [lhi-1:0] lz_lo, lz_hi;

      // The hi bits are easy because there's always an overflow.
      assign {ov_hi,lz_hi} = nlz_hi;

      if ( wlo == whi ) assign {ov_lo,lz_lo} = nlz_lo;
      else              assign lz_lo = nlz_lo;

      assign nlz[lhi-1:0] = ov_hi ? lz_lo : lz_hi;

      if ( wlo == whi )
        assign nlz[lhi+1:lhi] = { ov_hi && ov_lo,  ov_hi && !ov_lo };
      else
        assign nlz[lhi] = ov_hi;

   end

endmodule


//////////////////////////////////////////////////////////////////////////////
/// Testbench Code
//
// It is okay to modify the testbench code to facilitate the coding
// and debugging of your modules. Keep in mind that your submission
// will be tested using a different testbench, so no one will be
// accused of what is now called reward hacking (tampering with the
// test) for modifying the testbench below. If you do modify the
// testbench be sure to make sure all of the original tests are
// performed to make sure that your code passes the original
// testbench.

// cadence translate_off
`define SIMULATION_ON
// cadence translate_on

module cw_itof
  #( M_LJ_Method lj_way = M_lj_procedural, M_I_Format i_fmt = M_i_twos_comp,
    int w_i = 10, w_exp = 4, w_sig = 8, w_fp = w_exp + w_sig + 1 )
   ( output uwire [w_fp-1:0] f,
     input uwire [w_i-1:0] i );

   localparam bit fmt_twos_comp = i_fmt == M_i_twos_comp;

   uwire [w_i-1:0] i_left;
   localparam int w_nlz = $clog2(w_i);
   uwire [w_nlz-1:0] nlz;  // Number of leading zeros.

   localparam Rnd rnd = Rnd_to_0;

`ifdef SIMULATION_ON
   left_justify_wrapper #(M_lj_parts, w_i) lj( i_left, nlz, i );
`endif

   CW_fp_i2flt #( .sig_width(w_sig), .exp_width(w_exp), .isize(w_i),
                  .isign( fmt_twos_comp) )
   coa( .z(f), .a(i), .rnd(rnd),  .status() );

endmodule


// cadence translate_off

virtual class conv #(int wexp=6, wsig=10);
   // Convert between real and fp types using parameter-provided
   // exponent and significand sizes.

   localparam int w = 1 + wexp + wsig;
   localparam int bias_r = ( 1 << 11 - 1 ) - 1;
   localparam int w_sig_r = 52;
   localparam int w_exp_r = 11;
   localparam int bias_h = ( 1 << wexp - 1 ) - 1;

   static function logic [w-1:0] rtof( real r );
      logic [wsig-1:0] sig_f;
      logic [w_sig_r-wsig-2:0] sig_x;
      logic sig_x_msb;
      logic [w_exp_r-1:0] exp_r;
      logic sign_r;
      { sign_r, exp_r, sig_f, sig_x_msb, sig_x } = $realtobits(r);
      // So, what about a rounding mode? Not now!
      rtof = !r ? 0 : { sign_r, wexp'( exp_r + bias_h - bias_r ), sig_f };
   endfunction

   static function real ftor( logic [w-1:0] f );
      ftor = !f ? 0.0
        : $bitstoreal
          ( { f[w-1],
              w_exp_r'( bias_r + f[w-2:wsig] - bias_h ),
              f[wsig-1:0], (w_sig_r-wsig)'(0) } );
   endfunction

   static function int err_bits( logic [w-1:0] a, b );

      logic [wsig-1:0] sig_a, sig_b;
      logic [wsig+2:0] frac_a, frac_b, frac_diff;
      logic [wexp-1:0] exp_a, exp_b;
      logic s_a, s_b;
      int delta_e;

      if ( $isunknown(a) || $isunknown(b) ) return 1 << wexp;
      if ( a == b ) return 0;

      { s_a, exp_a, sig_a } = a;
      { s_b, exp_b, sig_b } = b;

      if ( exp_a == 0 || exp_b == 0 ) begin
         logic [wsig-1:0] sig = ~ ( sig_a | sig_b );
         return 1 + wsig - $clog2( sig + 1 );
      end

      delta_e = $abs( 0 + exp_a - exp_b );
      if ( delta_e > 1 ) return delta_e + wsig;
      frac_a = exp_a > exp_b ? { 2'b1, sig_a, 1'b0 } : { 3'b1, sig_a };
      frac_b = exp_b > exp_a ? { 2'b1, sig_b, 1'b0 } : { 3'b1, sig_b };
      frac_diff =
        s_a != s_b ? frac_a + frac_b :
        frac_a > frac_b ? frac_a - frac_b : frac_b - frac_a;
      return $clog2( frac_diff + 1 );

   endfunction

endclass


module testbench;

   localparam int n_tests = 1000;

   localparam int npsets = 3; // This MUST be set to the size of pset.
   // { w_exp, w_sig, w_i }
   localparam int pset[npsets][3] =
              '{
                { 7,  6,  3 },
                { 7,  8,  8 },
                { 8,  5,  12 }
                };

   localparam int nmsets = 2;
   localparam int nisets = 3;
   localparam M_LJ_Method mset[2] = '{ M_lj_parts, M_lj_custom };
   localparam M_I_Format iset[3] =
              '{ M_i_unsigned, M_i_twos_comp, M_i_sign_mag };

   string mtype_str[M_LJ_Method] =
          '{ M_lj_parts: "lj_parts", M_lj_custom: "lj_cust" };
   string mtype_abbr[M_LJ_Method] = '{ M_lj_parts: "ljp", M_lj_custom: "ljc" };
   string itype_str[M_I_Format] =
          '{ M_i_unsigned: "unsigned", M_i_twos_comp: "twos-comp",
             M_i_sign_mag: "sign-mag"};
   string itype_abbr[M_I_Format] =
          '{ M_i_unsigned: "uns", M_i_twos_comp: "2sc", M_i_sign_mag: "smg" };

   int t_errs_mod[M_LJ_Method][M_I_Format];
   int t_errs_each_f[M_LJ_Method][M_I_Format][int];
   int t_errs_each_nlz[M_LJ_Method][M_I_Format][int];
   int t_errs_each_lj[M_LJ_Method][M_I_Format][int];
   int t_errs_lj_f[M_LJ_Method][M_I_Format];
   int t_errs_lj_nlz[M_LJ_Method][M_I_Format];
   int t_errs_lj_lj[M_LJ_Method][M_I_Format];
   int t_errs_size_f[int];
   int t_errs_size_nlz[int];
   int t_errs_size_lj[int];
   int t_errs_each[M_LJ_Method][M_I_Format][int];

   localparam int nsets = npsets * nmsets * nisets;

   logic d[nsets:-1]; // Start / Done signals.

   int t_errs_f, t_errs_nlz, t_errs_lj;     // Total number of errors.
   initial begin
      t_errs_f = 0;
      t_errs_lj = 0;
      t_errs_nlz = 0;
      for ( int m=0; m<nmsets; m++ )
        for ( int mf=0; mf<nisets; mf++ )
          for ( int i=0; i<npsets; i++ ) begin
             t_errs_each_f[mset[m]][iset[mf]][i] = -1;
             t_errs_each_nlz[mset[m]][iset[mf]][i] = -1;
             t_errs_each_lj[mset[m]][iset[mf]][i] = -1;
             t_errs_lj_f[mset[m]][iset[mf]] = 0;
             t_errs_lj_nlz[mset[m]][iset[mf]] = 0;
             t_errs_lj_lj[mset[m]][iset[mf]] = 0;
        end

      d[-1] = 1;
   end

   final begin
      for ( int mi=0; mi<nmsets; mi++ )
        for ( int fi=0; fi<nisets; fi++ )
          for ( int i=0; i<npsets; i++ ) begin
             automatic M_LJ_Method m = mset[mi];
             automatic M_I_Format mf = iset[fi];
             t_errs_lj_f[m][mf] += t_errs_each_f[m][mf][i];
             t_errs_lj_nlz[m][mf] += t_errs_each_nlz[m][mf][i];
             t_errs_lj_lj[m][mf] += t_errs_lj_nlz[m][mf][i];
             if ( 0 )
             $write("Total %s %s w_exp=%0d, w_sig=%0d, w_i=%2d: Errors: %0d nlz, %0d lj, %0d f.\n",
                    mtype_str[m], itype_str[mf],
                    pset[i][0], pset[i][1], pset[i][2],
                    t_errs_each_nlz[m][mf][i], t_errs_each_lj[m][mf][i],
                    t_errs_each_f[m][mf][i]
                  );
          end
      for ( int mi=0; mi<nmsets; mi++ )
        for ( int fi=0; fi<nisets; fi++ ) begin
           automatic M_LJ_Method m = mset[mi];
           automatic M_I_Format mf = iset[fi];
      $write("Total %-8s %-9s Errors: %0d nlz, %0d lj, %0d f.\n",
                    mtype_str[m], itype_str[mf],
                    t_errs_lj_nlz[m][mf], t_errs_lj_lj[m][mf],
                    t_errs_lj_f[m][mf]
                  );
          end

      $write("Total All Configs:  Errors: %0d nlz, %0d lj, %0d f.\n",
             t_errs_nlz, t_errs_lj, t_errs_f);

   end

   for ( genvar m=0; m<nmsets; m++ )
     for ( genvar mf=0; mf<nisets; mf++ )
       for ( genvar i=0; i<npsets; i++ ) begin
          localparam int idx = m * npsets * nisets + mf * npsets + i;
          testbench_n
            #( .w_exp(pset[i][0]), .w_sig(pset[i][1]), .w_i(pset[i][2]),
               .pset(i), .mtype(mset[m]), .itype(iset[mf]) )
          t2( .done(d[idx]), .tstart(d[idx-1]) );
     end

endmodule


module testbench_n
  #( int w_exp = 5, w_sig = 8, w_i = 12,
     pset = 0, M_LJ_Method mtype = M_lj_parts, M_I_Format itype = M_i_unsigned )
   ( output logic done, input uwire tstart );

   // Number of sample outputs to print (whether correct or not).
   localparam int n_samples = 2;

   localparam int w_fp = 1 + w_sig + w_exp;

   localparam int w_nlz = $clog2(w_i)+1;

   function automatic int clz( logic [w_i-1:0] a );
      int nlz = 0;
      for ( int i=0; i<w_i; i++ ) if ( a[i] == 1 ) nlz = w_i - i - 1;
      clz = nlz;
   endfunction

   logic [w_i-1:0] ival;
   logic [w_fp-1:0] fval;

   tiger_itof #( mtype, itype, w_i, w_exp, w_sig ) mut_itof(fval,ival);
   //cw_itof #( mtype, itype, w_i, w_exp, w_sig ) mut_itof(fval,ival);

   initial begin

      automatic int n_tests = testbench.n_tests;
      automatic int n_err_f = 0, n_err_nlz = 0, n_err_lj = 0;
      automatic int n_f = 0;

      wait( tstart );

      $write("\nStarting Int %0s w_i=%0d, Left-Just %0s, FP Fmt w_exp=%0d, w_sig=%0d\n",
             testbench.itype_str[itype], w_i,
             testbench.mtype_str[mtype], w_exp, w_sig);

      for (int i=0; i<n_tests; i++ ) begin

         automatic int io2 = i / 2;
         logic [w_fp-1:0] shadow_fval;
         logic signed [w_i:0] shadow_ival; // For conversion, printing.
         logic shadow_sign, fval_sign;
         logic [w_sig-1:0] shadow_sig, fval_sig;
         logic [w_exp-1:0] shadow_exp, fval_exp;

         logic [w_i-1:0] shadow_i_left, shadow_i_abs;
         logic [w_nlz-1:0] shadow_nlz;

         bit err_f, err, err_nlz, err_lj;

         ival = i[0] ? io2 : ~io2;

         shadow_ival = itype == M_i_unsigned ? { 1'b0, ival } :
           itype == M_i_twos_comp ? { ival[w_i-1], ival } :
             itype == M_i_sign_mag ? { 2'b0, ival[w_i-2:0] }
               : 0;

         shadow_i_abs = shadow_ival < 0 ? -shadow_ival : shadow_ival;
         shadow_nlz = shadow_i_abs ? clz(shadow_i_abs) : 0;
         shadow_i_left = shadow_nlz ? shadow_i_abs << shadow_nlz : shadow_i_abs;

         shadow_fval = conv#(w_exp,w_sig)::rtof(real'(shadow_ival));
         if ( itype == M_i_sign_mag ) shadow_fval[w_fp-1] = ival[w_i-1];
         {shadow_sign,shadow_exp,shadow_sig} = shadow_fval;

         #1;
         {fval_sign,fval_exp,fval_sig} = fval;

         err_f = shadow_fval !== fval;

         err_nlz = shadow_i_abs && shadow_nlz !== mut_itof.lj.nlz;
         err_lj = shadow_i_left !== mut_itof.lj.justified;

         err = err_f || err_nlz || err_lj;

         if ( i < n_samples || err ) begin
            if ( err_f ) n_err_f++;
            if ( err_nlz ) n_err_nlz++;
            if ( err_lj ) n_err_lj++;

            if ( i < n_samples || err_f && n_err_f < 5
                 || err_nlz && n_err_nlz < 5
                 || err_lj && n_err_lj < 5 ) begin
              $write( "%s %s %s #(%0d,%0d,%0d) i=%0d='h%h, nlz,lj: {%d,'h%h} %s {%d,'h%h} (correct)\n",
                      err_nlz || err_lj ? "Error " : "Sample",
                      testbench.mtype_abbr[mtype],
                      testbench.itype_abbr[itype],
                      w_exp, w_sig, w_i,
                      shadow_ival, ival,
                      mut_itof.lj.nlz, mut_itof.lj.justified,
                      err_nlz || err_lj ? "!=" : "==",
                      shadow_nlz, shadow_i_left );
              $write( "%s %s %s #(%0d,%0d,%0d)  s,exp,sig: {%1d,%0d,'h%h} %s {%1d,%0d,'h%h} (correct)\n",
                      err_f ? "Error " : "Sample",
                      testbench.mtype_abbr[mtype],
                      testbench.itype_abbr[itype],
                      w_exp, w_sig, w_i,
                      fval_sign,fval_exp,fval_sig,
                      err_f ? "!=" : "==",
                      shadow_sign,shadow_exp,shadow_sig);
            end
         end

      end

      $write("Finished  %4s %3s exp=%0d, sig=%0d, wi=%0d. Errors: %0d nlz, %0d lj, %0d f\n",
             testbench.mtype_str[mtype], testbench.itype_str[itype],
             w_exp, w_sig, w_i,
             n_err_nlz, n_err_lj, n_err_f);
      //  $write("Frac gt %.4f\n", real'(n_f)/n_tests);

      testbench.t_errs_f += n_err_f;
      testbench.t_errs_nlz += n_err_nlz;
      testbench.t_errs_lj += n_err_lj;
      testbench.t_errs_each_f[mtype][itype][pset] = n_err_f;
      testbench.t_errs_each_nlz[mtype][itype][pset] = n_err_nlz;
      testbench.t_errs_each_lj[mtype][itype][pset] = n_err_lj;

      done = 1;
   end

endmodule



// cadence translate_on

`default_nettype wire

`ifdef SIMULATION_ON

`include "/apps/linux/cadence/DDIEXPORT23/GENUS231/share/synth/lib/chipware/sim/verilog/CW/CW_fp_i2flt.v"

`else

`include "/apps/linux/cadence/DDIEXPORT23/GENUS231/share/synth/lib/chipware/syn/CW/CW_fp_i2flt.v"

`endif