| Line 817... |
Line 817... |
"\t// if (IWIDTH < CWIDTH) then ...\n"
|
"\t// if (IWIDTH < CWIDTH) then ...\n"
|
"\t// However, this is the only (other) way I know to do it.\n"
|
"\t// However, this is the only (other) way I know to do it.\n"
|
"\tgenerate\n"
|
"\tgenerate\n"
|
"\tif (CWIDTH < IWIDTH+1)\n"
|
"\tif (CWIDTH < IWIDTH+1)\n"
|
"\tbegin\n"
|
"\tbegin\n"
|
|
"\t\twire\t[(CWIDTH):0]\tp3c_in;\n"
|
|
"\t\twire\t[(IWIDTH+1):0]\tp3d_in;\n"
|
|
"\t\tassign\tp3c_in = ir_coef_i + ir_coef_r;\n"
|
|
"\t\tassign\tp3d_in = r_dif_r + r_dif_i;\n"
|
|
"\n"
|
"\t\t// We need to pad these first two multiplies by an extra\n"
|
"\t\t// We need to pad these first two multiplies by an extra\n"
|
"\t\t// bit just to keep them aligned with the third,\n"
|
"\t\t// bit just to keep them aligned with the third,\n"
|
"\t\t// simpler, multiply.\n"
|
"\t\t// simpler, multiply.\n"
|
"\t\tshiftaddmpy #(CWIDTH+1,IWIDTH+2) p1(i_clk, i_ce,\n"
|
"\t\tshiftaddmpy #(CWIDTH+1,IWIDTH+2) p1(i_clk, i_ce,\n"
|
"\t\t\t\t{ir_coef_r[CWIDTH-1],ir_coef_r},\n"
|
"\t\t\t\t{ir_coef_r[CWIDTH-1],ir_coef_r},\n"
|
"\t\t\t\t{r_dif_r[IWIDTH],r_dif_r}, p_one);\n"
|
"\t\t\t\t{r_dif_r[IWIDTH],r_dif_r}, p_one);\n"
|
"\t\tshiftaddmpy #(CWIDTH+1,IWIDTH+2) p2(i_clk, i_ce,\n"
|
"\t\tshiftaddmpy #(CWIDTH+1,IWIDTH+2) p2(i_clk, i_ce,\n"
|
"\t\t\t\t{ir_coef_i[CWIDTH-1],ir_coef_i},\n"
|
"\t\t\t\t{ir_coef_i[CWIDTH-1],ir_coef_i},\n"
|
"\t\t\t\t{r_dif_i[IWIDTH],r_dif_i}, p_two);\n"
|
"\t\t\t\t{r_dif_i[IWIDTH],r_dif_i}, p_two);\n"
|
"\t\tshiftaddmpy #(CWIDTH+1,IWIDTH+2) p3(i_clk, i_ce,\n"
|
"\t\tshiftaddmpy #(CWIDTH+1,IWIDTH+2) p3(i_clk, i_ce,\n"
|
"\t\t\t\tir_coef_i+ir_coef_r,\n"
|
"\t\t\t\tp3c_in, p3d_in, p_three);\n"
|
"\t\t\t\tr_dif_r + r_dif_i,\n"
|
|
"\t\t\t\tp_three);\n"
|
|
"\tend else begin\n"
|
"\tend else begin\n"
|
|
"\t\twire\t[(CWIDTH):0]\tp3c_in;\n"
|
|
"\t\twire\t[(IWIDTH+1):0]\tp3d_in;\n"
|
|
"\t\tassign\tp3c_in = ir_coef_i + ir_coef_r;\n"
|
|
"\t\tassign\tp3d_in = r_dif_r + r_dif_i;\n"
|
|
"\n"
|
"\t\tshiftaddmpy #(IWIDTH+2,CWIDTH+1) p1a(i_clk, i_ce,\n"
|
"\t\tshiftaddmpy #(IWIDTH+2,CWIDTH+1) p1a(i_clk, i_ce,\n"
|
"\t\t\t\t{r_dif_r[IWIDTH],r_dif_r},\n"
|
"\t\t\t\t{r_dif_r[IWIDTH],r_dif_r},\n"
|
"\t\t\t\t{ir_coef_r[CWIDTH-1],ir_coef_r}, p_one);\n"
|
"\t\t\t\t{ir_coef_r[CWIDTH-1],ir_coef_r}, p_one);\n"
|
"\t\tshiftaddmpy #(IWIDTH+2,CWIDTH+1) p2a(i_clk, i_ce,\n"
|
"\t\tshiftaddmpy #(IWIDTH+2,CWIDTH+1) p2a(i_clk, i_ce,\n"
|
"\t\t\t\t{r_dif_i[IWIDTH], r_dif_i},\n"
|
"\t\t\t\t{r_dif_i[IWIDTH], r_dif_i},\n"
|
"\t\t\t\t{ir_coef_i[CWIDTH-1],ir_coef_i}, p_two);\n"
|
"\t\t\t\t{ir_coef_i[CWIDTH-1],ir_coef_i}, p_two);\n"
|
"\t\tshiftaddmpy #(IWIDTH+2,CWIDTH+1) p3a(i_clk, i_ce,\n"
|
"\t\tshiftaddmpy #(IWIDTH+2,CWIDTH+1) p3a(i_clk, i_ce,\n"
|
"\t\t\t\tr_dif_r+r_dif_i,\n"
|
"\t\t\t\tp3d_in, p3c_in, p_three);\n"
|
"\t\t\t\tir_coef_i+ir_coef_r,\n"
|
|
"\t\t\t\tp_three);\n"
|
|
"\tend\n"
|
"\tend\n"
|
"\tendgenerate\n"
|
"\tendgenerate\n"
|
"\n");
|
"\n");
|
fprintf(fp,
|
fprintf(fp,
|
"\t// These values are held in memory and delayed during the\n"
|
"\t// These values are held in memory and delayed during the\n"
|
| Line 935... |
Line 941... |
"\n"
|
"\n"
|
"endmodule\n");
|
"endmodule\n");
|
fclose(fp);
|
fclose(fp);
|
}
|
}
|
|
|
void build_stage(const char *fname, int stage, bool odd, int nbits, bool inv, int xtra) {
|
void build_hwbfly(const char *fname, int xtracbits) {
|
|
FILE *fp = fopen(fname, "w");
|
|
if (NULL == fp) {
|
|
fprintf(stderr, "Could not open \'%s\' for writing\n", fname);
|
|
perror("O/S Err was:");
|
|
return;
|
|
}
|
|
|
|
fprintf(fp,
|
|
"///////////////////////////////////////////////////////////////////////////\n"
|
|
"//\n"
|
|
"// Filename: hwbfly.v\n"
|
|
"//\n"
|
|
"// Project: %s\n"
|
|
"//\n"
|
|
"// Purpose: This routine is identical to the butterfly.v routine found\n"
|
|
"// in 'butterfly.v', save only that it uses the verilog \n"
|
|
"// operator '*' in hopes that the synthesizer would be able\n"
|
|
"// to optimize it with hardware resources.\n"
|
|
"//\n"
|
|
"// It is understood that a hardware multiply can complete its\n"
|
|
"// operation in a single clock.\n"
|
|
"//\n"
|
|
"//\n%s"
|
|
"//\n", prjname, creator);
|
|
fprintf(fp, "%s", cpyleft);
|
|
fprintf(fp,
|
|
"module hwbfly(i_clk, i_rst, i_ce, i_coef, i_left, i_right, i_aux,\n"
|
|
"\t\to_left, o_right, o_aux);\n"
|
|
"\t// Public changeable parameters ...\n"
|
|
"\tparameter IWIDTH=16,CWIDTH=IWIDTH+%d,OWIDTH=IWIDTH+1;\n"
|
|
"\t// Parameters specific to the core that should not be changed.\n"
|
|
"\tparameter\tSHIFT=0, ROUND=1;\n"
|
|
"\tinput\t\ti_clk, i_rst, i_ce;\n"
|
|
"\tinput\t\t[(2*CWIDTH-1):0]\ti_coef;\n"
|
|
"\tinput\t\t[(2*IWIDTH-1):0]\ti_left, i_right;\n"
|
|
"\tinput\t\ti_aux;\n"
|
|
"\toutput\twire\t[(2*OWIDTH-1):0]\to_left, o_right;\n"
|
|
"\toutput\treg\to_aux;\n"
|
|
"\n", xtracbits);
|
|
fprintf(fp,
|
|
"\twire\t[(OWIDTH-1):0] o_left_r, o_left_i, o_right_r, o_right_i;\n"
|
|
"\n"
|
|
"\treg\t[(2*IWIDTH-1):0] r_left, r_right;\n"
|
|
"\treg\t r_aux, r_aux_2;\n"
|
|
"\treg\t[(2*CWIDTH-1):0] r_coef, r_coef_2;\n"
|
|
"\twire\tsigned [(CWIDTH-1):0] r_coef_r, r_coef_i;\n"
|
|
"\tassign\tr_coef_r = r_coef_2[ (2*CWIDTH-1):(CWIDTH)];\n"
|
|
"\tassign\tr_coef_i = r_coef_2[ ( CWIDTH-1):0];\n"
|
|
"\twire signed [(IWIDTH-1):0] r_left_r, r_left_i, r_right_r, r_right_i;\n"
|
|
"\tassign\tr_left_r = r_left[ (2*IWIDTH-1):(IWIDTH)];\n"
|
|
"\tassign\tr_left_i = r_left[ (IWIDTH-1):0];\n"
|
|
"\tassign\tr_right_r = r_right[(2*IWIDTH-1):(IWIDTH)];\n"
|
|
"\tassign\tr_right_i = r_right[(IWIDTH-1):0];\n"
|
|
"\n"
|
|
"\treg signed [(IWIDTH):0] r_sum_r, r_sum_i, r_dif_r, r_dif_i;\n"
|
|
"\n"
|
|
"\treg [(2*IWIDTH+2):0] leftv, leftvv;\n"
|
|
"\n"
|
|
"\t// Set up the input to the multiply\n"
|
|
"\talways @(posedge i_clk)\n"
|
|
"\t\tif (i_rst)\n"
|
|
"\t\tbegin\n"
|
|
"\t\t\tr_aux <= 1'b0;\n"
|
|
"\t\t\tr_aux_2 <= 1'b0;\n"
|
|
"\t\tend else if (i_ce)\n"
|
|
"\t\tbegin\n"
|
|
"\t\t\t// One clock just latches the inputs\n"
|
|
"\t\t\tr_left <= i_left; // No change in # of bits\n"
|
|
"\t\t\tr_right <= i_right;\n"
|
|
"\t\t\tr_aux <= i_aux;\n"
|
|
"\t\t\tr_coef <= i_coef;\n"
|
|
"\t\t\t// Next clock adds/subtracts\n"
|
|
"\t\t\tr_sum_r <= r_left_r + r_right_r; // Now IWIDTH+1 bits\n"
|
|
"\t\t\tr_sum_i <= r_left_i + r_right_i;\n"
|
|
"\t\t\tr_dif_r <= r_left_r - r_right_r;\n"
|
|
"\t\t\tr_dif_i <= r_left_i - r_right_i;\n"
|
|
"\t\t\t// Other inputs are simply delayed on second clock\n"
|
|
"\t\t\tr_aux_2 <= r_aux;\n"
|
|
"\t\t\tr_coef_2<= r_coef;\n"
|
|
"\t\tend\n"
|
|
"\n\n");
|
|
fprintf(fp,
|
|
"\t// See comments in the butterfly.v source file for a discussion of\n"
|
|
"\t// these operations and the appropriate bit widths.\n\n");
|
|
fprintf(fp,
|
|
"\twire signed [(CWIDTH-1):0] ir_coef_r, ir_coef_i;\n"
|
|
"\tassign ir_coef_r = r_coef_2[(2*CWIDTH-1):CWIDTH];\n"
|
|
"\tassign ir_coef_i = r_coef_2[(CWIDTH-1):0];\n"
|
|
"\treg\tsigned [((IWIDTH+2)+(CWIDTH+1)-1):0] p_one, p_two, p_three;\n"
|
|
"\n"
|
|
"\treg\tsigned [(CWIDTH):0] p3c_in, p1c_in, p2c_in;\n"
|
|
"\treg\tsigned [(IWIDTH+1):0] p3d_in, p1d_in, p2d_in;\n"
|
|
"\treg\t[3:0] pipeline;\n"
|
|
"\n"
|
|
"\talways @(posedge i_clk)\n"
|
|
"\tbegin\n"
|
|
"\t\tif (i_rst)\n"
|
|
"\t\tbegin\n"
|
|
"\t\t\tpipeline <= 4'h0;\n"
|
|
"\t\t\tleftv <= 0;\n"
|
|
"\t\t\tleftvv <= 0;\n"
|
|
"\t\tend else if (i_clk)\n"
|
|
"\t\tbegin\n"
|
|
"\t\t\t// Second clock, pipeline = 1\n"
|
|
"\t\t\tp1c_in <= { ir_coef_r[(CWIDTH-1)], ir_coef_r };\n"
|
|
"\t\t\tp2c_in <= { ir_coef_i[(CWIDTH-1)], ir_coef_i };\n"
|
|
"\t\t\tp1d_in <= { r_dif_r[(IWIDTH)], r_dif_r };\n"
|
|
"\t\t\tp2d_in <= { r_dif_i[(IWIDTH)], r_dif_i };\n"
|
|
"\t\t\tp3c_in <= ir_coef_i + ir_coef_r;\n"
|
|
"\t\t\tp3d_in <= r_dif_r + r_dif_i;\n"
|
|
"\n "
|
|
"\t\t\tleftv <= { r_aux_2, r_sum_r, r_sum_i };\n"
|
|
"\n "
|
|
"\t\t\t// Third clock, pipeline = 3\n"
|
|
"\t\t\tp_one <= p1c_in * p1d_in;\n"
|
|
"\t\t\tp_two <= p2c_in * p2d_in;\n"
|
|
"\t\t\tp_three <= p3c_in * p3d_in;\n"
|
|
"\t\t\tleftvv <= leftv;\n"
|
|
"\n"
|
|
"\t\t\tpipeline <= { pipeline[2:0], 1'b1 };\n"
|
|
"\t\tend\n"
|
|
"\tend\n"
|
|
"\n");
|
|
|
|
fprintf(fp,
|
|
"\t// These values are held in memory and delayed during the\n"
|
|
"\t// multiply. Here, we recover them. During the multiply,\n"
|
|
"\t// values were multiplied by 2^(CWIDTH-2)*exp{-j*2*pi*...},\n"
|
|
"\t// therefore, the left_x values need to be right shifted by\n"
|
|
"\t// CWIDTH-2 as well. The additional bits come from a sign\n"
|
|
"\t// extension.\n"
|
|
"\twire\taux_s;\n"
|
|
"\twire\tsigned\t[(IWIDTH+CWIDTH):0] left_si, left_sr;\n"
|
|
"\treg\t\t[(2*IWIDTH+2):0] left_saved;\n"
|
|
"\tassign\tleft_sr = { {2{left_saved[2*(IWIDTH+1)-1]}}, left_saved[(2*(IWIDTH+1)-1):(IWIDTH+1)], {(CWIDTH-2){1'b0}} };\n"
|
|
"\tassign\tleft_si = { {2{left_saved[(IWIDTH+1)-1]}}, left_saved[((IWIDTH+1)-1):0], {(CWIDTH-2){1'b0}} };\n"
|
|
"\tassign\taux_s = left_saved[2*IWIDTH+2];\n"
|
|
"\n"
|
|
"\n"
|
|
"\treg signed [(CWIDTH+IWIDTH+3-1):0] b_left_r, b_left_i,\n"
|
|
"\t\t\t\t\t\tb_right_r, b_right_i;\n"
|
|
"\treg signed [(CWIDTH+IWIDTH+3-1):0] mpy_r, mpy_i;\n"
|
|
"\twire signed [(CWIDTH+IWIDTH+3-1):0] rnd;\n"
|
|
"\tgenerate\n"
|
|
"\tif ((ROUND==0)||(CWIDTH+IWIDTH-OWIDTH-SHIFT<2))\n"
|
|
"\t\tassign rnd = ({(CWIDTH+IWIDTH+3){1'b0}});\n"
|
|
"\telse if ((IWIDTH+CWIDTH)-(OWIDTH+SHIFT) == 2)\n"
|
|
"\t\tassign rnd = ({ {(OWIDTH+4+SHIFT){1'b0}},1'b1 });\n"
|
|
"\telse\n"
|
|
"\t\tassign rnd = ({ {(OWIDTH+4+SHIFT){1'b0}},1'b1,\n"
|
|
"\t\t\t\t{((IWIDTH+CWIDTH+3)-(OWIDTH+SHIFT+5)){1'b0}} });\n"
|
|
"\tendgenerate\n"
|
|
"\n");
|
|
|
|
fprintf(fp,
|
|
"\talways @(posedge i_clk)\n"
|
|
"\t\tif (i_rst)\n"
|
|
"\t\tbegin\n"
|
|
"\t\t\tleft_saved <= 0;\n"
|
|
"\t\t\tb_left_r <= 0;\n"
|
|
"\t\t\tb_left_i <= 0;\n"
|
|
"\t\t\tb_right_r <= 0;\n"
|
|
"\t\t\tb_right_i <= 0;\n"
|
|
"\t\t\to_aux <= 1'b0;\n"
|
|
"\t\tend else if (i_ce)\n"
|
|
"\t\tbegin\n"
|
|
"\t\t\t// First clock, recover all values\n"
|
|
"\t\t\tleft_saved <= leftvv;\n"
|
|
"\t\t\t// These values are IWIDTH+CWIDTH+3 bits wide\n"
|
|
"\t\t\t// although they only need to be (IWIDTH+1)\n"
|
|
"\t\t\t// + (CWIDTH) bits wide. (We've got two\n"
|
|
"\t\t\t// extra bits we need to get rid of.)\n"
|
|
"\t\t\tmpy_r <= p_one - p_two;\n"
|
|
"\t\t\tmpy_i <= p_three - p_one - p_two;\n"
|
|
"\n"
|
|
"\t\t\t// Second clock, round and latch for final clock\n"
|
|
"\t\t\tb_right_r <= mpy_r + rnd;\n"
|
|
"\t\t\tb_right_i <= mpy_i + rnd;\n"
|
|
"\t\t\tb_left_r <= { {2{left_sr[(IWIDTH+CWIDTH)]}},left_sr } + rnd;\n"
|
|
"\t\t\tb_left_i <= { {2{left_si[(IWIDTH+CWIDTH)]}},left_si } + rnd;\n"
|
|
"\n"
|
|
"\t\t\to_aux <= aux_s;\n"
|
|
"\t\tend\n"
|
|
"\n");
|
|
|
|
fprintf(fp,
|
|
"\t// Final step--remove unnecessary bits.\n"
|
|
"\tassign o_left_r = b_left_r[ (CWIDTH+IWIDTH-1-SHIFT-1):(CWIDTH+IWIDTH-OWIDTH-SHIFT-1)];\n"
|
|
"\tassign o_left_i = b_left_i[ (CWIDTH+IWIDTH-1-SHIFT-1):(CWIDTH+IWIDTH-OWIDTH-SHIFT-1)];\n"
|
|
"\tassign o_right_r = b_right_r[(CWIDTH+IWIDTH-1-SHIFT-1):(CWIDTH+IWIDTH-OWIDTH-SHIFT-1)];\n"
|
|
"\tassign o_right_i = b_right_i[(CWIDTH+IWIDTH-1-SHIFT-1):(CWIDTH+IWIDTH-OWIDTH-SHIFT-1)];\n"
|
|
"\n"
|
|
"\t// As a final step, we pack our outputs into two packed two's\n"
|
|
"\t// complement numbers per output word, so that each output word\n"
|
|
"\t// has (2*OWIDTH) bits in it, with the top half being the real\n"
|
|
"\t// portion and the bottom half being the imaginary portion.\n"
|
|
"\tassign\to_left = { o_left_r, o_left_i };\n"
|
|
"\tassign\to_right= { o_right_r,o_right_i};\n"
|
|
"\n"
|
|
"endmodule\n");
|
|
|
|
}
|
|
|
|
void build_stage(const char *fname, int stage, bool odd, int nbits, bool inv, int xtra, bool hwmpy=false) {
|
FILE *fstage = fopen(fname, "w");
|
FILE *fstage = fopen(fname, "w");
|
int cbits = nbits + xtra;
|
int cbits = nbits + xtra;
|
|
|
if ((cbits * 2) >= sizeof(long long)*8) {
|
if ((cbits * 2) >= sizeof(long long)*8) {
|
fprintf(stderr, "ERROR: CMEM Coefficient precision requested overflows long long data type.\n");
|
fprintf(stderr, "ERROR: CMEM Coefficient precision requested overflows long long data type.\n");
|
| Line 1128... |
Line 1338... |
"\t\t\t\to_sync <= 1'b0;\n"
|
"\t\t\t\to_sync <= 1'b0;\n"
|
"\t\t\tend else\n"
|
"\t\t\tend else\n"
|
"\t\t\t\to_sync <= 1'b0;\n"
|
"\t\t\t\to_sync <= 1'b0;\n"
|
"\t\tend\n"
|
"\t\tend\n"
|
"\n", (inv)?"i":"");
|
"\n", (inv)?"i":"");
|
|
if (hwmpy) {
|
|
fprintf(fstage,
|
|
"\thwbfly #(.IWIDTH(IWIDTH),.CWIDTH(CWIDTH),.OWIDTH(OWIDTH),\n"
|
|
"\t\t\t.SHIFT(BFLYSHIFT))\n"
|
|
"\t\tbfly(i_clk, i_rst, i_ce, ib_c,\n"
|
|
"\t\t\tib_a, ib_b, ib_sync, ob_a, ob_b, ob_sync);\n");
|
|
} else {
|
fprintf(fstage,
|
fprintf(fstage,
|
"\tbutterfly #(.IWIDTH(IWIDTH),.CWIDTH(CWIDTH),.OWIDTH(OWIDTH),\n"
|
"\tbutterfly #(.IWIDTH(IWIDTH),.CWIDTH(CWIDTH),.OWIDTH(OWIDTH),\n"
|
"\t\t\t.MPYDELAY(%d\'d%d),.LGDELAY(LGBDLY),.SHIFT(BFLYSHIFT))\n"
|
"\t\t\t.MPYDELAY(%d\'d%d),.LGDELAY(LGBDLY),.SHIFT(BFLYSHIFT))\n"
|
"\t\tbfly(i_clk, i_rst, i_ce, ib_c,\n"
|
"\t\tbfly(i_clk, i_rst, i_ce, ib_c,\n"
|
"\t\t\tib_a, ib_b, ib_sync, ob_a, ob_b, ob_sync);\n"
|
"\t\t\tib_a, ib_b, ib_sync, ob_a, ob_b, ob_sync);\n",
|
"endmodule\n",
|
|
lgdelay(nbits, xtra), bflydelay(nbits, xtra));
|
lgdelay(nbits, xtra), bflydelay(nbits, xtra));
|
}
|
}
|
|
fprintf(fstage, "endmodule\n");
|
|
}
|
|
|
void usage(void) {
|
void usage(void) {
|
fprintf(stderr,
|
fprintf(stderr,
|
"USAGE:\tfftgen [-f <size>] [-d dir] [-c cbits] [-n nbits] [-m mxbits] [-s01]\n"
|
"USAGE:\tfftgen [-f <size>] [-d dir] [-c cbits] [-n nbits] [-m mxbits] [-s01]\n"
|
// "\tfftgen -i\n"
|
// "\tfftgen -i\n"
|
| Line 1152... |
Line 1370... |
"\t-n <nbits>\tSets the number of bits in the twos complement input\n"
|
"\t-n <nbits>\tSets the number of bits in the twos complement input\n"
|
"\t\tto the FFT routine.\n"
|
"\t\tto the FFT routine.\n"
|
"\t-m <mxbits>\tSets the maximum bit width that the FFT should ever\n"
|
"\t-m <mxbits>\tSets the maximum bit width that the FFT should ever\n"
|
"\t\tproduce. Internal values greater than this value will be\n"
|
"\t\tproduce. Internal values greater than this value will be\n"
|
"\t\ttruncated to this value.\n"
|
"\t\ttruncated to this value.\n"
|
|
"\t-n <nbits>\tSets the bitwidth for values coming into the (i)FFT.\n"
|
|
"\t-p <nmpy>\tSets the number of stages that will use any hardware \n"
|
|
"\t\tmultiplication facility, instead of shift-add emulation.\n"
|
"\t-s\tSkip the final bit reversal stage. This is useful in\n"
|
"\t-s\tSkip the final bit reversal stage. This is useful in\n"
|
"\t\talgorithms that need to apply a filter without needing to do\n"
|
"\t\talgorithms that need to apply a filter without needing to do\n"
|
"\t\tbin shifting, as these algorithms can, with this option, just\n"
|
"\t\tbin shifting, as these algorithms can, with this option, just\n"
|
"\t\tmultiply by a bit reversed correlation sequence and then\n"
|
"\t\tmultiply by a bit reversed correlation sequence and then\n"
|
"\t\tinverse FFT the (still bit reversed) result.\n"
|
"\t\tinverse FFT the (still bit reversed) result. (You would need\n"
|
|
"\t\ta decimation in time inverse to do this, which this program does\n"
|
|
"\t\tnot yet provide.)\n"
|
"\t-S\tInclude the final bit reversal stage (default).\n"
|
"\t-S\tInclude the final bit reversal stage (default).\n"
|
|
"\t-x <xtrabits>\tUse this many extra bits internally, before any final\n"
|
|
"\t\trounding or truncation of the answer to the final number of bits.\n"
|
"\t-0\tA forward FFT (default), meaning that the coefficients are\n"
|
"\t-0\tA forward FFT (default), meaning that the coefficients are\n"
|
"\t\tgiven by e^{-j 2 pi k/N n }.\n"
|
"\t\tgiven by e^{-j 2 pi k/N n }.\n"
|
"\t-1\tAn inverse FFT, meaning that the coefficients are\n"
|
"\t-1\tAn inverse FFT, meaning that the coefficients are\n"
|
"\t\tgiven by e^{ j 2 pi k/N n }.\n");
|
"\t\tgiven by e^{ j 2 pi k/N n }.\n");
|
}
|
}
|
| Line 1172... |
Line 1397... |
// Obviously, the build_stage above.
|
// Obviously, the build_stage above.
|
// Copying the files of interest into the fft-core directory, from
|
// Copying the files of interest into the fft-core directory, from
|
// whatever directory this file is run out of.
|
// whatever directory this file is run out of.
|
int main(int argc, char **argv) {
|
int main(int argc, char **argv) {
|
int fftsize = -1, lgsize = -1;
|
int fftsize = -1, lgsize = -1;
|
int nbitsin = 16, xtracbits = 4;
|
int nbitsin = 16, xtracbits = 4, nummpy=0, nonmpy=2;
|
int nbitsout, maxbitsout = -1, xtrapbits=0;
|
int nbitsout, maxbitsout = -1, xtrapbits=0;
|
bool bitreverse = true, inverse=false, interactive = false,
|
bool bitreverse = true, inverse=false, interactive = false,
|
verbose_flag = false;
|
verbose_flag = false;
|
FILE *vmain;
|
FILE *vmain;
|
std::string coredir = "fft-core", cmdline = "";
|
std::string coredir = "fft-core", cmdline = "";
|
| Line 1262... |
Line 1487... |
exit(-1);
|
exit(-1);
|
}
|
}
|
nbitsin = atoi(argv[++argn]);
|
nbitsin = atoi(argv[++argn]);
|
j += 200;
|
j += 200;
|
break;
|
break;
|
|
case 'p':
|
|
if (argn+1 >= argc) {
|
|
printf("ERR: No number given for number of hardware multiply stages!\n\n");
|
|
exit(-1);
|
|
}
|
|
nummpy = atoi(argv[++argn]);
|
|
j += 200;
|
|
break;
|
case 'S':
|
case 'S':
|
bitreverse = true;
|
bitreverse = true;
|
break;
|
break;
|
case 's':
|
case 's':
|
bitreverse = false;
|
bitreverse = false;
|
| Line 1344... |
Line 1577... |
if (fftsize <= 2)
|
if (fftsize <= 2)
|
bitreverse = false;
|
bitreverse = false;
|
} if ((maxbitsout > 0)&&(nbitsout > maxbitsout))
|
} if ((maxbitsout > 0)&&(nbitsout > maxbitsout))
|
nbitsout = maxbitsout;
|
nbitsout = maxbitsout;
|
|
|
|
// Figure out how many multiply stages to use, and how many to skip
|
|
{
|
|
int lgv = lgval(fftsize);
|
|
|
|
nonmpy = lgv - nummpy;
|
|
if (nonmpy < 2) nonmpy = 2;
|
|
nummpy = lgv - nonmpy;
|
|
}
|
|
|
{
|
{
|
struct stat sbuf;
|
struct stat sbuf;
|
if (lstat(coredir.c_str(), &sbuf)==0) {
|
if (lstat(coredir.c_str(), &sbuf)==0) {
|
if (!S_ISDIR(sbuf.st_mode)) {
|
if (!S_ISDIR(sbuf.st_mode)) {
|
| Line 1484... |
Line 1725... |
fprintf(vmain, "\n\n");
|
fprintf(vmain, "\n\n");
|
|
|
{
|
{
|
std::string fname;
|
std::string fname;
|
char numstr[12];
|
char numstr[12];
|
|
bool mpystage;
|
|
|
|
// Last two stages are always non-multiply stages
|
|
// since the multiplies can be done by adds
|
|
mpystage = ((lgtmp-2) <= nummpy);
|
|
|
fname = coredir + "/";
|
fname = coredir + "/";
|
if (inverse) fname += "i";
|
if (inverse) fname += "i";
|
fname += "fftstage_e";
|
fname += "fftstage_e";
|
sprintf(numstr, "%d", fftsize);
|
sprintf(numstr, "%d", fftsize);
|
fname += numstr;
|
fname += numstr;
|
fname += ".v";
|
fname += ".v";
|
build_stage(fname.c_str(), fftsize/2, 0, nbits, inverse, xtracbits); // Even stage
|
build_stage(fname.c_str(), fftsize/2, 0, nbits, inverse, xtracbits, mpystage); // Even stage
|
|
|
fname = coredir + "/";
|
fname = coredir + "/";
|
if (inverse) fname += "i";
|
if (inverse) fname += "i";
|
fname += "fftstage_o";
|
fname += "fftstage_o";
|
sprintf(numstr, "%d", fftsize);
|
sprintf(numstr, "%d", fftsize);
|
fname += numstr;
|
fname += numstr;
|
fname += ".v";
|
fname += ".v";
|
build_stage(fname.c_str(), fftsize/2, 1, nbits, inverse, xtracbits); // Odd stage
|
build_stage(fname.c_str(), fftsize/2, 1, nbits, inverse, xtracbits, mpystage); // Odd stage
|
}
|
}
|
|
|
nbits += 1; // New number of input bits
|
nbits += 1; // New number of input bits
|
tmp_size >>= 1; lgtmp--;
|
tmp_size >>= 1; lgtmp--;
|
dropbit = 0;
|
dropbit = 0;
|
| Line 1531... |
Line 1777... |
fprintf(vmain, "\n\n");
|
fprintf(vmain, "\n\n");
|
|
|
{
|
{
|
std::string fname;
|
std::string fname;
|
char numstr[12];
|
char numstr[12];
|
|
bool mpystage;
|
|
|
|
mpystage = ((lgtmp-2) <= nummpy);
|
|
|
fname = coredir + "/";
|
fname = coredir + "/";
|
if (inverse) fname += "i";
|
if (inverse) fname += "i";
|
fname += "fftstage_e";
|
fname += "fftstage_e";
|
sprintf(numstr, "%d", tmp_size);
|
sprintf(numstr, "%d", tmp_size);
|
fname += numstr;
|
fname += numstr;
|
fname += ".v";
|
fname += ".v";
|
build_stage(fname.c_str(), tmp_size/2, 0, nbits+xtrapbits, inverse, xtracbits); // Even stage
|
build_stage(fname.c_str(), tmp_size/2, 0,
|
|
nbits+xtrapbits, inverse, xtracbits,
|
|
mpystage); // Even stage
|
|
|
fname = coredir + "/";
|
fname = coredir + "/";
|
if (inverse) fname += "i";
|
if (inverse) fname += "i";
|
fname += "fftstage_o";
|
fname += "fftstage_o";
|
sprintf(numstr, "%d", tmp_size);
|
sprintf(numstr, "%d", tmp_size);
|
fname += numstr;
|
fname += numstr;
|
fname += ".v";
|
fname += ".v";
|
build_stage(fname.c_str(), tmp_size/2, 1, nbits+xtrapbits, inverse, xtracbits); // Odd stage
|
build_stage(fname.c_str(), tmp_size/2, 1,
|
|
nbits+xtrapbits, inverse, xtracbits,
|
|
mpystage); // Odd stage
|
}
|
}
|
|
|
|
|
dropbit ^= 1;
|
dropbit ^= 1;
|
nbits = obits;
|
nbits = obits;
|
| Line 1639... |
Line 1892... |
std::string fname;
|
std::string fname;
|
|
|
fname = coredir + "/butterfly.v";
|
fname = coredir + "/butterfly.v";
|
build_butterfly(fname.c_str(), xtracbits);
|
build_butterfly(fname.c_str(), xtracbits);
|
|
|
|
if (nummpy > 0) {
|
|
fname = coredir + "/hwbfly.v";
|
|
build_hwbfly(fname.c_str(), xtracbits);
|
|
}
|
|
|
fname = coredir + "/shiftaddmpy.v";
|
fname = coredir + "/shiftaddmpy.v";
|
build_multiply(fname.c_str());
|
build_multiply(fname.c_str());
|
|
|
fname = coredir + "/qtrstage.v";
|
fname = coredir + "/qtrstage.v";
|
build_quarters(fname.c_str());
|
build_quarters(fname.c_str());
|