多彩编程 多彩编程MZPH · CODE BLOG
ARTICLE DETAIL

文章详情

深耕前端与后端开发技术的一线实战笔记与踩坑复盘。

【集创赛国一】手把手教你 FPGA 图像处理算法【第四讲】3×3 滑动窗口搭建 + 均值滤波、高斯滤波 Verilog 实现

【集创赛国一】手把手教你 FPGA 图像处理算法【第四讲】3×3 滑动窗口搭建 + 均值滤波、高斯滤波 Verilog 实现 前言几乎所有图像算法高斯滤波、均值滤波、中值滤波、Sobel 边缘检测、形态学腐蚀膨胀都依赖 3×3 邻域窗口。软件 OpenCV直接访问二维数组下标随便取周围 9 个像素。FPGA 硬件像素是串行流同一时刻只有 1 个当前像素想要拿到上下两行像素必须使用行 FIFO行缓存做数据缓存拼接出 3 行图像数据生成 3×3 窗口矩阵。本讲先讲解行 FIFO 原理然后实现均值滤波、高斯滤波输出处理后像素并且严格同步vs/hs/de时序信号。1、3×3 滑动窗口硬件原理1.1 行缓存 FIFO 架构图像一行像素宽度为COL使用两个同步 FIFO 缓存前两行图像。• 新像素不断输入• FIFO1 存储上一行全部像素• FIFO2 存储上上行全部像素每个时钟周期同时输出三行各 3 个像素拼成 3×3 矩阵matrix_22是窗口中心像素。边界处理图像最前几行 / 最末行、最左 / 最右列没有邻域像素本工程采用像素复制填充避免边界黑边、异常噪点。⚠️重点矩阵生成会引入时钟延迟后续卷积运算还会消耗流水线节拍vs/hs/de时序信号必须严格同步打拍否则图像错位撕裂。2 、滤波算法原理2.1 均值滤波均值滤波用于抑制高斯随机噪声窗口内全部像素求和取平均。输出公式FPGA 避免除法利用等价系数 先左移等价乘以 1024再乘以系数 114最后截取高位完成除法。2.2 高斯滤波高斯滤波为加权平滑距离中心像素越远权重越低。本工程使用自定义加权卷积核计算流程分为 4 级流水线1 级每个像素与对应权重做乘法2 级按行累加求和3 级三行结果总求和4 级移位归一化、饱和截断输出 8bit 灰度结果。提示中间运算位宽必须充分扩展防止乘法、加法溢出。3、 FPGA 工程踩坑总结✅行 FIFO 深度等于图像行宽度FIFO 读写使能需要配合行、列计数器row_cnt / col_cnt✅图像边界一定要做填充处理直接卷积会出现边缘噪点本模块内部已经做边界复制✅卷积乘加运算必须扩展寄存器位宽防止溢出✅流水线多少拍vs/hs/de时序信号就必须同步打多少拍不能复用原始输入时序✅彩色图像建议 RGB 转 YCbCr只对 Y 亮度通道滤波Cb、Cr 色度通道单独打拍对齐❌不要直接对 RGB 三通道同时滤波资源开销大容易出现色彩偏移。附录完整 Verilog 源码module matrix_3x3_10bit #( parameter COL 640, parameter ROW 480 ) ( input clk, input rst_n, input valid_in,//输入数据有效信号 input signed [9:0] din, //输入的图像数据将一帧的数据从左到右然后从上到下依次输入 output reg [9:0] matrix_11, output reg [9:0] matrix_12, output reg [9:0] matrix_13, output reg [9:0] matrix_21, output reg [9:0] matrix_22, output reg [9:0] matrix_23, output reg [9:0] matrix_31, output reg [9:0] matrix_32, output reg [9:0] matrix_33 ); reg [9:0] col_cnt; reg [9:0] row_cnt; always (posedge clk or negedge rst_n) if(rst_n 1b0) col_cnt 11d0; else if(col_cnt COL-1 valid_in 1b1) col_cnt 11d0; else if(valid_in 1b1) col_cnt col_cnt 1b1; else col_cnt col_cnt; always (posedge clk or negedge rst_n) if(rst_n 1b0) row_cnt 11d0; else if(row_cnt ROW-1 col_cnt COL-1 valid_in 1b1) row_cnt 11d0; else if(col_cnt COL-1 valid_in 1b1) row_cnt row_cnt 1b1; wire [9:0] q_1; wire [9:0] q_2; wire [9:0] dout_r2; wire [9:0] dout_r1; wire [9:0] dout_r0; assign dout_r2 din; assign dout_r1 q_1; assign dout_r0 q_2; wire wr_en_1; wire rd_en_1; wire wr_en_2; wire rd_en_2; assign wr_en_1 (row_cnt ROW - 1) ? valid_in : 1b0; //不写最后1行 assign rd_en_1 (row_cnt 0) ? valid_in : 1b0; //从第1行开始读 assign wr_en_2 (row_cnt ROW - 2) ? valid_in : 1b0; //不写最后2行 assign rd_en_2 (row_cnt 1) ? valid_in : 1b0; //从第2行开始读 FIFO_SC_10bit_Top u1_FIFO_SC_10bit_Top( .data(din), //input [9:0] Data .clock(clk), //input Clk .wrreq(wr_en_1), //input WrEn .rdreq(rd_en_1), //input RdEn .aclr(~rst_n), //input Reset .q(q_1), //output [9:0] Q .empty(), //output Empty .full() //output Full ); FIFO_SC_10bit_Top u2_FIFO_SC_10bit_Top( .data(din), //input [9:0] Data .clock(clk), //input Clk .wrreq(wr_en_2), //input WrEn .rdreq(rd_en_2), //input RdEn .aclr(~rst_n), //input Reset .q(q_2), //output [9:0] Q .empty(), //output Empty .full() //output Full ); always (posedge clk or negedge rst_n) begin if(!rst_n) begin {matrix_11, matrix_12, matrix_13} {10d0, 10d0, 10d0}; {matrix_21, matrix_22, matrix_23} {10d0, 10d0, 10d0}; {matrix_31, matrix_32, matrix_33} {10d0, 10d0, 10d0}; end //------------------------------------------------------------------------- 第1排矩阵 else if(row_cnt 0)begin if(col_cnt 0) begin //第1个矩阵 {matrix_11, matrix_12, matrix_13} {dout_r2, dout_r2, dout_r2}; {matrix_21, matrix_22, matrix_23} {dout_r2, dout_r2, dout_r2}; {matrix_31, matrix_32, matrix_33} {dout_r2, dout_r2, dout_r2}; end else begin //剩余矩阵 {matrix_11, matrix_12, matrix_13} {matrix_12, matrix_13, dout_r2}; {matrix_21, matrix_22, matrix_23} {matrix_22, matrix_23, dout_r2}; {matrix_31, matrix_32, matrix_33} {matrix_32, matrix_33, dout_r2}; end end //------------------------------------------------------------------------- 第2排矩阵 else if(row_cnt 1)begin if(col_cnt 0) begin //第1个矩阵 {matrix_11, matrix_12, matrix_13} {dout_r1, dout_r1, dout_r1}; {matrix_21, matrix_22, matrix_23} {dout_r1, dout_r1, dout_r1}; {matrix_31, matrix_32, matrix_33} {dout_r2, dout_r2, dout_r2}; end else begin //剩余矩阵 {matrix_11, matrix_12, matrix_13} {matrix_12, matrix_13, dout_r1}; {matrix_21, matrix_22, matrix_23} {matrix_22, matrix_23, dout_r1}; {matrix_31, matrix_32, matrix_33} {matrix_32, matrix_33, dout_r2}; end end //------------------------------------------------------------------------- 剩余矩阵 else begin if(col_cnt 0) begin //第1个矩阵 {matrix_11, matrix_12, matrix_13} {dout_r0, dout_r0, dout_r0}; {matrix_21, matrix_22, matrix_23} {dout_r1, dout_r1, dout_r1}; {matrix_31, matrix_32, matrix_33} {dout_r2, dout_r2, dout_r2}; end else begin //剩余矩阵 {matrix_11, matrix_12, matrix_13} {matrix_12, matrix_13, dout_r0}; {matrix_21, matrix_22, matrix_23} {matrix_22, matrix_23, dout_r1}; {matrix_31, matrix_32, matrix_33} {matrix_32, matrix_33, dout_r2}; end end end endmodule module image_mean_10bit ( input wire i_clk , //时钟输入 input wire i_rst_n , //复位 input wire i_hs , input wire i_vs , input wire i_de , //图像有效显示区域 input wire [9:0] i_data , //输入RGB图像 output wire o_hs , output wire o_vs , output wire o_de , //图像有效显示区域 output wire [9:0] o_data //输出16bits的灰度图像 ); //wire or reg define reg [13:0] data_add ;//和 reg [20:0] data_mean ;//均值滤波的结果 //3x3矩阵的变量 wire [9:0] matrix_11 ; wire [9:0] matrix_12 ; wire [9:0] matrix_13 ; wire [9:0] matrix_21 ; wire [9:0] matrix_22 ; wire [9:0] matrix_23 ; wire [9:0] matrix_31 ; wire [9:0] matrix_32 ; wire [9:0] matrix_33 ; //延时相关变量 reg [1:0] hs_r ; reg [1:0] vs_r ; reg [1:0] de_r ; //main code //加法一个时钟周期 always (posedge i_clk or negedge i_rst_n) begin if(!i_rst_n) data_add14d0; else data_addmatrix_11matrix_12matrix_13 matrix_21matrix_22matrix_23 matrix_31matrix_32matrix_33; end //乘法一个时钟周期 移位相加的乘法运算 //先乘以1024再除以9 1024/9 113.77 114 7b111_0010 6432162 always (posedge i_clk or negedge i_rst_n) begin if(!i_rst_n) data_mean20d0; else data_mean(data_add6)(data_add5)(data_add4)(data_add1) ; end //视频时序延时打两拍 always (posedge i_clk or negedge i_rst_n) begin if(!i_rst_n) begin hs_r2d0; vs_r2d0; de_r2d0; end else begin hs_r{hs_r[0],i_hs}; vs_r{vs_r[0],i_vs}; de_r{de_r[0],i_de}; end end //输出视频时序 assign o_hs hs_r[1] ; assign o_vs vs_r[1] ; assign o_de de_r[1] ; assign o_data de_r[1] ? data_mean[19:10] : 10d0 ;//截取16bits作为输出 //3x3矩阵生成模块 matrix_3x3_10bit #( . COL (640 ) , . ROW (480 ) )matrix_3x3_10bit_inst ( . clk (i_clk ) , . rst_n (i_rst_n ) , . valid_in (i_de ) , . din (i_data ) , . matrix_11 (matrix_11 ) , //3x3矩阵内的9个数 . matrix_12 (matrix_12 ) , . matrix_13 (matrix_13 ) , . matrix_21 (matrix_21 ) , . matrix_22 (matrix_22 ) , . matrix_23 (matrix_23 ) , . matrix_31 (matrix_31 ) , . matrix_32 (matrix_32 ) , . matrix_33 (matrix_33 ) ); endmodule module guassian_filter_proc #( parameter [10:0] IMG_HDISP 11d1280, // 640*480 parameter [10:0] IMG_VDISP 11d720 ) ( input wire clk , input wire rst_n , // Image data prepared to be processed input wire per_img_vsync , // Prepared Image data vsync valid signal input wire per_img_href , // Prepared Image data href vaild signal input wire [7:0] per_img_gray , // Prepared Image brightness input // Image data has been processed output reg post_img_vsync , // processed Image data vsync valid signal output reg post_img_href , // processed Image data href vaild signal output reg [7:0] post_img_gray // processed Image brightness output ); //---------------------------------------------------------------------- // Generate 8Bit 3X3 Matrix wire matrix_img_vsync; wire matrix_img_href; wire [7:0] matrix_p11; wire [7:0] matrix_p12; wire [7:0] matrix_p13; wire [7:0] matrix_p21; wire [7:0] matrix_p22; wire [7:0] matrix_p23; wire [7:0] matrix_p31; wire [7:0] matrix_p32; wire [7:0] matrix_p33; // 注意此处需要例化8bit 3×3矩阵模块 matrix_generate_3x3_8bit matrix_generate_3x3_8bit u_matrix_generate_3x3_8bit ( //global clock .clk (clk), //cmos video pixel clock .rst_n (rst_n), //global reset //Image data prepred to be processd .per_frame_vsync (per_img_vsync), //Prepared Image data vsync valid signal .per_frame_href (per_img_href), //Prepared Image data href vaild signal .per_frame_clken (per_img_href), //Prepared Image data output/capture enable clock .per_img_y (per_img_gray), //Prepared Image brightness input //Image data has been processd .matrix_frame_vsync (matrix_img_vsync), //Processed Image data vsync valid signal .matrix_frame_href (matrix_img_href), //Processed Image data href vaild signal .matrix_frame_clken (), //Processed Image data output/capture enable clock .matrix_p11(matrix_p11), .matrix_p12(matrix_p12), .matrix_p13(matrix_p13), //3X3 Matrix output .matrix_p21(matrix_p21), .matrix_p22(matrix_p22), .matrix_p23(matrix_p23), .matrix_p31(matrix_p31), .matrix_p32(matrix_p32), .matrix_p33(matrix_p33) ); //---------------------------------------------------------------------- //////////1拍 //---------------------------------------------------------------------- // [g11,g12,g13] [109,115,109] // g [g21,g22,g23] [115,122,115] // [g31,g32,g33] [109,115,109] localparam g11 8d100; localparam g12 8d119; localparam g13 8d100; localparam g21 8d119; localparam g22 8d142; localparam g23 8d119; localparam g31 8d100; localparam g32 8d119; localparam g33 8d100; reg [16:0] mult_g11; reg [16:0] mult_g21; reg [16:0] mult_g31; reg [16:0] mult_g12; reg [16:0] mult_g22; reg [16:0] mult_g32; reg [16:0] mult_g13; reg [16:0] mult_g23; reg [16:0] mult_g33; always (posedge clk) begin mult_g11 matrix_p11 * g11; mult_g21 matrix_p12 * g21; mult_g31 matrix_p13 * g31; mult_g12 matrix_p21 * g12; mult_g22 matrix_p22 * g22; mult_g32 matrix_p23 * g32; mult_g13 matrix_p31 * g13; mult_g23 matrix_p32 * g23; mult_g33 matrix_p33 * g33; end //---------------------------------------------------------------------- //////////2拍 reg [18:0] weight1; reg [18:0] weight2; reg [18:0] weight3; always (posedge clk) begin weight1 mult_g11 mult_g21 mult_g31; weight2 mult_g12 mult_g22 mult_g32; weight3 mult_g13 mult_g23 mult_g33; end //---------------------------------------------------------------------- /////////////3拍////////////////////////// reg [18:0] weight_sum; always (posedge clk) begin weight_sum weight1 weight2 weight3; end //////////////////4拍///////////////////////////////// reg [7:0] sum; always (posedge clk) begin if(weight_sum[18:10] 255) post_img_gray8d255; else post_img_gray weight_sum[17:10]; end //---------------------------------------------------------------------- // lag 3 clocks signal sync localparam C_CLK_LATENCY 4; reg [C_CLK_LATENCY-1:0] matrix_img_vsync_r1; reg [C_CLK_LATENCY-1:0] matrix_img_href_r1; always (posedge clk or negedge rst_n) begin if(!rst_n) begin matrix_img_vsync_r1 {C_CLK_LATENCY{1b0}}; matrix_img_href_r1 {C_CLK_LATENCY{1b0}}; end else begin matrix_img_vsync_r1 {matrix_img_vsync_r1[C_CLK_LATENCY-2:0],matrix_img_vsync}; matrix_img_href_r1 {matrix_img_href_r1[C_CLK_LATENCY-2:0],matrix_img_href}; end end always (posedge clk or negedge rst_n) begin if(!rst_n) begin post_img_vsync 1b0; post_img_href 1b0; end else begin post_img_vsync matrix_img_vsync_r1[2]; post_img_href matrix_img_href_r1[2]; end end endmodule使用提示matrix_3x3_10bit需要例化 FIFO IP FIFO_SC_10bit_Top位宽 10bit深度等于图像行宽度guassian_filter_proc依赖 8bit 版本 3×3 矩阵模块matrix_generate_3x3_8bit均值滤波流水线 2 拍高斯滤波流水线 4 拍输出时序直接使用模块输出o_hs/o_vs/o_de禁止复用输入时序信号彩色场景用法RGB 转 YCbCrY 通道送入滤波模块Cb/Cr 通道做对应拍数延时和输出像素时序对齐。
返回列表