java图片文字识别技术

代码参考:http://blog.youkuaiyun.com/ycb1689/article/details/8520954

其实我觉得识别率不是很高,至少没有达到95%,而且图片稍微做点模糊处理就识别不了了,但是够我使用的了。

下载相关文件,http://pan.baidu.com/s/1jIi9rPc

主要有:

(1)tesseract-ocr-setup-3.01-1.exe(需要安装,安装很简单,一直下一步就可以了)

(2)jai_imageio-1.1-alpha.jar 和 swingx-1.6.1.jar(放到项目中)

(3)chi_sim.traineddata(中文语言包,里面也进行了压缩,解压后把chi_sim.traineddata放到testeract安装目录下的tessdata目录下),如下图:



接下来就是编码了,直接参考人家的了:

ImageIOHelper类

import java.awt.image.BufferedImage;  
import java.io.File;  
import java.io.IOException;  
import java.util.Iterator;  
import java.util.Locale;  
  
import javax.imageio.IIOImage;  
import javax.imageio.ImageIO;  
import javax.imageio.ImageReader;  
import javax.imageio.ImageWriteParam;  
import javax.imageio.ImageWriter;  
import javax.imageio.metadata.IIOMetadata;  
import javax.imageio.stream.ImageInputStream;  
import javax.imageio.stream.ImageOutputStream;  
  
import com.sun.media.imageio.plugins.tiff.TIFFImageWriteParam;  
  
public class ImageIOHelper {    
        
    public static File createImage(File imageFile, String imageFormat) {    
        File tempFile = null;    
        try {    
            Iterator readers = ImageIO.getImageReadersByFormatName(imageFormat);    
            ImageReader reader = (ImageReader)readers.next();    
            
            ImageInputStream iis = ImageIO.createImageInputStream(imageFile);    
            reader.setInput(iis);    
            //Read the stream metadata    
            IIOMetadata streamMetadata = reader.getStreamMetadata();    
                
            //Set up the writeParam    
            TIFFImageWriteParam tiffWriteParam = new TIFFImageWriteParam(Locale.CHINESE);    
            tiffWriteParam.setCompressionMode(ImageWriteParam.MODE_DISABLED);    
                
            //Get tif writer and set output to file    
            Iterator writers = ImageIO.getImageWritersByFormatName("tiff");    
            ImageWriter writer = (ImageWriter)writers.next();    
                
            BufferedImage bi = reader.read(0);    
            IIOImage image = new IIOImage(bi,null,reader.getImageMetadata(0));    
            tempFile = tempImageFile(imageFile);    
            ImageOutputStream ios = ImageIO.createImageOutputStream(tempFile);    
            writer.setOutput(ios);    
            writer.write(streamMetadata, image, tiffWriteParam);    
            ios.close();    
                
            writer.dispose();    
            reader.dispose();    
                
        } catch (IOException e) {    
            e.printStackTrace();    
        }    
        return tempFile;    
    }    
    
    private static File tempImageFile(File imageFile) {    
        String path = imageFile.getPath();    
        StringBuffer strB = new StringBuffer(path);    
        strB.insert(path.lastIndexOf('.'),0);    
        return new File(strB.toString().replaceFirst("(?<=//.)(//w+)$", "tif"));    
    }    
    
}  

OCR类:

public class OCR {    
    private final String LANG_OPTION = "-l";  //英文字母小写l,并非数字1    
    private final String EOL = System.getProperty("line.separator");    
    private String tessPath = "<span style="color: rgb(0, 0, 255); font-family: Consolas, "Courier New", Courier, mono, serif; background-color: rgb(248, 248, 248);">C://Program Files (x86)//Tesseract-OCR</span>"; //注:如果不是默认安装,需要修改这个路径   
    //private String tessPath = new File("tesseract").getAbsolutePath();    
        
    public String recognizeText(File imageFile,String imageFormat)throws Exception{    
        File tempImage = ImageIOHelper.createImage(imageFile,imageFormat);    
        File outputFile = new File(imageFile.getParentFile(),"output");    
        StringBuffer strB = new StringBuffer();    
        List cmd = new ArrayList();    
        if(OS.isWindowsXP()){    
            cmd.add(tessPath+"//tesseract");    
        }else if(OS.isLinux()){    
            cmd.add("tesseract");    
        }else{    
            cmd.add(tessPath+"//tesseract");    
        }    
        cmd.add("");    
        cmd.add(outputFile.getName());    
        cmd.add(LANG_OPTION);    
        cmd.add("chi_sim");    
        //cmd.add("eng");    
            
        ProcessBuilder pb = new ProcessBuilder();    
        pb.directory(imageFile.getParentFile());    
            
        cmd.set(1, tempImage.getName());    
        pb.command(cmd);    
        pb.redirectErrorStream(true);    
            
        Process process = pb.start();    
        //tesseract.exe 1.jpg 1 -l chi_sim    
        int w = process.waitFor();    
            
        //删除临时正在工作文件    
        tempImage.delete();    
            
        if(w==0){    
            BufferedReader in = new BufferedReader(new InputStreamReader(new FileInputStream(outputFile.getAbsolutePath()+".txt"),"UTF-8"));    
                
            String str;    
            while((str = in.readLine())!=null){    
                strB.append(str).append(EOL);    
            }    
            in.close();    
        }else{    
            String msg;    
            switch(w){    
                case 1:    
                    msg = "Errors accessing files.There may be spaces in your image's filename.";    
                    break;    
                case 29:    
                    msg = "Cannot recongnize the image or its selected region.";    
                    break;    
                case 31:    
                    msg = "Unsupported image format.";    
                    break;    
                default:    
                    msg = "Errors occurred.";    
            }    
            tempImage.delete();    
            throw new RuntimeException(msg);    
        }    
        new File(outputFile.getAbsolutePath()+".txt").delete();    
        return strB.toString();    
    }    
}    


OcrTest。类

import java.io.File;  
import java.io.IOException;  
  
public class OcrTest {  
  
 public static void main(String[] args) {  
        String path = "E://test11.png";  //图片路径     
        System.out.println("ORC Test Begin......");  
        try {       
            String valCode = new OCR().recognizeText(new File(path), "png");       
            System.out.println(valCode);       
        } catch (IOException e) {       
            e.printStackTrace();       
        } catch (Exception e) {    
            e.printStackTrace();    
        }         
        System.out.println("ORC Test End......");  
    }    
  
}  


好了,自己试试吧



ImageComparerUI——基于Java语言实现的相似图像识别,基于直方图比较算法。 import java.awt.BorderLayout; import java.awt.Color; import java.awt.Dimension; import java.awt.FlowLayout; import java.awt.Font; import java.awt.Graphics; import java.awt.Graphics2D; import java.awt.Image; import java.awt.MediaTracker; import java.awt.event.ActionEvent; import java.awt.event.ActionListener; import java.awt.image.BufferedImage; import java.io.File; import java.io.IOException; import javax.imageio.ImageIO; import javax.swing.JButton; import javax.swing.JComponent; import javax.swing.JFileChooser; import javax.swing.JFrame; import javax.swing.JPanel; public class ImageComparerUI extends JComponent implements ActionListener { /** * */ private static final long serialVersionUID = 1L; private JButton browseBtn; private JButton histogramBtn; private JButton compareBtn; private Dimension mySize; // image operator private MediaTracker tracker; private BufferedImage sourceImage; private BufferedImage candidateImage; private double simility; // command constants public final static String BROWSE_CMD = "Browse..."; public final static String HISTOGRAM_CMD = "Histogram Bins"; public final static String COMPARE_CMD = "Compare Result"; public ImageComparerUI() { JPanel btnPanel = new JPanel(); btnPanel.setLayout(new FlowLayout(FlowLayout.LEFT)); browseBtn = new JButton("Browse..."); histogramBtn = new JButton("Histogram Bins"); compareBtn = new JButton("Compare Result"); // buttons btnPanel.add(browseBtn); btnPanel.add(histogramBtn); btnPanel.add(compareBtn); // setup listener... browseBtn.addActionListener(this); histogramBtn.addActionListener(this); compareBtn.addActionListener(this); mySize = new Dimension(620, 500); JFrame demoUI = new JFrame("Similiar Image Finder"); demoUI.getContentPane().setLayout(new BorderLayout()); demoUI.getContentPane().add(this, BorderLayout.CENTER); demoUI.getContentPane().add(btnPanel, BorderLayout.SOUTH); demoUI.setDefaultCloseOperation(JFrame.EXIT_ON_CLOSE); demoUI.pack(); demoUI.setVisible(true); } public void paint(Graphics g) { Graphics2D g2 = (Graphics2D) g; if(sourceImage != null) { Image scaledImage = sourceImage.getScaledInstance(300, 300, Image.SCALE_FAST); g2.drawImage(scaledImage, 0, 0, 300, 300, null); } if(candidateImage != null) { Image scaledImage = candidateImage.getScaledInstance(300, 330, Image.SCALE_FAST); g2.drawImage(scaledImage, 310, 0, 300, 300, null); } // display compare result info here Font myFont = new Font("Serif", Font.BOLD, 16); g2.setFont(myFont); g2.setPaint(Color.RED); g2.drawString("The degree of similarity : " + simility, 50, 350); } public void actionPerformed(ActionEvent e) { if(BROWSE_CMD.equals(e.getActionCommand())) { JFileChooser chooser = new JFileChooser(); chooser.showOpenDialog(null); File f = chooser.getSelectedFile(); BufferedImage bImage = null; if(f == null) return; try { bImage = ImageIO.read(f); } catch (IOException e1) { e1.printStackTrace(); } tracker = new MediaTracker(this); tracker.addImage(bImage, 1); // blocked 10 seconds to load the image data try { if (!tracker.waitForID(1, 10000)) { System.out.println("Load error."); System.exit(1); }// end if } catch (InterruptedException ine) { ine.printStackTrace(); System.exit(1); } // end catch if(sourceImage == null) { sourceImage = bImage; }else if(candidateImage == null) { candidateImage = bImage; } else { sourceImage = null; candidateImage = null; }
评论
添加红包

请填写红包祝福语或标题

红包个数最小为10个

红包金额最低5元

当前余额3.43前往充值 >
需支付:10.00
成就一亿技术人!
领取后你会自动成为博主和红包主的粉丝 规则
hope_wisdom
发出的红包
实付
使用余额支付
点击重新获取
扫码支付
钱包余额 0

抵扣说明:

1.余额是钱包充值的虚拟货币,按照1:1的比例进行支付金额的抵扣。
2.余额无法直接购买下载,可以购买VIP、付费专栏及课程。

余额充值