图片提取文字功能很神奇?Java几行代码搞定它!
阅读本文大概需要 4 分钟。
来自:blog.csdn.net/weixin_44671737/article/details/110000864
摘要
一、tesseract-ocr介绍
OCR engine - libtesseract and a command line program - tesseract.
输入(一张图片) 有用信息提取(比如一个图片上只有一个字,那其他留白的是无用,这个字上每个色素是有效的并且相关) 找出文字/线条 字符分类集 输入与分类集对比找出最接近的 输出识别结果
二、安装tesseract
三、使用命令行
tesseract 1606150081.png 1606150081 -l chi_sim
tesseract D:\company\ruigushop\spring-2s\test.png stdout -l chi_sim
四、程序实现(Python)
上传图片 -> 保存 ->对上传的图片执行tesseract指令->获取识别结果
# coding=utf-8
from flask import Flask, request
import os
import datetime
import time
app = Flask(__name__)
def get_time_stamp():
times = datetime.datetime.now().strftime('%Y-%m-%d %H:%M:%S')
array = time.strptime(times, "%Y-%m-%d %H:%M:%S")
time_stamp = int(time.mktime(array))
return time_stamp
@app.route('/image/extract', methods=['POST'])
def pure_rec():
file = request.files.get('file')
ts = str(get_time_stamp())
up_path = os.path.join(ts + file.filename)
file.save(up_path)
cmd = "tesseract "+up_path+" " + ts + " -l chi_sim"
print(cmd)
os.system(cmd)
with open(ts+".txt", 'r+', encoding="utf-8") as f:
result = f.read()
return result
if __name__ == '__main__':
app.run(debug=True)
五、程序实现(Java)
package com.lbh.web.controller;
/*
* Copyright@lbhbinhao@163.com
* Author:liubinhao
* Date:2020/11/23
* ++++ ______ @author liubinhao ______ ______
* +++/ /| / /| / /|
* +/_____/ | /_____/ | /_____/ |
* | | | | | | | | |
* | | | | | |________| | |
* | | | | | / | | |
* | | | | |/___________| | |
* | | |___________________ | |____________| | |
* | | / / | | | | | | |
* | |/ _________________/ / | | / | | /
* |_________________________|/b |_____|/ |_____|/
*/
import org.springframework.web.bind.annotation.PostMapping;
import org.springframework.web.bind.annotation.RequestParam;
import org.springframework.web.bind.annotation.RestController;
import org.springframework.web.multipart.MultipartFile;
import java.io.BufferedReader;
import java.io.File;
import java.io.IOException;
import java.io.InputStreamReader;
@RestController
public class LiteralExtractController {
@PostMapping("/image/extract")
public String reg(@RequestParam("file")MultipartFile file) throws IOException {
String result = "";
String filename = file.getOriginalFilename();
File save = new File(System.getProperty("user.dir")+"\\"+filename);
if (!save.exists()){
save.createNewFile();
}
file.transferTo(save);
String cmd = String.format("tesseract %s stdout -l %s",System.getProperty("user.dir")+"\\"+filename,"chi_sim");
result = cmd(cmd);
return result;
}
public static String cmd(String cmd) {
BufferedReader br = null;
try {
Process p = Runtime.getRuntime().exec(cmd);
br = new BufferedReader(new InputStreamReader(p.getInputStream()));
String line = null;
StringBuilder sb = new StringBuilder();
while ((line = br.readLine()) != null) {
sb.append(line + "\n");
}
return sb.toString();
} catch (Exception e) {
e.printStackTrace();
}
finally
{
if (br != null)
{
try {
br.close();
} catch (Exception e) {
e.printStackTrace();
}
}
}
return null;
}
}
六、实验测试
七、总结
推荐阅读:
王思聪花100万装了台电脑,跑分“全球第四”!看完装机清单我人傻了
最近面试BAT,整理一份面试资料《Java面试BATJ通关手册》,覆盖了Java核心技术、JVM、Java并发、SSM、微服务、数据库、数据结构等等。
朕已阅