diff --git a/code/crane/crane-api/crane-test/src/main/java/com/crane/test/1-122700001-OCR-LF-C01.jpg b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/1-122700001-OCR-LF-C01.jpg new file mode 100644 index 00000000..3e1428a2 Binary files /dev/null and b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/1-122700001-OCR-LF-C01.jpg differ diff --git a/code/crane/crane-api/crane-test/src/main/java/com/crane/test/1-122720001-OCR-AH-A01.jpg b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/1-122720001-OCR-AH-A01.jpg new file mode 100644 index 00000000..d559c093 Binary files /dev/null and b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/1-122720001-OCR-AH-A01.jpg differ diff --git a/code/crane/crane-api/crane-test/src/main/java/com/crane/test/20dryvan_lg.png b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/20dryvan_lg.png new file mode 100644 index 00000000..13bc3160 Binary files /dev/null and b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/20dryvan_lg.png differ diff --git a/code/crane/crane-api/crane-test/src/main/java/com/crane/test/Test.java b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/Test.java index c003bff7..fd4c2712 100644 --- a/code/crane/crane-api/crane-test/src/main/java/com/crane/test/Test.java +++ b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/Test.java @@ -11,19 +11,30 @@ import java.util.Base64; public class Test{ public static void main(String[] args) throws IOException { - String API_URL = "http://localhost:8080/ocr"; - String imagePath = "./demo.jpg"; + String API_URL = "http://localhost:8866/ocr"; + String imagePath = "D:\\workspace\\code\\crane\\crane-api\\crane-test\\src\\main\\java\\com\\crane\\test\\tempImageNYfUPw-750x563.webp"; File file = new File(imagePath); + long base64StartTime = System.nanoTime(); + byte[] fileContent = java.nio.file.Files.readAllBytes(file.toPath()); String base64Image = Base64.getEncoder().encodeToString(fileContent); + long base64EndTime = System.nanoTime(); + long durationMs1 = (base64EndTime - base64StartTime) / 1_000_000; // 毫秒 + System.out.println("图片转 BASE64 耗时: " + durationMs1 + " ms"); ObjectMapper objectMapper = new ObjectMapper(); ObjectNode payload = objectMapper.createObjectNode(); payload.put("file", base64Image); payload.put("fileType", 1); - OkHttpClient client = new OkHttpClient(); + + OkHttpClient client = new OkHttpClient.Builder() + .connectTimeout(300, java.util.concurrent.TimeUnit.SECONDS) // 连接超时,建议20-30秒 + .writeTimeout(300, java.util.concurrent.TimeUnit.SECONDS) // 写请求体超时 + .readTimeout(300, java.util.concurrent.TimeUnit.SECONDS) // 读取响应超时,OCR服务响应较慢可设置长点 + .build(); + MediaType JSON = MediaType.get("application/json; charset=utf-8"); RequestBody body = RequestBody.create(JSON, payload.toString()); @@ -32,25 +43,35 @@ public class Test{ .post(body) .build(); + long startTime = System.nanoTime(); + try (Response response = client.newCall(request).execute()) { + long endTime = System.nanoTime(); + long durationMs = (endTime - startTime) / 1_000_000; // 毫秒 + + System.out.println("⏱️ 请求耗时: " + durationMs + " ms"); + if (response.isSuccessful()) { String responseBody = response.body().string(); + JsonNode root = objectMapper.readTree(responseBody); JsonNode result = root.get("result"); JsonNode ocrResults = result.get("ocrResults"); + for (int i = 0; i < ocrResults.size(); i++) { JsonNode item = ocrResults.get(i); - JsonNode prunedResult = item.get("prunedResult"); - System.out.println("Pruned Result [" + i + "]: " + prunedResult.toString()); - - String ocrImageBase64 = item.get("ocrImage").asText(); - byte[] ocrImageBytes = Base64.getDecoder().decode(ocrImageBase64); - String ocrImgPath = "ocr_result_" + i + ".jpg"; - try (FileOutputStream fos = new FileOutputStream(ocrImgPath)) { - fos.write(ocrImageBytes); - System.out.println("Saved OCR image to: " + ocrImgPath); + System.out.println(prunedResult); + // 获取 rec_texts 并打印 + JsonNode recTexts = prunedResult.get("rec_texts"); + if (recTexts != null && recTexts.isArray()) { + System.out.println("📦 rec_texts:"); + for (JsonNode text : recTexts) { + System.out.println(" ➤ " + text.asText()); + } + } else { + System.out.println("❗ 未检测到 rec_texts 字段"); } } } else { diff --git a/code/crane/crane-api/crane-test/src/main/java/com/crane/test/demo.jpg b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/demo.jpg new file mode 100644 index 00000000..cc7009e6 Binary files /dev/null and b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/demo.jpg differ diff --git a/code/crane/crane-api/crane-test/src/main/java/com/crane/test/image.png b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/image.png new file mode 100644 index 00000000..3865858b Binary files /dev/null and b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/image.png differ diff --git a/code/crane/crane-api/crane-test/src/main/java/com/crane/test/images.jpg b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/images.jpg new file mode 100644 index 00000000..0fb5d75d Binary files /dev/null and b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/images.jpg differ diff --git a/code/crane/crane-api/crane-test/src/main/java/com/crane/test/namorni-preprava-namorni-doprava-kamionova-doprava-kontejneru-2-1024x768.jpg b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/namorni-preprava-namorni-doprava-kamionova-doprava-kontejneru-2-1024x768.jpg new file mode 100644 index 00000000..943e8316 Binary files /dev/null and b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/namorni-preprava-namorni-doprava-kamionova-doprava-kontejneru-2-1024x768.jpg differ diff --git a/code/crane/crane-api/crane-test/src/main/java/com/crane/test/tempImageNYfUPw-750x563.webp b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/tempImageNYfUPw-750x563.webp new file mode 100644 index 00000000..4faaf96a Binary files /dev/null and b/code/crane/crane-api/crane-test/src/main/java/com/crane/test/tempImageNYfUPw-750x563.webp differ diff --git a/code/crane/crane-api/ocr_result_0.jpg b/code/crane/crane-api/ocr_result_0.jpg new file mode 100644 index 00000000..43c931bd Binary files /dev/null and b/code/crane/crane-api/ocr_result_0.jpg differ diff --git a/notes/后端开发/paddlepaddle/paddleocr.md b/notes/后端开发/paddlepaddle/paddleocr.md new file mode 100644 index 00000000..e69de29b diff --git a/notes/后端开发/paddlepaddle/paddlex.md b/notes/后端开发/paddlepaddle/paddlex.md new file mode 100644 index 00000000..41abf896 --- /dev/null +++ b/notes/后端开发/paddlepaddle/paddlex.md @@ -0,0 +1,327 @@ +> 注:该项目背景为 《识别集装箱号》 + + + +# 相关网站 + +相关网站: + +- github:https://github.com/PaddlePaddle/PaddleX +- 文档:https://paddlepaddle.github.io/PaddleX/latest/index.html + +# 1. 介绍 + +PaddleX 3.0 是基于飞桨框架构建的低代码开发工具,它集成了众多**开箱即用的预训练模型**,可以实现模型从训练到推理的**全流程开发**,支持国内外**多款主流硬件**,助力AI 开发者进行产业实践。 + +作用: + +- 部署服务端,方便客户端调用 + + + +# 2. 安装 + +注意:安装 PaddleX 前需要安装 PaddlePaddle + +## 2.1 安装 PaddlePaddle + +```bash +# CPU 版本 +python -m pip install paddlepaddle==3.0.0 -i https://www.paddlepaddle.org.cn/packages/stable/cpu/ + +# GPU 版本,需显卡驱动程序版本 ≥450.80.02(Linux)或 ≥452.39(Windows) +python -m pip install paddlepaddle-gpu==3.0.0 -i https://www.paddlepaddle.org.cn/packages/stable/cu118/ + +# GPU 版本,需显卡驱动程序版本 ≥550.54.14(Linux)或 ≥550.54.14(Windows) +python -m pip install paddlepaddle-gpu==3.0.0 -i https://www.paddlepaddle.org.cn/packages/stable/cu126/ +``` + +> ❗无需关注物理机上的 CUDA 版本,只需关注显卡驱动程序版本。更多飞桨 Wheel 版本信息,请参考[飞桨官网](https://www.paddlepaddle.org.cn/install/quick?docurl=/documentation./docs/zh/install/pip/linux-pip.html)。 + +## 2.2 安装 PaddleX + +### 2.2.1 命令行安装 + +```bash +pip install "paddlex[base]" +``` + + + +# 3. 使用 + +## 3.1 部署项目 + +> 参考文档:https://paddlepaddle.github.io/PaddleX/latest/pipeline_deploy/serving.html#12 + +### 3.1.1 安装服务化部署插件 + +执行如下命令,安装服务化部署插件: + +```bash +paddlex --install serving +``` + +### 3.1.2 生成配置文件 + +使用相关命令,其中 OCR 当前为通用 OCR 产线,需要不同的产线就修改不同的名称即可。 + +```bash +paddlex --get_pipeline_config OCR --save_path ./OCR.yaml +``` + +生成成功会提示 + +```bash +The pipeline config has been saved to: OCR.yaml +``` + +生成的默认配置如下: + +```yaml +pipeline_name: OCR + +text_type: general + +use_doc_preprocessor: True +use_textline_orientation: True + +SubPipelines: + DocPreprocessor: + pipeline_name: doc_preprocessor + use_doc_orientation_classify: True + use_doc_unwarping: True + SubModules: + DocOrientationClassify: + module_name: doc_text_orientation + model_name: PP-LCNet_x1_0_doc_ori + model_dir: null + DocUnwarping: + module_name: image_unwarping + model_name: UVDoc + model_dir: null + +SubModules: + TextDetection: + module_name: text_detection + model_name: PP-OCRv5_server_det + model_dir: null + limit_side_len: 64 + limit_type: min + max_side_limit: 4000 + thresh: 0.3 + box_thresh: 0.6 + unclip_ratio: 1.5 + TextLineOrientation: + module_name: textline_orientation + model_name: PP-LCNet_x1_0_textline_ori + model_dir: null + batch_size: 6 + TextRecognition: + module_name: text_recognition + model_name: PP-OCRv5_server_rec + model_dir: null + batch_size: 6 + score_thresh: 0.0 +``` + +**修改配置文件** + +注意将训练好的模型路径配置 `./pipeline/rec_inference` 和 `./pipeline/det_inference` + +```yaml +# 整体 pipeline 名称,用于识别产线名称 +pipeline_name: OCR + +# 文本类型,可选 general(通用)或 others,决定一些后处理策略 +text_type: general + +# 是否使用文档预处理模块(方向分类 + 扭曲矫正) +# 如无严重倾斜/扫描件,建议设为 False,可显著提升速度 +use_doc_preprocessor: False + +# 是否对检测出的文本行进行角度分类(如 0° / 90° / 180°) +# mobile 模型一般建议关闭 +use_textline_orientation: False + +# 正式的 OCR 主流程模块 +SubModules: + + # 文本检测模块(通常是基于 DB 的检测器) + TextDetection: + module_name: text_detection + model_name: PP-OCRv5_server_det # 使用的是 server 版大模型,精度高但速度慢 + model_dir: ./pipeline/det_inference # 本地模型文件夹路径(需包含 model.pdmodel 等) + limit_side_len: 64 # 输入图最短边 resize 到此尺寸(可适当调大提高效果) + limit_type: min + max_side_limit: 4000 # 最长边最大不超过此值 + thresh: 0.3 # 检测 binarize 阈值 + box_thresh: 0.6 # 文本框置信度阈值 + unclip_ratio: 1.5 # DB 算法的后处理参数,控制框扩大比例 + + # 文本识别模块(通常是 CRNN + CTC 或 SVTR 模型) + TextRecognition: + module_name: text_recognition + model_name: PP-OCRv5_server_rec # 同样使用的是 server 版识别模型 + model_dir: ./pipeline/rec_inference + batch_size: 6 # 一次识别图块的数量,适当调大可提高 GPU 利用率 + score_thresh: 0.0 # 识别结果置信度下限,低于此不输出 +``` + +### 3.1.3 运行服务器 + +通过 PaddleX CLI 运行服务器: + +```bash +paddlex --serve --pipeline {产线名称或产线配置文件路径} [{其他命令行选项}] +``` + +参考命令: + +```bash +paddlex --serve --pipeline .\OCR.yaml --device gpu:0 --port 8866 +``` + +可以看到类似以下展示的信息,即代表成功 + +``` +INFO: Started server process [63108] +INFO: Waiting for application startup. +INFO: Application startup complete. +INFO: Uvicorn running on http://0.0.0.0:8080 (Press CTRL+C to quit) +``` + +### 3.1.4 调用服务 + +对于服务提供的主要操作: + +- HTTP请求方法为POST。 +- 请求体和响应体均为JSON数据(JSON对象)。 +- 当请求处理成功时,响应状态码为`200`,响应体的属性如下: + +| 名称 | 类型 | 含义 | +| :---------- | :-------- | :---------------------------- | +| `logId` | `string` | 请求的UUID。 | +| `errorCode` | `integer` | 错误码。固定为`0`。 | +| `errorMsg` | `string` | 错误说明。固定为`"Success"`。 | +| `result` | `object` | 操作结果。 | + +- 当请求处理未成功时,响应体的属性如下: + +| 名称 | 类型 | 含义 | +| :---------- | :-------- | :------------------------- | +| `logId` | `string` | 请求的UUID。 | +| `errorCode` | `integer` | 错误码。与响应状态码相同。 | +| `errorMsg` | `string` | 错误说明。 | + +服务提供的主要操作如下: + +- **`infer`** + +获取图像OCR结果。 + +``` +POST /ocr +``` + +- 请求体的属性如下: + +| 名称 | 类型 | 含义 | 是否必填 | +| :-------------------------- | :----------------- | :----------------------------------------------------------- | :------- | +| `file` | `string` | 服务器可访问的图像文件或PDF文件的URL,或上述类型文件内容的Base64编码结果。默认对于超过10页的PDF文件,只有前10页的内容会被处理。 要解除页数限制,请在产线配置文件中添加以下配置:`Serving: extra: max_num_input_imgs: null ` | 是 | +| `fileType` | `integer` | `null` | 文件类型。`0`表示PDF文件,`1`表示图像文件。若请求体无此属性,则将根据URL推断文件类型。 | 否 | +| `visualize` | `boolean` | `null` | 是否返回可视化结果图以及处理过程中的中间图像等。传入 `true`:返回图像。传入 `false`:不返回图像。若请求体中未提供该参数或传入 `null`:遵循产线配置文件`Serving.visualize` 的设置。 例如,在产线配置文件中添加如下字段: `Serving: visualize: False `将默认不返回图像,通过请求体中的`visualize`参数可以覆盖默认行为。如果请求体和配置文件中均未设置(或请求体传入`null`、配置文件中未设置),则默认返回图像。 | 否 | +| `useDocOrientationClassify` | `boolean` | `null` | 请参阅产线对象中 `predict` 方法的 `use_doc_orientation_classify` 参数相关说明。 | 否 | +| `useDocUnwarping` | `boolean` | `null` | 请参阅产线对象中 `predict` 方法的 `use_doc_unwarping` 参数相关说明。 | 否 | +| | | | | +| `useTextlineOrientation` | `boolean` | `null` | 请参阅产线对象中 `predict` 方法的 `use_textline_orientation` 参数相关说明。 | 否 | +| `textDetLimitSideLen` | `integer` | `null` | 请参阅产线对象中 `predict` 方法的 `text_det_limit_side_len` 参数相关说明。 | 否 | +| `textDetLimitType` | `string` | `null` | 请参阅产线对象中 `predict` 方法的 `text_det_limit_type` 参数相关说明。 | 否 | +| `textDetThresh` | `number` | `null` | 请参阅产线对象中 `predict` 方法的 `text_det_thresh` 参数相关说明。 | 否 | +| `textDetBoxThresh` | `number` | `null` | 请参阅产线对象中 `predict` 方法的 `text_det_box_thresh` 参数相关说明。 | 否 | +| `textDetUnclipRatio` | `number` | `null` | 请参阅产线对象中 `predict` 方法的 `text_det_unclip_ratio` 参数相关说明。 | 否 | +| `textRecScoreThresh` | `number` | `null` | 请参阅产线对象中 `predict` 方法的 `text_rec_score_thresh` 参数相关说明。 | 否 | + +- 请求处理成功时,响应体的`result`具有如下属性: + +| 名称 | 类型 | 含义 | +| :----------- | :------- | :----------------------------------------------------------- | +| `ocrResults` | `object` | OCR结果。数组长度为1(对于图像输入)或实际处理的文档页数(对于PDF输入)。对于PDF输入,数组中的每个元素依次表示PDF文件中实际处理的每一页的结果。 | +| `dataInfo` | `object` | 输入数据信息。 | + +`ocrResults`中的每个元素为一个`object`,具有如下属性: + +| 名称 | 类型 | 含义 | +| :---------------------- | :---------------- | :----------------------------------------------------------- | +| `prunedResult` | `object` | 产线对象的 `predict` 方法生成结果的 JSON 表示中 `res` 字段的简化版本,其中去除了 `input_path` 和 `page_index` 字段。 | +| `ocrImage` | `string` | `null` | OCR结果图,其中标注检测到的文本位置。图像为JPEG格式,使用Base64编码。 | +| `docPreprocessingImage` | `string` | `null` | 可视化结果图像。图像为JPEG格式,使用Base64编码。 | +| `inputImage` | `string` | `null` | 输入图像。图像为JPEG格式,使用Base64编码。 | + + + +**Java调用实例** + +```java +import okhttp3.*; +import com.fasterxml.jackson.databind.ObjectMapper; +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.node.ObjectNode; + +import java.io.File; +import java.io.FileOutputStream; +import java.io.IOException; +import java.util.Base64; + +public class Main { + public static void main(String[] args) throws IOException { + String API_URL = "http://localhost:8080/ocr"; + String imagePath = "./demo.jpg"; + + File file = new File(imagePath); + byte[] fileContent = java.nio.file.Files.readAllBytes(file.toPath()); + String base64Image = Base64.getEncoder().encodeToString(fileContent); + + ObjectMapper objectMapper = new ObjectMapper(); + ObjectNode payload = objectMapper.createObjectNode(); + payload.put("file", base64Image); + payload.put("fileType", 1); + + OkHttpClient client = new OkHttpClient(); + MediaType JSON = MediaType.get("application/json; charset=utf-8"); + RequestBody body = RequestBody.create(JSON, payload.toString()); + + Request request = new Request.Builder() + .url(API_URL) + .post(body) + .build(); + + try (Response response = client.newCall(request).execute()) { + if (response.isSuccessful()) { + String responseBody = response.body().string(); + JsonNode root = objectMapper.readTree(responseBody); + JsonNode result = root.get("result"); + + JsonNode ocrResults = result.get("ocrResults"); + for (int i = 0; i < ocrResults.size(); i++) { + JsonNode item = ocrResults.get(i); + + JsonNode prunedResult = item.get("prunedResult"); + System.out.println("Pruned Result [" + i + "]: " + prunedResult.toString()); + + String ocrImageBase64 = item.get("ocrImage").asText(); + byte[] ocrImageBytes = Base64.getDecoder().decode(ocrImageBase64); + String ocrImgPath = "ocr_result_" + i + ".jpg"; + try (FileOutputStream fos = new FileOutputStream(ocrImgPath)) { + fos.write(ocrImageBytes); + System.out.println("Saved OCR image to: " + ocrImgPath); + } + } + } else { + System.err.println("Request failed with HTTP code: " + response.code()); + } + } + } +} +``` +