基线:全量源码首提(D1 版本控制落地,含第一轮优化 B1-B4/A1/A4/缩放修复)
This commit is contained in:
@@ -0,0 +1,355 @@
|
||||
// ============================================================
|
||||
// GlmOcrClient:智谱 GLM-OCR 云端接口(版面解析 layout_parsing)
|
||||
// · POST {BaseUrl} Authorization: Bearer {ApiKey}
|
||||
// {"model":"glm-ocr","file":"data:image/png;base64,..."}
|
||||
// · 响应 layout_details[页][{label,bbox_2d(归一化),content,width,height}]
|
||||
// → label=text 的块转 OcrItem(bbox 映射回裁剪图像素坐标)
|
||||
// · 手写字识别(模型内置,无需专用参数);价格 ~0.2 元/百万 tokens
|
||||
// · 纯 C# HTTP,无需 Python;引擎切换由 OcrClient 路由
|
||||
// ============================================================
|
||||
using System;
|
||||
using System.Collections.Generic;
|
||||
using System.Drawing;
|
||||
using System.Drawing.Imaging;
|
||||
using System.IO;
|
||||
using System.Net;
|
||||
using System.Text;
|
||||
using System.Threading.Tasks;
|
||||
using System.Web.Script.Serialization;
|
||||
|
||||
namespace pdfbt
|
||||
{
|
||||
/// <summary>智谱 GLM-OCR 客户端(单实例)。</summary>
|
||||
internal sealed class GlmOcrClient
|
||||
{
|
||||
private readonly JavaScriptSerializer _json = new JavaScriptSerializer();
|
||||
private string _apiKey;
|
||||
private string _baseUrl;
|
||||
|
||||
/// <summary>状态提示(未配置Key/请求失败等),接 OcrClient.StatusChanged。</summary>
|
||||
public event Action<string> StatusChanged;
|
||||
|
||||
public GlmOcrClient()
|
||||
{
|
||||
_baseUrl = ReadAppSetting("GlmBaseUrl",
|
||||
"https://open.bigmodel.cn/api/paas/v4/layout_parsing");
|
||||
try
|
||||
{
|
||||
// net48 下显式启用 TLS1.2(大模型 API 必须)
|
||||
ServicePointManager.SecurityProtocol =
|
||||
ServicePointManager.SecurityProtocol | SecurityProtocolType.Tls12;
|
||||
}
|
||||
catch { }
|
||||
}
|
||||
|
||||
/// <summary>Key 由 OcrClient 从 baidu_ocr.json 的 glm_api_key 注入(热更新)。</summary>
|
||||
public void SetApiKey(string key)
|
||||
{
|
||||
_apiKey = key == null ? "" : key.Trim();
|
||||
}
|
||||
|
||||
public bool IsConfigured
|
||||
{
|
||||
get { return !string.IsNullOrWhiteSpace(_apiKey); }
|
||||
}
|
||||
|
||||
private static string ReadAppSetting(string key, string def)
|
||||
{
|
||||
try
|
||||
{
|
||||
var v = System.Configuration.ConfigurationManager.AppSettings[key];
|
||||
return string.IsNullOrWhiteSpace(v) ? def : v;
|
||||
}
|
||||
catch { return def; }
|
||||
}
|
||||
|
||||
/// <summary>识别一块位图 → 文本块列表(带 bbox 像素坐标)。</summary>
|
||||
public Task<List<OcrItem>> RecognizeAsync(Bitmap crop, int timeoutMs)
|
||||
{
|
||||
if (crop == null || crop.Width <= 0 || crop.Height <= 0)
|
||||
return Task.FromResult(new List<OcrItem>());
|
||||
|
||||
if (!IsConfigured)
|
||||
{
|
||||
Notify("GLM OCR 未配置 API Key(OCR 设置窗口里填智谱 API Key)");
|
||||
var tcs0 = new TaskCompletionSource<List<OcrItem>>();
|
||||
tcs0.SetException(new Exception(
|
||||
"智谱 GLM-OCR 未配置 API Key。\r\n请到 open.bigmodel.cn 获取 Key," +
|
||||
"在 OCR 设置窗口选择引擎\"智谱 GLM-OCR\"并填入 Key 后保存。"));
|
||||
return tcs0.Task;
|
||||
}
|
||||
|
||||
return Task.Factory.StartNew(() =>
|
||||
{
|
||||
var sw = System.Diagnostics.Stopwatch.StartNew();
|
||||
try
|
||||
{
|
||||
var result = RequestAndParse(crop, timeoutMs);
|
||||
Notify("智谱GLM-OCR·" + result.Count + "条 " + sw.ElapsedMilliseconds + "ms");
|
||||
return result;
|
||||
}
|
||||
catch
|
||||
{
|
||||
throw;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
private List<OcrItem> RequestAndParse(Bitmap crop, int timeoutMs)
|
||||
{
|
||||
byte[] png;
|
||||
using (var ms = new MemoryStream())
|
||||
{
|
||||
crop.Save(ms, ImageFormat.Png);
|
||||
png = ms.ToArray();
|
||||
}
|
||||
// 服务端期望 data URI 或 URL(SDK 自动包装 data URI,此处显式包装)
|
||||
string fileField = "data:image/png;base64," + Convert.ToBase64String(png);
|
||||
|
||||
string body = _json.Serialize(new Dictionary<string, object>
|
||||
{
|
||||
{ "model", "glm-ocr" },
|
||||
{ "file", fileField }
|
||||
});
|
||||
|
||||
var req = (HttpWebRequest)WebRequest.Create(_baseUrl);
|
||||
req.Method = "POST";
|
||||
req.ContentType = "application/json";
|
||||
req.Headers.Add("Authorization", "Bearer " + _apiKey);
|
||||
req.Timeout = timeoutMs;
|
||||
req.ReadWriteTimeout = timeoutMs;
|
||||
req.Proxy = null; // 尊重直连,避免系统代理拖慢
|
||||
var bytes = Encoding.UTF8.GetBytes(body);
|
||||
req.ContentLength = bytes.Length;
|
||||
using (var rs = req.GetRequestStream())
|
||||
{
|
||||
rs.Write(bytes, 0, bytes.Length);
|
||||
}
|
||||
|
||||
string respText;
|
||||
try
|
||||
{
|
||||
using (var resp = (HttpWebResponse)req.GetResponse())
|
||||
using (var sr = new StreamReader(resp.GetResponseStream(),
|
||||
new UTF8Encoding(false)))
|
||||
{
|
||||
respText = sr.ReadToEnd();
|
||||
}
|
||||
}
|
||||
catch (WebException wex)
|
||||
{
|
||||
string detail = "";
|
||||
try
|
||||
{
|
||||
if (wex.Response != null)
|
||||
using (var sr = new StreamReader(wex.Response.GetResponseStream(),
|
||||
new UTF8Encoding(false)))
|
||||
detail = sr.ReadToEnd();
|
||||
}
|
||||
catch { }
|
||||
string hint = wex.Response == null
|
||||
? ("网络不可达: " + wex.Message)
|
||||
: ("HTTP " + (int)((HttpWebResponse)wex.Response).StatusCode + " " + detail);
|
||||
throw new Exception("GLM OCR 请求失败: " + hint);
|
||||
}
|
||||
|
||||
return ParseResponse(respText, crop.Width, crop.Height);
|
||||
}
|
||||
|
||||
/// <summary>解析 layout_parsing 响应 → OcrItem 列表(text 块)。</summary>
|
||||
private List<OcrItem> ParseResponse(string respText, int cropW, int cropH)
|
||||
{
|
||||
var msg = _json.Deserialize<Dictionary<string, object>>(respText);
|
||||
if (msg == null) throw new Exception("GLM 响应解析失败(空响应)");
|
||||
if (msg.ContainsKey("error"))
|
||||
{
|
||||
var err = msg["error"] as Dictionary<string, object>;
|
||||
string code = err != null && err.ContainsKey("code") ? Convert.ToString(err["code"]) : "?";
|
||||
string message = err != null && err.ContainsKey("message") ? Convert.ToString(err["message"]) : respText;
|
||||
throw new Exception("GLM OCR 错误 [" + code + "]: " + message);
|
||||
}
|
||||
|
||||
var items = new List<OcrItem>();
|
||||
var pages = msg.ContainsKey("layout_details")
|
||||
? msg["layout_details"] as System.Collections.ArrayList : null;
|
||||
if (pages == null || pages.Count == 0)
|
||||
throw new Exception("GLM 响应没有 layout_details(内容: " +
|
||||
(respText.Length > 300 ? respText.Substring(0, 300) : respText) + ")");
|
||||
|
||||
// 单页输入 → 取第一页的块
|
||||
var blocks = pages[0] as System.Collections.ArrayList;
|
||||
if (blocks == null) return items;
|
||||
|
||||
foreach (Dictionary<string, object> b in blocks)
|
||||
{
|
||||
string label = b.ContainsKey("label") ? Convert.ToString(b["label"]) : "";
|
||||
string content = b.ContainsKey("content") ? Convert.ToString(b["content"]) : "";
|
||||
if (string.IsNullOrWhiteSpace(content)) continue;
|
||||
|
||||
var bbox = b.ContainsKey("bbox_2d") ? b["bbox_2d"] as System.Collections.ArrayList : null;
|
||||
if (bbox == null || bbox.Count < 4) continue;
|
||||
|
||||
// ★ 实测返回像素坐标;官方文档描述归一化(0~1)。按值域自动兼容:
|
||||
// 四个值全部 ≤1.5 → 归一化,乘图尺寸;否则直接当像素用。
|
||||
float a0 = Convert.ToSingle(bbox[0]), a1 = Convert.ToSingle(bbox[1]);
|
||||
float a2 = Convert.ToSingle(bbox[2]), a3 = Convert.ToSingle(bbox[3]);
|
||||
bool normalized = a0 <= 1.5f && a1 <= 1.5f && a2 <= 1.5f && a3 <= 1.5f;
|
||||
float x1 = normalized ? a0 * cropW : a0;
|
||||
float y1 = normalized ? a1 * cropH : a1;
|
||||
float x2 = normalized ? a2 * cropW : a2;
|
||||
float y2 = normalized ? a3 * cropH : a3;
|
||||
// 裁剪到图内(超界坐标画在页面外会导致蒙版不可见)
|
||||
x1 = Math.Max(0, Math.Min(cropW, x1));
|
||||
y1 = Math.Max(0, Math.Min(cropH, y1));
|
||||
x2 = Math.Max(0, Math.Min(cropW, x2));
|
||||
y2 = Math.Max(0, Math.Min(cropH, y2));
|
||||
|
||||
if (label == "table")
|
||||
{
|
||||
// ★ 框选多行文字常被判定为表格块(content 为 <table> HTML):
|
||||
// 每个 <tr> 一条文本(单元格空格拼接),行高均分定位。
|
||||
// 必须在 CleanHtml 剥标签之前解析结构。
|
||||
AddTableItems(items, content, x1, y1, x2, y2);
|
||||
continue;
|
||||
}
|
||||
if (label != "text") continue; // 图片/公式块跳过
|
||||
content = CleanHtmlShared(content); // 居中块会返回 <div align=..>..</div> 等 HTML
|
||||
if (string.IsNullOrWhiteSpace(content)) continue;
|
||||
|
||||
// 多行文本块拆行(content 按 \n,块内均匀分布);清理 Markdown 标记
|
||||
// ★ 行间上下各收缩 15%(≤3px):蒙版之间留空隙,不连成一片
|
||||
string[] lines = content.Split(new[] { '\r', '\n' },
|
||||
StringSplitOptions.RemoveEmptyEntries);
|
||||
float lineH = (y2 - y1) / lines.Length;
|
||||
float pad = Math.Min(lineH * 0.15f, 3f);
|
||||
for (int i = 0; i < lines.Length; i++)
|
||||
{
|
||||
string text = CleanMarkdownShared(lines[i]).Trim();
|
||||
if (text.Length == 0) continue;
|
||||
float ty1 = y1 + lineH * i + pad;
|
||||
float ty2 = y1 + lineH * (i + 1) - pad;
|
||||
if (ty2 - ty1 < 2) { ty1 = y1 + lineH * i; ty2 = ty1 + Math.Max(2, lineH); }
|
||||
items.Add(new OcrItem
|
||||
{
|
||||
Text = text,
|
||||
Box = new[]
|
||||
{
|
||||
new PointF(x1, ty1),
|
||||
new PointF(x2, ty1),
|
||||
new PointF(x2, ty2),
|
||||
new PointF(x1, ty2)
|
||||
},
|
||||
Score = 1f
|
||||
});
|
||||
}
|
||||
}
|
||||
return items;
|
||||
}
|
||||
|
||||
private static readonly System.Text.RegularExpressions.Regex _trRegex =
|
||||
new System.Text.RegularExpressions.Regex("<tr[^>]*>(.*?)</tr>",
|
||||
System.Text.RegularExpressions.RegexOptions.Singleline |
|
||||
System.Text.RegularExpressions.RegexOptions.IgnoreCase |
|
||||
System.Text.RegularExpressions.RegexOptions.Compiled);
|
||||
private static readonly System.Text.RegularExpressions.Regex _tdRegex =
|
||||
new System.Text.RegularExpressions.Regex("<t[dh][^>]*>(.*?)</t[dh]>",
|
||||
System.Text.RegularExpressions.RegexOptions.Singleline |
|
||||
System.Text.RegularExpressions.RegexOptions.IgnoreCase |
|
||||
System.Text.RegularExpressions.RegexOptions.Compiled);
|
||||
|
||||
/// <summary>表格块 → 每 tr 一条 item(td 内容空格拼接),行高均分 bbox。</summary>
|
||||
private static void AddTableItems(List<OcrItem> items, string tableHtml,
|
||||
float x1, float y1, float x2, float y2)
|
||||
{
|
||||
var rows = new List<string>();
|
||||
foreach (System.Text.RegularExpressions.Match tr in _trRegex.Matches(tableHtml))
|
||||
{
|
||||
var cells = new List<string>();
|
||||
foreach (System.Text.RegularExpressions.Match td in _tdRegex.Matches(tr.Groups[1].Value))
|
||||
{
|
||||
string cell = CleanHtmlShared(td.Groups[1].Value).Trim();
|
||||
if (cell.Length > 0) cells.Add(cell);
|
||||
}
|
||||
if (cells.Count > 0) rows.Add(string.Join(" ", cells.ToArray()));
|
||||
}
|
||||
if (rows.Count == 0) return;
|
||||
float rowH = (y2 - y1) / rows.Count;
|
||||
float pad = Math.Min(rowH * 0.15f, 3f); // 行间留空隙,蒙版不连片
|
||||
for (int i = 0; i < rows.Count; i++)
|
||||
{
|
||||
float ty1 = y1 + rowH * i + pad;
|
||||
float ty2 = y1 + rowH * (i + 1) - pad;
|
||||
if (ty2 - ty1 < 2) { ty1 = y1 + rowH * i; ty2 = ty1 + Math.Max(2, rowH); }
|
||||
items.Add(new OcrItem
|
||||
{
|
||||
Text = rows[i],
|
||||
Box = new[]
|
||||
{
|
||||
new PointF(x1, ty1),
|
||||
new PointF(x2, ty1),
|
||||
new PointF(x2, ty2),
|
||||
new PointF(x1, ty2)
|
||||
},
|
||||
Score = 1f
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
private static readonly System.Text.RegularExpressions.Regex _htmlTag =
|
||||
new System.Text.RegularExpressions.Regex("<[^>]*>",
|
||||
System.Text.RegularExpressions.RegexOptions.Compiled);
|
||||
|
||||
/// <summary>
|
||||
/// 剥离 GLM content 里的 HTML 片段(居中块返回 div 等)与常见实体;
|
||||
/// 标签替换为空格,避免前后文字误粘连;连续空格合并。
|
||||
/// </summary>
|
||||
internal static string CleanHtmlShared(string s)
|
||||
{
|
||||
if (string.IsNullOrEmpty(s)) return s;
|
||||
if (s.IndexOf('<') >= 0)
|
||||
{
|
||||
s = _htmlTag.Replace(s, " ");
|
||||
s = s.Replace(" ", " ").Replace("&", "&")
|
||||
.Replace("<", "<").Replace(">", ">")
|
||||
.Replace(""", "\"").Replace("'", "'");
|
||||
}
|
||||
while (s.Contains(" ")) s = s.Replace(" ", " ");
|
||||
return s;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// 清理 GLM 返回 content 里的 Markdown 标记(★保守规则,不伤正文):
|
||||
/// · 行首 "##"/"#" 标题井号(仅行首;"2#配电箱" 中间的 # 保留)
|
||||
/// · 成对 **加粗** 星号、行首 "- "/"* " 列表标记
|
||||
/// </summary>
|
||||
internal static string CleanMarkdownShared(string s)
|
||||
{
|
||||
if (string.IsNullOrEmpty(s)) return s;
|
||||
// 行首标题:# 开头且后跟空格/更多#/行尾 → 剥掉行首所有 #
|
||||
string t = s.TrimStart();
|
||||
if (t.StartsWith("#"))
|
||||
{
|
||||
int i = 0;
|
||||
while (i < t.Length && t[i] == '#') i++;
|
||||
// "#..." 后是空格或行尾才视为标题标记
|
||||
if (i >= t.Length || t[i] == ' ' || t[i] == '\t')
|
||||
t = t.Substring(i).TrimStart();
|
||||
// 否则(如 "#5 线缆")保留原样
|
||||
else t = s;
|
||||
}
|
||||
// 行首列表标记
|
||||
if (t.StartsWith("- ") || t.StartsWith("* "))
|
||||
t = t.Substring(2);
|
||||
// 成对 ** (加粗) 去星号
|
||||
if (t.Contains("**"))
|
||||
t = t.Replace("**", "");
|
||||
return t;
|
||||
}
|
||||
|
||||
private void Notify(string s)
|
||||
{
|
||||
var h = StatusChanged;
|
||||
if (h != null) h(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user