343 lines
13 KiB
Python
343 lines
13 KiB
Python
# -*- coding: utf-8 -*-
|
||
"""
|
||
Web backend for the OCR table -> online spreadsheet pipeline.
|
||
|
||
The OCR call mirrors `api.py` as closely as possible:
|
||
* same module-level imports (alibabacloud_ocr_api20210707, credentials,
|
||
tea_openapi, darabonba_stream, tea_util, etc.)
|
||
* same `Sample` class with `create_client` (only difference: credentials
|
||
are parameterized instead of hardcoded)
|
||
* same `main` shape: build client -> read body stream -> build
|
||
`RecognizeTableOcrRequest` -> call `recognize_table_ocr_with_options`
|
||
* response: we use `to_map()` (the proper TeaModel way) instead of
|
||
`print(json.dumps(resp, default=str))`, which produces invalid JSON
|
||
(Python repr, not JSON) in the original.
|
||
|
||
Endpoints:
|
||
GET / -> index.html
|
||
POST /api/convert -> image + aliyun AK -> full pipeline -> {sheet_data, ocr_result}
|
||
POST /api/load_xlsx -> .xlsx file -> {sheet_data}
|
||
POST /api/load_json -> OCR-style .json file -> {sheet_data}
|
||
GET /api/load_demo -> 111.html / 111.xlsx -> {sheet_data}
|
||
GET /sheet -> sheet.html (legacy standalone template)
|
||
GET /api/health -> health check
|
||
"""
|
||
import json
|
||
import os
|
||
import re
|
||
import sys
|
||
import tempfile
|
||
import traceback
|
||
|
||
from flask import Flask, jsonify, request, send_from_directory
|
||
|
||
# Make sibling modules importable
|
||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||
sys.path.insert(0, HERE)
|
||
|
||
# ---------- Aliyun SDK imports (mirror api.py) ----------
|
||
from alibabacloud_credentials.client import Client as CredentialClient
|
||
from alibabacloud_credentials.models import Config as CredentialConfig
|
||
from alibabacloud_darabonba_stream.client import Client as StreamClient
|
||
from alibabacloud_ocr_api20210707 import models as ocr_api_20210707_models
|
||
from alibabacloud_ocr_api20210707.client import Client as ocr_api20210707Client
|
||
from alibabacloud_tea_openapi import models as open_api_models
|
||
from alibabacloud_tea_util import models as util_models
|
||
|
||
from generate_sheet import xlsx_to_spreadsheet_data # noqa: E402
|
||
|
||
|
||
# ---------- Sample class: same shape as api.py ----------
|
||
class Sample:
|
||
def __init__(self):
|
||
pass
|
||
|
||
@staticmethod
|
||
def create_client(access_key_id: str, access_key_secret: str) -> ocr_api20210707Client:
|
||
"""
|
||
Mirrors api.py's create_client, but credentials are parameters
|
||
instead of hardcoded. Everything else (config shape, endpoint)
|
||
is identical to api.py.
|
||
"""
|
||
credentialsConfig = CredentialConfig(
|
||
type='access_key',
|
||
access_key_id=access_key_id,
|
||
access_key_secret=access_key_secret,
|
||
)
|
||
credentialsClient = CredentialClient(credentialsConfig)
|
||
config = open_api_models.Config(
|
||
credential=credentialsClient
|
||
)
|
||
config.endpoint = f'ocr-api.cn-hangzhou.aliyuncs.com'
|
||
return ocr_api20210707Client(config)
|
||
|
||
@staticmethod
|
||
def recognize_table(image_path: str, access_key_id: str, access_key_secret: str) -> dict:
|
||
"""
|
||
Mirrors api.py's main(), but:
|
||
* image path is a parameter
|
||
* credentials are parameters
|
||
* returns the structured response dict (via to_map) instead of
|
||
printing invalid-JSON string
|
||
"""
|
||
client = Sample.create_client(access_key_id, access_key_secret)
|
||
body_stream = StreamClient.read_from_file_path(image_path)
|
||
recognize_table_ocr_request = ocr_api_20210707_models.RecognizeTableOcrRequest(
|
||
body=body_stream
|
||
)
|
||
runtime = util_models.RuntimeOptions()
|
||
resp = client.recognize_table_ocr_with_options(recognize_table_ocr_request, runtime)
|
||
# `to_map()` is the documented way to get a JSON-serializable dict
|
||
# from a TeaModel response. This is what api.py's `default=str`
|
||
# hack was trying to do (and failing at, since str(resp) yields
|
||
# a Python repr, not JSON).
|
||
return resp.to_map() if hasattr(resp, 'to_map') else _tea_fallback(resp)
|
||
|
||
|
||
def _tea_fallback(resp):
|
||
"""Last-resort: serialize via str(). May produce Python repr, not JSON."""
|
||
try:
|
||
return json.loads(json.dumps(resp, default=str))
|
||
except Exception:
|
||
return {"_raw": str(resp)}
|
||
|
||
|
||
# ---------- Flask app ----------
|
||
app = Flask(__name__, static_folder=None)
|
||
|
||
|
||
def run_subprocess(cmd: list) -> tuple[int, str, str]:
|
||
import subprocess
|
||
p = subprocess.run(cmd, capture_output=True, text=True, cwd=HERE)
|
||
return p.returncode, p.stdout, p.stderr
|
||
|
||
|
||
@app.route("/")
|
||
def index():
|
||
return send_from_directory(HERE, "index.html")
|
||
|
||
|
||
@app.route("/sheet")
|
||
@app.route("/sheet.html")
|
||
def sheet():
|
||
return send_from_directory(HERE, "sheet.html")
|
||
|
||
|
||
@app.route("/111.html")
|
||
def sheet_111():
|
||
return send_from_directory(HERE, "111.html")
|
||
|
||
|
||
@app.route("/api/health")
|
||
def health():
|
||
return jsonify({"ok": True, "service": "shudao"})
|
||
|
||
|
||
@app.route("/api/convert", methods=["POST"])
|
||
def api_convert():
|
||
if "image" not in request.files:
|
||
return jsonify({"ok": False, "stage": "input", "error": "缺少图片(字段名 image)"}), 400
|
||
|
||
access_key_id = (request.form.get("access_key_id") or "").strip()
|
||
access_key_secret = (request.form.get("access_key_secret") or "").strip()
|
||
if not access_key_id or not access_key_secret:
|
||
return jsonify({"ok": False, "stage": "input", "error": "缺少阿里云 AccessKey ID / Secret"}), 400
|
||
|
||
image = request.files["image"]
|
||
if not image.filename:
|
||
return jsonify({"ok": False, "stage": "input", "error": "图片文件为空"}), 400
|
||
|
||
workdir = tempfile.mkdtemp(prefix="shudao_")
|
||
img_path = os.path.join(workdir, "input" + os.path.splitext(image.filename)[1].lower() or ".png")
|
||
json_path = os.path.join(workdir, "ocr.json")
|
||
xlsx_path = os.path.join(workdir, "ocr.xlsx")
|
||
|
||
try:
|
||
image.save(img_path)
|
||
|
||
# ---- Step 1: OCR (mirrors api.py exactly) ----
|
||
try:
|
||
ocr_raw = Sample.recognize_table(img_path, access_key_id, access_key_secret)
|
||
except Exception as e:
|
||
return jsonify({
|
||
"ok": False,
|
||
"stage": "ocr",
|
||
"error": f"OCR 调用失败:{e}",
|
||
"trace": traceback.format_exc(),
|
||
}), 500
|
||
|
||
# The Aliyun SDK returns {body: {Data: "<json string>", RequestId}, headers, statusCode}
|
||
# json_to_excel.py expects the body shape: {Data: "<json string>", RequestId}
|
||
# (it then json.loads the Data string to get prism_tablesInfo).
|
||
ocr_body = ocr_raw.get("body") if isinstance(ocr_raw, dict) else None
|
||
if not isinstance(ocr_body, dict):
|
||
return jsonify({
|
||
"ok": False,
|
||
"stage": "ocr",
|
||
"error": "OCR 响应格式异常:缺少 body 字段",
|
||
"raw_type": type(ocr_raw).__name__,
|
||
}), 500
|
||
|
||
with open(json_path, "w", encoding="utf-8") as f:
|
||
json.dump(ocr_body, f, ensure_ascii=False, indent=2)
|
||
|
||
# ---- Step 2: json -> xlsx ----
|
||
rc, out, err = run_subprocess(
|
||
[sys.executable, os.path.join(HERE, "json_to_excel.py"), json_path, xlsx_path]
|
||
)
|
||
if rc != 0:
|
||
return jsonify({
|
||
"ok": False,
|
||
"stage": "json_to_excel",
|
||
"error": err or out or "json_to_excel 失败",
|
||
}), 500
|
||
|
||
# ---- Step 3: xlsx -> sheet data ----
|
||
try:
|
||
sheet_data = xlsx_to_spreadsheet_data(xlsx_path)
|
||
except Exception as e:
|
||
return jsonify({
|
||
"ok": False,
|
||
"stage": "xlsx_to_sheet",
|
||
"error": f"xlsx 解析失败:{e}",
|
||
"trace": traceback.format_exc(),
|
||
}), 500
|
||
|
||
return jsonify({
|
||
"ok": True,
|
||
"sheet_data": sheet_data,
|
||
"ocr_result": ocr_raw,
|
||
})
|
||
except Exception as e:
|
||
return jsonify({
|
||
"ok": False,
|
||
"stage": "unhandled",
|
||
"error": f"未处理异常:{e}",
|
||
"trace": traceback.format_exc(),
|
||
}), 500
|
||
|
||
|
||
# -------------------- Direct file imports (skip OCR) --------------------
|
||
@app.route("/api/load_xlsx", methods=["POST"])
|
||
def api_load_xlsx():
|
||
"""Directly parse an uploaded .xlsx file into sheet data."""
|
||
if "file" not in request.files:
|
||
return jsonify({"ok": False, "error": "缺少文件(字段名 file)"}), 400
|
||
f = request.files["file"]
|
||
if not f.filename:
|
||
return jsonify({"ok": False, "error": "文件为空"}), 400
|
||
if not f.filename.lower().endswith(".xlsx"):
|
||
return jsonify({"ok": False, "error": "请上传 .xlsx 文件"}), 400
|
||
|
||
workdir = tempfile.mkdtemp(prefix="shudao_xlsx_")
|
||
xlsx_path = os.path.join(workdir, f.filename)
|
||
try:
|
||
f.save(xlsx_path)
|
||
sheet_data = xlsx_to_spreadsheet_data(xlsx_path)
|
||
return jsonify({"ok": True, "sheet_data": sheet_data})
|
||
except Exception as e:
|
||
return jsonify({
|
||
"ok": False,
|
||
"stage": "xlsx_parse",
|
||
"error": f"xlsx 解析失败:{e}",
|
||
"trace": traceback.format_exc(),
|
||
}), 500
|
||
|
||
|
||
@app.route("/api/load_json", methods=["POST"])
|
||
def api_load_json():
|
||
"""Parse an uploaded Aliyun-OCR-style JSON file -> xlsx -> sheet data."""
|
||
if "file" not in request.files:
|
||
return jsonify({"ok": False, "error": "缺少文件(字段名 file)"}), 400
|
||
f = request.files["file"]
|
||
if not f.filename:
|
||
return jsonify({"ok": False, "error": "文件为空"}), 400
|
||
|
||
workdir = tempfile.mkdtemp(prefix="shudao_json_")
|
||
json_path = os.path.join(workdir, "input.json")
|
||
xlsx_path = os.path.join(workdir, "out.xlsx")
|
||
try:
|
||
f.save(json_path)
|
||
# Validate the JSON structure
|
||
with open(json_path, "r", encoding="utf-8") as fp:
|
||
data = json.load(fp)
|
||
# Accept either the body shape {Data: "<json string>", RequestId: ...}
|
||
# or the already-parsed OCR shape with prism_tablesInfo at the top level
|
||
if isinstance(data, dict) and "Data" in data and isinstance(data["Data"], str):
|
||
ocr_body = data # already in body shape
|
||
elif isinstance(data, dict) and "prism_tablesInfo" in data:
|
||
# Wrap into body shape so json_to_excel can handle it
|
||
ocr_body = {"Data": json.dumps(data, ensure_ascii=False)}
|
||
else:
|
||
return jsonify({
|
||
"ok": False,
|
||
"stage": "json_validate",
|
||
"error": "JSON 格式不符合预期(既不是 Aliyun OCR body 也没有 prism_tablesInfo)",
|
||
}), 400
|
||
with open(json_path, "w", encoding="utf-8") as fp:
|
||
json.dump(ocr_body, fp, ensure_ascii=False, indent=2)
|
||
|
||
rc, out, err = run_subprocess(
|
||
[sys.executable, os.path.join(HERE, "json_to_excel.py"), json_path, xlsx_path]
|
||
)
|
||
if rc != 0:
|
||
return jsonify({
|
||
"ok": False,
|
||
"stage": "json_to_excel",
|
||
"error": err or out or "json_to_excel 失败",
|
||
}), 500
|
||
|
||
sheet_data = xlsx_to_spreadsheet_data(xlsx_path)
|
||
return jsonify({"ok": True, "sheet_data": sheet_data})
|
||
except Exception as e:
|
||
return jsonify({
|
||
"ok": False,
|
||
"stage": "json_load",
|
||
"error": f"JSON 加载失败:{e}",
|
||
"trace": traceback.format_exc(),
|
||
}), 500
|
||
|
||
|
||
@app.route("/api/load_demo", methods=["GET"])
|
||
def api_load_demo():
|
||
"""Load the project-local 111.html's embedded sheet data (if any),
|
||
or fall back to re-running generate_sheet on 111.xlsx."""
|
||
# Approach 1: extract {{DATA}} from 111.html (cleanest, no re-conversion)
|
||
candidate_html = os.path.join(HERE, "111.html")
|
||
if os.path.exists(candidate_html):
|
||
try:
|
||
with open(candidate_html, "r", encoding="utf-8") as fp:
|
||
content = fp.read()
|
||
# Find the first { ... } that contains the "name" key
|
||
# The data is in a `const rawData = {...};` block
|
||
m = re.search(r"const\s+rawData\s*=\s*(\{.*?\});\s*//\s*Normalize", content, re.DOTALL)
|
||
if m:
|
||
sheet_data = json.loads(m.group(1))
|
||
return jsonify({"ok": True, "sheet_data": sheet_data, "source": "111.html"})
|
||
except Exception:
|
||
pass # fall through
|
||
|
||
# Approach 2: re-run generate_sheet on 111.xlsx
|
||
candidate_xlsx = os.path.join(HERE, "111.xlsx")
|
||
if not os.path.exists(candidate_xlsx):
|
||
return jsonify({
|
||
"ok": False,
|
||
"error": "示例数据不可用(既没有 111.html 也没有 111.xlsx)",
|
||
}), 404
|
||
try:
|
||
sheet_data = xlsx_to_spreadsheet_data(candidate_xlsx)
|
||
return jsonify({"ok": True, "sheet_data": sheet_data, "source": "111.xlsx"})
|
||
except Exception as e:
|
||
return jsonify({
|
||
"ok": False,
|
||
"stage": "xlsx_parse",
|
||
"error": f"111.xlsx 解析失败:{e}",
|
||
"trace": traceback.format_exc(),
|
||
}), 500
|
||
|
||
|
||
if __name__ == "__main__":
|
||
port = int(os.environ.get("PORT", "5000"))
|
||
print(f" * Serving Shudao on http://127.0.0.1:{port}")
|
||
app.run(host="127.0.0.1", port=port, debug=False)
|