From 97a3883d5fe1b98ad4adf6814dab291709a78708 Mon Sep 17 00:00:00 2001 From: Nemoyuzx <103307047+Nemoyuzx@users.noreply.github.com> Date: Mon, 7 Sep 2026 17:29:43 +0800 Subject: [PATCH 1/4] feat: improve local review workflow with privacy guard --- .gitignore | 11 + AGENTS.md | 7 +- README.md | 6 +- docs/ARCHITECTURE.md | 24 ++ docs/OPERATIONS.md | 21 +- pyproject.toml | 2 +- .../restore_standard_solutions_from_backup.py | 93 ++++++ scripts/setup.sh | 8 +- src/cuoti/app.py | 62 +++- src/cuoti/cli.py | 21 ++ src/cuoti/db.py | 115 ++++++- src/cuoti/ingest.py | 7 +- src/cuoti/macos_service.py | 291 ++++++++++++++++++ src/cuoti/models.py | 130 ++++++++ src/cuoti/pdf_export.py | 132 ++++++-- src/cuoti/pdf_worker.py | 30 ++ src/cuoti/rich_text.py | 47 ++- src/cuoti/static/app.css | 26 +- src/cuoti/static/app.js | 6 + src/cuoti/templates/base.html | 6 +- src/cuoti/templates/dashboard.html | 22 +- src/cuoti/templates/detail.html | 33 +- src/cuoti/templates/import.html | 53 ++++ src/cuoti/templates/pdf.html | 33 +- src/cuoti/templates/review.html | 67 +++- tests/test_app.py | 120 +++++++- tests/test_db.py | 143 ++++++++- tests/test_ingest.py | 14 + tests/test_macos_service.py | 24 ++ tests/test_pdf.py | 46 +++ tests/test_privacy.py | 59 ++++ tests/test_rich_text.py | 23 +- uv.lock | 11 + ...1\224\231\351\242\230\346\234\254.command" | 9 + 34 files changed, 1575 insertions(+), 127 deletions(-) create mode 100644 scripts/restore_standard_solutions_from_backup.py create mode 100644 src/cuoti/macos_service.py create mode 100644 src/cuoti/pdf_worker.py create mode 100644 src/cuoti/templates/import.html create mode 100644 tests/test_macos_service.py create mode 100644 tests/test_privacy.py diff --git a/.gitignore b/.gitignore index 11364d0..c1c21fa 100644 --- a/.gitignore +++ b/.gitignore @@ -19,6 +19,17 @@ dist/ *.sqlite3-shm *.sqlite3-wal *.db +*.jpg +*.jpeg +*.JPG +*.JPEG +*.png +*.PNG +*.heic +*.HEIC +*.pdf +*.PDF +local-data/ data/inbox/* !data/inbox/.gitkeep output/* diff --git a/AGENTS.md b/AGENTS.md index 76dd40e..1577151 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -4,10 +4,10 @@ ## 用户发来错题照片时 -1. 用原生图片理解能力逐张查看。多题同页必须拆成多条记录;连续多图属于同一题时可合并理解,但每条记录保留来源页。收录范围取“明显错题/订正题”与“题号被圈出的题”的并集;题号被圈时即使未看到红笔订正也必须入库。 -2. 忠实提取题干、选项、学生错误答案、正确答案和批注。所有公式写成 LaTeX:行内 `$...$`,独立公式 `$$...$$`。 +1. 用原生图片理解能力逐张查看。多题同页必须拆成多条记录;连续多图属于同一题时可合并理解,但每条记录保留来源页。收录范围取“明显错题/订正题”与“题号被圈出的题”的并集;题号被圈时即使未看到红笔订正也必须入库。答案位置空白但旁边有明确对钩,且题号未圈、没有红笔订正或其他错误标记时,表示该题已经掌握,必须排除;禁止仅因没有手写答案就把它识别成错题。 +2. 忠实提取题干、选项、学生错误答案、正确答案和批注。`wrong_answer` 只填写学生最终写出的具体答案;答题区为空白、只有思路但没有最终答案,或只有红笔写出的答案/订正但看不到可确认的原答案时统一写 `不会`,禁止写“待确认(见原图红笔答案或订正)”“思路错误”或作答过程描述。能看清原来的具体错误答案时忠实记录,作答过程错误只写入 `error_reason`。有答案或解析 PDF 时,`correct_answer` 必须写 PDF 给出的明确最终答案或结论,禁止写“见解析”“见标准解析”“待确认”等占位文本;`analysis` 必须忠实转写对应 PDF 原解,禁止沿用概括版或自行编写。若 PDF 对应关系或文字无法可靠确认,两个字段留空并停止自动确认,交由人工复核。选择题的 `correct_answer` 以 `A`、`B`、`C`、`D` 等裸字母开头,不写成 `(A)` 或 `(A)`;公式内部括号照常保留。所有公式写成 LaTeX:行内 `$...$`,独立公式 `$$...$$`。独立公式与前后正文之间只换一行,不留空白行;连续独立公式之间也不留空白行。 3. 科目只能是 `数学`、`英语`、`408`、`政治`。使用下方标准板块;看不清或不能确定时写 `待确认`,不能臆造。 -4. 原图没有解析时,补出步骤完整、可独立理解的解析;指出具体错因,给出短知识点标签。 +4. 有答案/标准解析照片或 PDF 时,`analysis` 必须按原解顺序忠实转写方法、计算和结论,禁止概括改写或用自己解法替代。只有原图确实没有解析时才补出步骤完整、可独立理解的解析,并标明“【补充解析】”。若标准解析看不清、缺页或对应不确定,停止自行解答,保留解析图、置为低置信度待复核。`error_reason` 只能根据照片中可见的学生解答过程填写具体错误;没有解答过程、只有题号被圈、答案被订正、未作答或只有标准解析时必须留空,禁止写“题号被圈或答案被红笔订正,需要对照标准解析复盘”等占位话术。黑笔圈题按用户约定保留“再做一次”。给出短知识点标签。 5. 按 `docs/extraction-example.json` 生成临时 JSON,用 Pydantic 校验并导入: ```bash @@ -37,6 +37,7 @@ - PDF 修改后必须导出两种版本,用 `pdftoppm` 转成 PNG 并目视检查公式、中文、图片、分页和留白。 - 每次改动运行 `./.venv/bin/pytest`;涉及界面时再做真实浏览器检查。 - 运行时产物只放 `data/inbox`、`tmp`、`output` 或各科 `错题_auto`,不要散落到源码目录。 +- GitHub 只允许同步通用代码、通用文档和测试。个人错题数据及其元信息一律留在本机,包括照片、题目/答案内容、批次日期、原始文件名、题号清单、数据库记录 ID、JSON、Markdown 镜像、备份和导出文件;提交或推送前必须检查暂存区和 Git 跟踪文件,发现这些内容立即停止。 ## 当前架构状态 diff --git a/README.md b/README.md index 7f45622..8de9b2f 100644 --- a/README.md +++ b/README.md @@ -34,11 +34,13 @@ cd cuoti-auto ./scripts/setup.sh ``` -安装脚本会创建 `.venv`、安装 Python 与 npm 依赖、初始化四科数据库,并在 macOS 桌面创建“打开错题本.command”快捷方式。之后也可以手动启动: +安装脚本会创建 `.venv`、安装 Python 与 npm 依赖、初始化四科数据库,并在 macOS 桌面创建“打开错题本.command”快捷方式。macOS 上同时会安装用户级 LaunchAgent:登录后自动启动服务并打开一次浏览器,服务异常退出后会自动拉起。也可以手动管理: ```bash .venv/bin/cuoti doctor -.venv/bin/cuoti serve +.venv/bin/cuoti service status +.venv/bin/cuoti service install +.venv/bin/cuoti service uninstall ``` 浏览器会打开 。 diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 527a722..8e5c61a 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -38,6 +38,30 @@ 用户要求每科在对应桌面目录独立建库。网页查询时依次读取四个小库后合并,因此既满足物理隔离,也保留统一筛选。数据库启用 WAL 和外键;`source_hash + question_text` 唯一约束用于阻止同一照片同一题重复入库。 +## 选项标签不变量 + +`questions.options_json` 只保存选项正文,不保存 `A.`、`B.` 等序号。导入、复核保存和数据库更新通过 `normalize_options` 去除与当前位置相符的历史前缀;网页、实时预览、Markdown 和 PDF 在渲染时通过 `option_label` 统一补回标签。空字符串会作为缺失选项的占位保留,渲染时隐藏但不改变后续字母;人工编辑中的普通空行则由 `parse_options_text` 忽略。这样可避免 `A. A. ...` 重复、选项错位,并保证所有出口编号一致。 + +`questions.correct_answer` 中位于字段开头的选择题字母只保存裸标签,例如 `C` 或 `C $O(n)$`,不保存 `(C)`、`(C)`。导入模型和数据库更新统一通过 `normalize_correct_answer` 清理开头标签;规则锚定字段开头,因此不会改动公式或正文内部的括号。 + +## 复核图片角色 + +`images.image_role` 区分 `question`(题目)、`work`(学生作答/订正)、`solution`(答案/标准解析)和待细分的 `supplement`。复核页将题目与作答照片放在前组、答案解析照片放在后组并提供快捷跳转;答案解析图永远不进入纯题 PDF 的可选图片列表。新增附图必须通过 `SubjectStore.add_image` 明确写入角色。 + +`analysis` 的内容来源遵循严格优先级:已关联的标准解析页 > 无标准解析时的补充解析。有标准解析页时只做忠实转写,不允许用人工摘要或模型自解替换书中的方法。无法确定页面对应或公式时,保留解析图并降为低置信度待复核,不得生成看似已核对的文字。 + +## 行间公式间距不变量 + +结构化富文本字段和选项在 Pydantic 导入、数据库更新及网页/PDF 渲染入口统一经过 `normalize_rich_text_spacing`。独立公式 `$$...$$` 与前后正文之间只保留一个换行,不能出现空白行;连续独立公式同样紧邻。Markdown 代码围栏内部不执行该规则,避免改变示例程序的原始格式。历史数据迁移脚本位于 `tmp/normalize_display_math_spacing.py`。 + +## 错因证据不变量 + +`questions.error_reason` 只记录能从学生解答过程确认的具体错误。题号圈选、红笔订正、未作答和标准解析只能决定收录或辅助讲解,不能据此推断错因;没有作答过程时字段保持空字符串。识别模型、结构化导入和复核保存通过 `normalize_error_reason` 清除已知的泛化占位话术,复核输入框保持为空。“再做一次”是用户指定的黑笔圈题复习标记,不属于待核对占位话术,予以保留。 + +## 错误答案不变量 + +`questions.wrong_answer` 只保存学生最终写出的具体答案,不保存“待确认(见原图红笔答案或订正)”“思路错误”等状态性占位话术,也不保存作答过程描述。识别到答题区为空白、只有思路未形成最终答案,或只有红笔答案/订正而无法确认学生原答案时统一存为“不会”;能看清具体原错误答案时仍忠实保存。作答过程错误写入 `error_reason`。结构化导入和复核保存统一通过 `normalize_wrong_answer` 执行基础规则。 + ## 可扩展点 新增 OCR 供应商时返回 `ExtractionBatch` 并加入 `ingest_path` 的适配器表即可。不要改变核心表来适配供应商。若以后增加复习算法,应写入 `attempts`,不要覆盖历史答案。 diff --git a/docs/OPERATIONS.md b/docs/OPERATIONS.md index ff979b8..35c0cb7 100644 --- a/docs/OPERATIONS.md +++ b/docs/OPERATIONS.md @@ -10,6 +10,16 @@ cd /path/to/cuoti-auto 如果端口被占用,可临时使用 `CUOTI_PORT=8877 .venv/bin/cuoti serve`。 +### macOS 登录自启与常驻 + +```bash +.venv/bin/cuoti service install +.venv/bin/cuoti service status +.venv/bin/cuoti service uninstall +``` + +`install` 会在 `~/Library/LaunchAgents/` 安装两个用户级任务:`com.nemoyu.cuoti-auto` 常驻检查 Web 服务并在异常退出后自动拉起;`com.nemoyu.cuoti-auto.open` 每次登录只等待服务就绪并打开一次浏览器。由于 macOS 会禁止 launchd 直接读取桌面下的项目和数据库,监视器在需要时会通过 Terminal 的桌面访问权限启动后台 supervisor,无需给 Python 开启“完全磁盘访问”。日志保存在 `output/logs/`,重启服务不会反复弹出新标签页。 + ## 对比复核 - 首页点“开始对比复核”,左侧查看原始照片,右侧直接修改题干、选项、答案、解析和分类。 @@ -19,11 +29,12 @@ cd /path/to/cuoti-auto ## 科目分页与 PDF 导出 -- 四科使用独立页面:`/subject/math`、`/subject/english`、`/subject/cs408`、`/subject/politics`。科目通过顶部页签切换,筛选表单不再包含科目下拉框。 +- 四科使用独立页面:`/subject/math`、`/subject/english`、`/subject/cs408`、`/subject/politics`。科目通过顶部页签切换,筛选表单不再包含科目下拉框。图片上传统一使用独立的 `/import` 页面,科目页不内嵌上传表单;导入完成后停留在导入页并提供对应科目与复核入口。 - 题库卡片、待复核队列和 PDF 导出共用统一顺序:科目 → 章节 → 小节/板块 → 题号。数字按自然数比较,因此第 2 章排在第 10 章之前;`待确认`内容放在末尾。 - 题库桌面端固定每行两题,选项横向排列且不叠加浏览器自动序号。点击题目后先显示题干、答案和解析,原图与编辑表单默认折叠在页底。 - 题号在网页、复核、Markdown 和 PDF 中同时显示建档日期。Markdown 三反引号代码块会在网页和 PDF 中渲染为独立代码区域;解答、应用、计算、证明等长题在 PDF 中独占一列。 - PDF 由单线程后台队列生成,不占用页面请求。顶部导航栏的“PDF 导出”浮窗轮询 `/api/exports/{job_id}`,用圆环显示进度、完成后显示对勾并提供下载。 +- 公式密集的大批次会每 24 题启动一个独立 WeasyPrint 子进程,串行渲染后合并为一个 PDF;子进程结束即释放内存,避免数百道公式题使 Web 服务常驻数 GB 排版缓存。 - 完整错题本版只输出结构化题目、错误答案、正确答案、解析、错因和知识点,不带原始拍照页;纯题版只会带复核页中人工勾选的无答案题图。 - 任务记录保存在当前服务进程内,服务重启后旧任务进度会失效,但已生成的 PDF 仍保留在 `output/pdf/`。 @@ -49,6 +60,7 @@ cd /path/to/cuoti-auto - 一页 1-3 题最利于题目切分;连续过程要按页码顺序命名。 - 批改符号、错误答案和正确答案都要入镜。反光严重或焦外的照片先重拍。 - 入库时同时查找订正痕迹和题号圈选:题号被圈出的题一律收录,与其是否能看清原错答无关。 +- 答案位置为空但有明确对钩,且题号未圈、没有红笔订正或其他错误标记时,按“已经掌握”排除;空白本身不能作为错题证据。 - 自动结果进入“待复核”;详情页复核后改为“已复核”。 - 原始照片可能含答案,默认不进入纯题 PDF。几何图、材料图等无答案插图可在详情页勾选“纯题 PDF 图片”。 @@ -63,6 +75,13 @@ cd /path/to/cuoti-auto 停止写入后复制四个 `wrong_questions.sqlite3` 以及 `assets/`、`markdown/` 即可完整恢复。WAL 模式运行中备份时优先用 SQLite 在线备份 API,不要只复制主库而漏掉 `-wal`。 +## 本地批次记录与隐私边界 + +- 批次 JSON、图片映射、审计结果和一次性修复脚本只保存在被 Git 忽略的 `tmp/` 中;题目照片、数据库、Markdown、备份和导出文件只保存在各科本地 `错题_auto` 或运行时目录中。 +- 写入前使用 SQLite 在线备份。解析照片的 `image_role` 是 `solution`,不会进入纯题 PDF。 +- 挂接 `solution` 图片不等于文字解析已核对。批量导入时必须逐题按标准解析页忠实转写 `analysis`;若页面缺失、对应不确定或公式无法辨认,应留空、降低置信度并交由人工复核,禁止改用自写摘要。 +- GitHub 只同步通用程序、通用文档与测试。不得提交或推送批次日期、原始文件名、题号清单、记录 ID、题目/答案映射、个人路径、照片、数据库、Markdown 镜像或导出文件。 + ## 常见问题 - 网页公式还是 `$...$`:运行 `npm install`,确认 `/vendor/katex/katex.min.js` 能访问。 diff --git a/pyproject.toml b/pyproject.toml index 623aa54..4369469 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -29,7 +29,7 @@ dependencies = [ ] [project.optional-dependencies] -dev = ["httpx>=0.28,<1", "pytest>=8.4,<10"] +dev = ["httpx>=0.28,<1", "pypdf>=6,<7", "pytest>=8.4,<10"] [project.scripts] cuoti = "cuoti.cli:main" diff --git a/scripts/restore_standard_solutions_from_backup.py b/scripts/restore_standard_solutions_from_backup.py new file mode 100644 index 0000000..b5e372c --- /dev/null +++ b/scripts/restore_standard_solutions_from_backup.py @@ -0,0 +1,93 @@ +"""Restore solution fields that were replaced by the review placeholder. + +This is a narrow recovery tool for the accidental provenance quarantine. It +copies only ``correct_answer`` and ``analysis`` (plus their earlier confidence) +from a user-selected online SQLite backup, and only for rows whose current +analysis is the exact quarantine placeholder. All later edits to the question, +options, student answer, classification, notes and image links are preserved. +""" + +from __future__ import annotations + +import argparse +import sqlite3 +from datetime import datetime +from pathlib import Path + +from cuoti.config import Settings +from cuoti.db import SubjectStore + + +PENDING = "标准解析待人工核对(见已关联解析图)" + + +def online_backup(source: Path, target: Path) -> None: + target.parent.mkdir(parents=True, exist_ok=True) + with sqlite3.connect(source) as src, sqlite3.connect(target) as dst: + src.backup(dst) + + +def restore(subject: str, source_backup: Path) -> tuple[int, int]: + settings = Settings.load() + store = SubjectStore(subject, settings) + store.initialize() + + stamp = datetime.now().astimezone().strftime("%Y%m%d-%H%M%S") + safety_backup = store.root / "backups" / f"{stamp}-before-standard-solution-restore.sqlite3" + online_backup(store.db_path, safety_backup) + + with sqlite3.connect(source_backup) as connection: + connection.row_factory = sqlite3.Row + earlier = { + int(row["id"]): row + for row in connection.execute( + "SELECT id, source_file, question_text, correct_answer, analysis, confidence " + "FROM questions" + ) + } + + restored = 0 + skipped = 0 + for record in store.list(): + if record.analysis.strip() != PENDING: + continue + old = earlier.get(record.id) + if ( + old is None + or old["source_file"] != record.source_file + or old["question_text"] != record.question_text + or not str(old["analysis"]).strip() + or str(old["analysis"]).strip() == PENDING + ): + skipped += 1 + continue + store.update( + record.id, + { + "correct_answer": str(old["correct_answer"]).strip(), + "analysis": str(old["analysis"]).strip(), + "confidence": float(old["confidence"]), + "status": "待复核", + }, + ) + restored += 1 + + print( + f"subject={subject} restored={restored} skipped={skipped} " + f"safety_backup={safety_backup}" + ) + return restored, skipped + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--subject", required=True, choices=("数学", "英语", "408", "政治")) + parser.add_argument("--backup", required=True, type=Path) + args = parser.parse_args() + if not args.backup.is_file(): + parser.error(f"backup not found: {args.backup}") + restore(args.subject, args.backup.resolve()) + + +if __name__ == "__main__": + main() diff --git a/scripts/setup.sh b/scripts/setup.sh index 3c3654b..54f2ac3 100755 --- a/scripts/setup.sh +++ b/scripts/setup.sh @@ -22,6 +22,10 @@ npm install chmod +x "$PROJECT_DIR/打开错题本.command" "$PROJECT_DIR/scripts/setup.sh" ln -sfn "$PROJECT_DIR/打开错题本.command" "$HOME/Desktop/打开错题本.command" +if [ "$(uname -s)" = "Darwin" ]; then + "$PROJECT_DIR/.venv/bin/cuoti" service install +fi + echo -echo "安装完成。双击桌面的“打开错题本.command”,或运行:" -echo " $PROJECT_DIR/.venv/bin/cuoti serve" +echo "安装完成。macOS 会在登录后自动运行并打开错题本。" +echo " 查看状态:$PROJECT_DIR/.venv/bin/cuoti service status" diff --git a/src/cuoti/app.py b/src/cuoti/app.py index 7266066..dc99d41 100644 --- a/src/cuoti/app.py +++ b/src/cuoti/app.py @@ -15,6 +15,7 @@ from .db import SubjectStore, initialize_all, list_all, natural_text_sort_key from .export_jobs import export_job_output, get_export_job, start_export_job from .ingest import ingest_path +from .models import normalize_options, option_label, parse_options_text from .pdf_export import export_pdf from .rich_text import render_web_rich_text from .taxonomy import data_structure_choices, politics_choices @@ -24,12 +25,19 @@ ensure_directories(settings) initialize_all(settings) +MATH_SECTION_SHORT_LABELS = { + "高等数学": "高数", + "线性代数": "线代", + "概率论与数理统计": "概率", +} + PACKAGE_ROOT = Path(__file__).parent templates = Environment( loader=FileSystemLoader(PACKAGE_ROOT / "templates"), autoescape=select_autoescape(["html"]), ) templates.filters["richtext"] = lambda value: Markup(render_web_rich_text(value or "")) +templates.filters["option_label"] = option_label app = FastAPI(title="错题_auto", version="0.1.0") app.mount("/static", StaticFiles(directory=PACKAGE_ROOT / "static"), name="static") @@ -65,9 +73,10 @@ async def render_preview(request: Request) -> JSONResponse: raise HTTPException(400, "预览字段格式错误") if kind == "options": - options = [line.strip() for line in value.splitlines() if line.strip()] + options = parse_options_text(value) rendered = "".join( - f"{render_web_rich_text(option)}" for option in options + f'{option_label(index)}.{render_web_rich_text(option)}' + for index, option in enumerate(options) if option ) else: rendered = render_web_rich_text(value) @@ -82,6 +91,21 @@ def home(request: Request, subject: str = "") -> RedirectResponse: return RedirectResponse(f"/subject/{SUBJECT_SLUGS[selected]}{suffix}", status_code=303) +@app.get("/import", response_class=HTMLResponse) +def import_page(request: Request, imported: int = 0, subjects: str = "") -> HTMLResponse: + """Render the standalone ingestion workspace outside every subject library.""" + subject_by_slug = {slug: name for name, slug in SUBJECT_SLUGS.items()} + imported_subjects = [ + (subject_by_slug[slug], slug) + for slug in dict.fromkeys(item.strip() for item in subjects.split(",")) + if slug in subject_by_slug + ] + return render( + "import.html", request, + imported_count=max(0, imported), imported_subjects=imported_subjects, + ) + + @app.get("/subject/{subject_slug}", response_class=HTMLResponse) def subject_dashboard( request: Request, @@ -102,8 +126,29 @@ def subject_dashboard( } questions = list_all(filters, settings) all_items = list_all({"subject": subject}, settings) + chapters = sorted({item.chapter for item in all_items}, key=natural_text_sort_key) + chapter_sections = { + chapter: sorted( + {item.section for item in all_items if item.chapter == chapter}, + key=natural_text_sort_key, + ) + for chapter in chapters + } + chapter_options = [ + { + "value": chapter, + "label": ( + f"{' / '.join(MATH_SECTION_SHORT_LABELS.get(value, value) for value in chapter_sections[chapter])}" + f" · {chapter}" + if subject == "数学" + else chapter + ), + } + for chapter in chapters + ] facets = { - "chapters": sorted({item.chapter for item in all_items}, key=natural_text_sort_key), + "chapters": chapters, + "chapter_options": chapter_options, "sections": sorted({item.section for item in all_items}, key=natural_text_sort_key), "statuses": sorted({item.status for item in all_items}), } @@ -231,7 +276,7 @@ def question_update( "section": section.strip() or "待确认", "question_type": question_type.strip() or "未知题型", "question_text": question_text.strip(), - "options": [line.strip() for line in options.splitlines() if line.strip()], + "options": parse_options_text(options), "wrong_answer": wrong_answer.strip(), "correct_answer": correct_answer.strip(), "analysis": analysis.strip(), @@ -300,8 +345,13 @@ def upload_import( with target.open("wb") as stream: shutil.copyfileobj(upload.file, stream) saved.extend(ingest_path(target, provider, settings)) - subject = saved[0][0] if saved else "408" - return RedirectResponse(f"/subject/{SUBJECT_SLUGS[subject]}?imported=1", status_code=303) + imported_slugs = ",".join( + SUBJECT_SLUGS[subject] for subject in dict.fromkeys(subject for subject, _ in saved) + ) + return RedirectResponse( + f"/import?{urlencode({'imported': len(saved), 'subjects': imported_slugs})}", + status_code=303, + ) @app.get("/media/{subject}/{relative_path:path}") diff --git a/src/cuoti/cli.py b/src/cuoti/cli.py index ce5dff7..ea082e3 100644 --- a/src/cuoti/cli.py +++ b/src/cuoti/cli.py @@ -28,6 +28,11 @@ def build_parser() -> argparse.ArgumentParser: imported.add_argument("--source", type=Path, required=True) serve = sub.add_parser("serve", help="启动本地错题本网页") serve.add_argument("--no-open", action="store_true") + service = sub.add_parser("service", help="管理 macOS 登录自启与常驻服务") + service.add_argument( + "action", + choices=("install", "status", "uninstall", "open-when-ready"), + ) export = sub.add_parser("export", help="导出 PDF") export.add_argument("--variant", choices=("practice", "notebook"), required=True) export.add_argument("--subject", choices=SUBJECTS) @@ -85,6 +90,22 @@ def main(argv: list[str] | None = None) -> int: ], cwd=settings.project_root) except KeyboardInterrupt: return 0 + if args.command == "service": + from .macos_service import install, open_when_ready, status, uninstall + + if args.action == "install": + service_path, opener_path = install(settings) + print(f"已安装常驻服务:{service_path}") + print(f"已安装登录打开器:{opener_path}") + return 0 + if args.action == "status": + return status(settings) + if args.action == "uninstall": + service_path, opener_path = uninstall() + print(f"已移除服务配置:{service_path}") + print(f"已移除打开器配置:{opener_path}") + return 0 + return open_when_ready(settings) if args.command == "export": questions = list_all({"subject": args.subject or "", "chapter": args.chapter, "section": args.section}, settings) print(export_pdf(questions, args.variant, args.output, settings)) diff --git a/src/cuoti/db.py b/src/cuoti/db.py index 01b6a5f..979bb64 100644 --- a/src/cuoti/db.py +++ b/src/cuoti/db.py @@ -10,7 +10,17 @@ from typing import Any, Iterator from .config import SUBJECTS, Settings, ensure_directories -from .models import ExtractedQuestion, QuestionRecord, utc_now +from .models import ( + ExtractedQuestion, + QuestionRecord, + normalize_correct_answer, + normalize_error_reason, + normalize_options, + normalize_wrong_answer, + option_label, + utc_now, +) +from .rich_text import normalize_rich_text_spacing SCHEMA = """ @@ -44,6 +54,7 @@ relative_path TEXT NOT NULL, original_name TEXT NOT NULL, page_index INTEGER NOT NULL DEFAULT 1, + image_role TEXT NOT NULL DEFAULT 'question' CHECK(image_role IN ('question', 'work', 'solution', 'supplement')), include_in_practice INTEGER NOT NULL DEFAULT 0, UNIQUE(question_id, relative_path) ); @@ -107,6 +118,14 @@ def initialize(self) -> None: columns = {row["name"] for row in conn.execute("PRAGMA table_info(images)")} if "include_in_practice" not in columns: conn.execute("ALTER TABLE images ADD COLUMN include_in_practice INTEGER NOT NULL DEFAULT 0") + if "image_role" not in columns: + conn.execute("ALTER TABLE images ADD COLUMN image_role TEXT NOT NULL DEFAULT 'question'") + # Historical attached pages used a stable filename suffix. Mark + # them conservatively until the batch-specific migration can + # distinguish handwritten work from standard solutions. + conn.execute( + "UPDATE images SET image_role='supplement' WHERE relative_path LIKE '%-supplement.%'" + ) @contextmanager def connection(self) -> Iterator[sqlite3.Connection]: @@ -153,7 +172,9 @@ def insert(self, question: ExtractedQuestion, source_hash: str, source_file: str question_id = int(cursor.lastrowid) if image_path: conn.execute( - "INSERT OR IGNORE INTO images(question_id, relative_path, original_name, page_index) VALUES (?, ?, ?, ?)", + """INSERT OR IGNORE INTO images( + question_id, relative_path, original_name, page_index, image_role + ) VALUES (?, ?, ?, ?, 'question')""", (question_id, image_path, Path(source_file).name, question.source_page), ) self.write_markdown(question_id) @@ -165,8 +186,8 @@ def get(self, question_id: int) -> QuestionRecord | None: row = conn.execute("SELECT * FROM questions WHERE id=?", (question_id,)).fetchone() if not row: return None - images, practice_images = self._image_paths(conn, question_id) - return _row_to_record(row, images, practice_images) + images, practice_images, question_images, solution_images = self._image_paths(conn, question_id) + return _row_to_record(row, images, practice_images, question_images, solution_images) def list(self, filters: dict[str, str] | None = None) -> list[QuestionRecord]: self.initialize() @@ -194,8 +215,8 @@ def list(self, filters: dict[str, str] | None = None) -> list[QuestionRecord]: rows = conn.execute(sql, params).fetchall() result = [] for row in rows: - images, practice_images = self._image_paths(conn, row["id"]) - result.append(_row_to_record(row, images, practice_images)) + images, practice_images, question_images, solution_images = self._image_paths(conn, row["id"]) + result.append(_row_to_record(row, images, practice_images, question_images, solution_images)) return sorted(result, key=question_sort_key) def update(self, question_id: int, fields: dict[str, Any]) -> None: @@ -204,8 +225,19 @@ def update(self, question_id: int, fields: dict[str, Any]) -> None: "correct_answer", "analysis", "error_reason", "difficulty", "confidence", "status", } updates = {k: v for k, v in fields.items() if k in allowed} + for field in ("question_text", "wrong_answer", "correct_answer", "analysis", "error_reason"): + if field in updates: + updates[field] = ( + normalize_correct_answer(updates[field]) + if field == "correct_answer" + else normalize_wrong_answer(updates[field]) + if field == "wrong_answer" + else normalize_error_reason(updates[field]) + if field == "error_reason" + else normalize_rich_text_spacing(str(updates[field])).strip() + ) if "options" in fields: - updates["options_json"] = json.dumps(fields["options"], ensure_ascii=False) + updates["options_json"] = json.dumps(normalize_options(fields["options"]), ensure_ascii=False) if "knowledge_points" in fields: updates["knowledge_points_json"] = json.dumps(fields["knowledge_points"], ensure_ascii=False) if not updates: @@ -219,7 +251,11 @@ def update(self, question_id: int, fields: dict[str, Any]) -> None: def set_practice_images(self, question_id: int, relative_paths: list[str]) -> None: selected = set(relative_paths) with self.connection() as conn: - rows = conn.execute("SELECT relative_path FROM images WHERE question_id=?", (question_id,)).fetchall() + rows = conn.execute( + """SELECT relative_path FROM images + WHERE question_id=? AND image_role!='solution'""", + (question_id,), + ).fetchall() allowed = {row["relative_path"] for row in rows} conn.execute("UPDATE images SET include_in_practice=0 WHERE question_id=?", (question_id,)) for relative_path in selected & allowed: @@ -228,6 +264,30 @@ def set_practice_images(self, question_id: int, relative_paths: list[str]) -> No (question_id, relative_path), ) + def add_image( + self, + question_id: int, + relative_path: str, + original_name: str, + page_index: int, + image_role: str = "question", + ) -> None: + """Attach a derived review image with an explicit semantic role.""" + if image_role not in {"question", "work", "solution", "supplement"}: + raise ValueError(f"未知图片类型:{image_role}") + self.initialize() + with self.connection() as conn: + conn.execute( + """INSERT INTO images( + question_id, relative_path, original_name, page_index, image_role, include_in_practice + ) VALUES (?, ?, ?, ?, ?, 0) + ON CONFLICT(question_id, relative_path) DO UPDATE SET + original_name=excluded.original_name, + page_index=excluded.page_index, + image_role=excluded.image_role""", + (question_id, relative_path, original_name, page_index, image_role), + ) + def delete(self, question_id: int) -> Path | None: """Remove one mistaken record while keeping a recoverable local backup.""" self.initialize() @@ -269,14 +329,19 @@ def delete(self, question_id: int) -> Path | None: return backup_path @staticmethod - def _image_paths(conn: sqlite3.Connection, question_id: int) -> tuple[list[str], list[str]]: + def _image_paths( + conn: sqlite3.Connection, question_id: int, + ) -> tuple[list[str], list[str], list[str], list[str]]: rows = conn.execute( - "SELECT relative_path, include_in_practice FROM images WHERE question_id=? ORDER BY page_index, id", + """SELECT relative_path, include_in_practice, image_role + FROM images WHERE question_id=? ORDER BY page_index, id""", (question_id,), ).fetchall() return ( [row["relative_path"] for row in rows], [row["relative_path"] for row in rows if row["include_in_practice"]], + [row["relative_path"] for row in rows if row["image_role"] != "solution"], + [row["relative_path"] for row in rows if row["image_role"] == "solution"], ) def write_markdown(self, question_id: int) -> Path: @@ -284,8 +349,16 @@ def write_markdown(self, question_id: int) -> Path: if record is None: raise KeyError(question_id) path = self.root / "markdown" / f"{question_id:06d}.md" - options = "\n".join(f"- {item}" for item in record.options) or "(无选项)" - images = "\n".join(f"![原题图片](../{item})" for item in record.image_paths) + options = "\n".join( + f"- {option_label(index)}. {item}" + for index, item in enumerate(record.options) if item + ) or "(无选项)" + question_images = "\n".join( + f"![题目或作答照片](../{item})" for item in record.question_image_paths + ) + solution_images = "\n".join( + f"![答案或解析照片](../{item})" for item in record.solution_image_paths + ) knowledge = "、".join(record.knowledge_points) or "待补充" content = f"""--- id: {record.id} @@ -307,7 +380,7 @@ def write_markdown(self, question_id: int) -> Path: {options} -{images} +{question_images} ## 我的错误答案 @@ -321,6 +394,8 @@ def write_markdown(self, question_id: int) -> Path: {record.analysis or '待补充'} +{solution_images} + ## 错因与知识点 - 错因:{record.error_reason or '待补充'} @@ -334,16 +409,24 @@ def get_without_init(self, question_id: int) -> QuestionRecord | None: row = conn.execute("SELECT * FROM questions WHERE id=?", (question_id,)).fetchone() if not row: return None - images, practice_images = self._image_paths(conn, question_id) - return _row_to_record(row, images, practice_images) + images, practice_images, question_images, solution_images = self._image_paths(conn, question_id) + return _row_to_record(row, images, practice_images, question_images, solution_images) -def _row_to_record(row: sqlite3.Row, images: list[str], practice_images: list[str]) -> QuestionRecord: +def _row_to_record( + row: sqlite3.Row, + images: list[str], + practice_images: list[str], + question_images: list[str], + solution_images: list[str], +) -> QuestionRecord: data = dict(row) data["options"] = json.loads(data.pop("options_json") or "[]") data["knowledge_points"] = json.loads(data.pop("knowledge_points_json") or "[]") data["image_paths"] = images data["practice_image_paths"] = practice_images + data["question_image_paths"] = question_images + data["solution_image_paths"] = solution_images data["needs_review"] = data["status"] == "待复核" return QuestionRecord.model_validate(data) diff --git a/src/cuoti/ingest.py b/src/cuoti/ingest.py index e3c272b..5d8dcae 100644 --- a/src/cuoti/ingest.py +++ b/src/cuoti/ingest.py @@ -76,10 +76,15 @@ def normalize_display_image(source: Path, target: Path) -> int: ANALYSIS_PROMPT = """你正在整理中国考研纸质错题。逐题读取图片,不能把多道题合成一道。 +先判断题目是否应收录:明显错题、订正题和题号被圈出的题需要收录;答案位置空白但旁边有明确对钩,且题号未圈、没有红笔订正或其他错误标记时,表示已经掌握,必须排除,禁止仅因没有手写答案就把它当作错题。 必须忠实保留题干、选项、学生错误答案、标准正确答案和手写批注;数学公式写成 LaTeX,行内用 $...$、独立公式用 $$...$$。 +若输入中包含答案或标准解析页,correct_answer 必须填写该页给出的明确最终答案或结论,禁止写“见解析”“见标准解析”“待确认”或任何指向图片的占位文字;analysis 必须按原解的顺序、方法和计算步骤忠实转写,不得概括、改写、缩写或用自己的解法替代。仅当输入中确实没有标准解析时才能自行补充,并在解析开头标明“【补充解析】”。标准解析看不清、缺页或与题目无法对应时,不得猜测或改用自写解析;correct_answer 和 analysis 留空,confidence 不得高于 0.50,needs_review 必须为 true,并停止自动确认交由人工复核。 +“学生错误答案”字段只填写学生最终写出的具体答案。若学生答题区为空白、只有解题思路而未形成最终答案,或只看到红笔写出的答案/订正而没有可确认的原答案,统一写“不会”,禁止写“待确认(见原图红笔答案或订正)”“思路错误”或作答过程描述;若能看清原来的具体错误答案,则忠实记录原答案。作答过程中的错误只写入 error_reason。 +选择题的正确答案以 A、B、C、D 等裸字母开头,不要写成 (A)、(A)等带括号形式。 +独立公式的 $$...$$ 与前后正文之间只换一行,不要留空白行;连续独立公式之间也不要插入空白行。 科目只能是 数学、英语、408、政治。章节与板块尽量使用考研常见命名。 若原图缺少解析,请给出可独立理解、步骤完整的解析;不要伪造看不清的内容,看不清处写 [无法辨认] 并将 needs_review 设为 true。 -error_reason 要总结真正的知识或方法漏洞,knowledge_points 使用短标签。source_page 从 1 开始。 +error_reason 只能根据图片中可见的学生解答过程总结具体知识或方法错误;若没有解答过程、只有题号圈选、答案订正、未作答或标准解析,则 error_reason 必须为空字符串,不能写复盘、待核对或未独立完成等占位话术。knowledge_points 使用短标签。source_page 从 1 开始。 """ diff --git a/src/cuoti/macos_service.py b/src/cuoti/macos_service.py new file mode 100644 index 0000000..2ecf6be --- /dev/null +++ b/src/cuoti/macos_service.py @@ -0,0 +1,291 @@ +from __future__ import annotations + +import os +import plistlib +import re +import signal +import shlex +import subprocess +import sys +import tempfile +import time +from pathlib import Path +from urllib.error import URLError +from urllib.request import urlopen + +from .config import Settings + + +SERVICE_LABEL = "com.nemoyu.cuoti-auto" +OPENER_LABEL = "com.nemoyu.cuoti-auto.open" +LAUNCHCTL = Path("/bin/launchctl") +OPEN = Path("/usr/bin/open") + + +def _require_macos() -> None: + if sys.platform != "darwin": + raise RuntimeError("自动常驻服务目前只支持 macOS LaunchAgent") + + +def _browser_url(settings: Settings) -> str: + host = settings.host + if host in {"0.0.0.0", "::", "[::]"}: + host = "127.0.0.1" + return f"http://{host}:{settings.port}" + + +def _launch_agent_paths() -> tuple[Path, Path]: + root = Path.home() / "Library" / "LaunchAgents" + return root / f"{SERVICE_LABEL}.plist", root / f"{OPENER_LABEL}.plist" + + +def _runtime_paths() -> dict[str, Path]: + root = Path.home() / "Library" / "Application Support" / "cuoti-auto" + return { + "root": root, + "monitor": root / "monitor.zsh", + "launcher": root / "start-service.command", + "supervisor": root / "supervisor.zsh", + "opener": root / "open-when-ready.zsh", + "pid": root / "supervisor.pid", + } + + +def build_launch_agent_plists( + settings: Settings, + executable: Path | None = None, + runtime_root: Path | None = None, +) -> tuple[dict[str, object], dict[str, object]]: + """构造常驻服务与单次打开器。 + + 打开器不设 KeepAlive,因此服务异常重启时不会反复打开浏览器标签; + 它只会在 LaunchAgent 登录会话加载时执行一次。 + """ + del executable # 服务由 Terminal 启动,避免 LaunchAgent 被 macOS 拒绝访问桌面目录。 + runtime_root = runtime_root or _runtime_paths()["root"] + log_root = settings.project_root / "output" / "logs" + environment = { + "CUOTI_PROJECT_ROOT": str(settings.project_root), + "CUOTI_DESKTOP": str(settings.desktop), + "CUOTI_HOST": settings.host, + "CUOTI_PORT": str(settings.port), + "CUOTI_OPENAI_MODEL": settings.openai_model, + "PATH": os.environ.get( + "PATH", + "/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin", + ), + "PYTHONUNBUFFERED": "1", + } + common: dict[str, object] = { + "WorkingDirectory": str(runtime_root), + "EnvironmentVariables": environment, + "LimitLoadToSessionType": "Aqua", + "ProcessType": "Background", + "ThrottleInterval": 5, + } + service = { + **common, + "Label": SERVICE_LABEL, + "ProgramArguments": ["/bin/zsh", str(runtime_root / "monitor.zsh")], + "RunAtLoad": True, + "KeepAlive": True, + "StandardOutPath": str(log_root / "service.log"), + "StandardErrorPath": str(log_root / "service.error.log"), + } + opener = { + **common, + "Label": OPENER_LABEL, + "ProgramArguments": ["/bin/zsh", str(runtime_root / "open-when-ready.zsh")], + "RunAtLoad": True, + "StandardOutPath": str(log_root / "opener.log"), + "StandardErrorPath": str(log_root / "opener.error.log"), + } + return service, opener + + +def _write_plist(path: Path, payload: dict[str, object]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with tempfile.NamedTemporaryFile(dir=path.parent, prefix=f".{path.name}.", delete=False) as handle: + temporary = Path(handle.name) + plistlib.dump(payload, handle, sort_keys=False) + temporary.chmod(0o644) + temporary.replace(path) + + +def _write_script(path: Path, content: str) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(content, encoding="utf-8") + path.chmod(0o700) + + +def _write_runtime_scripts(settings: Settings) -> dict[str, Path]: + """生成不位于受保护桌面目录中的轻量启动脚本。 + + macOS 会拒绝 launchd 直接读取 Desktop 下的 Python 虚拟环境和数据库。 + 健康监视器因此只访问 localhost;需要拉起服务时,交由用户已授权的 + Terminal 启动后台 supervisor,无需给 Python 授予“完全磁盘访问”。 + """ + paths = _runtime_paths() + url = shlex.quote(_browser_url(settings)) + launcher = shlex.quote(str(paths["launcher"])) + supervisor = shlex.quote(str(paths["supervisor"])) + pid_path = shlex.quote(str(paths["pid"])) + executable = shlex.quote(str(settings.project_root / ".venv" / "bin" / "cuoti")) + service_log = shlex.quote(str(settings.project_root / "output" / "logs" / "service.log")) + service_error = shlex.quote(str(settings.project_root / "output" / "logs" / "service.error.log")) + + _write_script(paths["monitor"], f"""#!/bin/zsh +set -u +URL={url} +LAUNCHER={launcher} +while true; do + if ! /usr/bin/curl --fail --silent --max-time 1 "$URL" >/dev/null 2>&1; then + /usr/bin/open -g "$LAUNCHER" + /bin/sleep 8 + else + /bin/sleep 3 + fi +done +""") + _write_script(paths["launcher"], f"""#!/bin/zsh +set -u +URL={url} +SUPERVISOR={supervisor} +PID_FILE={pid_path} +if /usr/bin/curl --fail --silent --max-time 1 "$URL" >/dev/null 2>&1; then + exit 0 +fi +if [ -f "$PID_FILE" ]; then + SUPERVISOR_PID="$(/bin/cat "$PID_FILE" 2>/dev/null || true)" + if [ -n "$SUPERVISOR_PID" ] && /bin/kill -0 "$SUPERVISOR_PID" 2>/dev/null; then + exit 0 + fi + /bin/rm -f "$PID_FILE" +fi +/usr/bin/nohup /bin/zsh "$SUPERVISOR" /dev/null 2>&1 & +exit 0 +""") + _write_script(paths["supervisor"], f"""#!/bin/zsh +set -u +URL={url} +PID_FILE={pid_path} +CUOTI={executable} +SERVICE_LOG={service_log} +SERVICE_ERROR={service_error} +echo $$ > "$PID_FILE" +trap '/bin/rm -f "$PID_FILE"' EXIT INT TERM +while true; do + if /usr/bin/curl --fail --silent --max-time 1 "$URL" >/dev/null 2>&1; then + /bin/sleep 3 + continue + fi + "$CUOTI" serve --no-open >>"$SERVICE_LOG" 2>>"$SERVICE_ERROR" + /bin/sleep 3 +done +""") + _write_script(paths["opener"], f"""#!/bin/zsh +set -u +URL={url} +for attempt in {{1..240}}; do + if /usr/bin/curl --fail --silent --max-time 1 "$URL" >/dev/null 2>&1; then + exec /usr/bin/open "$URL" + fi + /bin/sleep 0.25 +done +echo "等待服务就绪超时:$URL" >&2 +exit 1 +""") + return paths + + +def _launchctl(*arguments: str, check: bool = True) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [str(LAUNCHCTL), *arguments], + check=check, + text=True, + capture_output=True, + ) + + +def install(settings: Settings) -> tuple[Path, Path]: + _require_macos() + executable = settings.project_root / ".venv" / "bin" / "cuoti" + if not executable.is_file() or not os.access(executable, os.X_OK): + raise RuntimeError(f"未找到可执行的错题服务:{executable}") + + log_root = settings.project_root / "output" / "logs" + log_root.mkdir(parents=True, exist_ok=True) + runtime_paths = _write_runtime_scripts(settings) + service_path, opener_path = _launch_agent_paths() + service, opener = build_launch_agent_plists(settings, executable, runtime_paths["root"]) + _write_plist(service_path, service) + _write_plist(opener_path, opener) + + domain = f"gui/{os.getuid()}" + for label in (OPENER_LABEL, SERVICE_LABEL): + _launchctl("bootout", f"{domain}/{label}", check=False) + for path in (service_path, opener_path): + _launchctl("bootstrap", domain, str(path)) + _launchctl("enable", f"{domain}/{SERVICE_LABEL}") + _launchctl("enable", f"{domain}/{OPENER_LABEL}") + _launchctl("kickstart", "-k", f"{domain}/{SERVICE_LABEL}") + return service_path, opener_path + + +def uninstall() -> tuple[Path, Path]: + _require_macos() + service_path, opener_path = _launch_agent_paths() + domain = f"gui/{os.getuid()}" + for label in (OPENER_LABEL, SERVICE_LABEL): + _launchctl("bootout", f"{domain}/{label}", check=False) + pid_path = _runtime_paths()["pid"] + try: + pid_text = pid_path.read_text(encoding="utf-8").strip() + if pid_text.isascii() and pid_text.isdigit(): + os.kill(int(pid_text), signal.SIGTERM) + except (FileNotFoundError, ProcessLookupError, PermissionError): + pass + pid_path.unlink(missing_ok=True) + service_path.unlink(missing_ok=True) + opener_path.unlink(missing_ok=True) + return service_path, opener_path + + +def _service_state(label: str) -> tuple[bool, str, str]: + domain = f"gui/{os.getuid()}" + result = _launchctl("print", f"{domain}/{label}", check=False) + if result.returncode != 0: + return False, "unloaded", "" + state_match = re.search(r"\bstate = ([^\n]+)", result.stdout) + pid_match = re.search(r"\bpid = ([0-9]+)", result.stdout) + return True, state_match.group(1).strip() if state_match else "loaded", pid_match.group(1) if pid_match else "" + + +def status(settings: Settings) -> int: + _require_macos() + loaded, state, pid = _service_state(SERVICE_LABEL) + endpoint_ok = endpoint_ready(_browser_url(settings), timeout=1.0) + pid_text = f" pid={pid}" if pid else "" + print(f"LaunchAgent: {'已加载' if loaded else '未加载'} ({state}{pid_text})") + print(f"网页: {'可访问' if endpoint_ok else '不可访问'} ({_browser_url(settings)})") + return 0 if loaded and endpoint_ok else 1 + + +def endpoint_ready(url: str, timeout: float = 1.0) -> bool: + try: + with urlopen(url, timeout=timeout) as response: + return 200 <= response.status < 400 + except (OSError, URLError): + return False + + +def open_when_ready(settings: Settings, wait_seconds: float = 60.0) -> int: + _require_macos() + url = _browser_url(settings) + deadline = time.monotonic() + wait_seconds + while time.monotonic() < deadline: + if endpoint_ready(url): + return subprocess.call([str(OPEN), url]) + time.sleep(0.25) + print(f"等待服务就绪超时:{url}", file=sys.stderr) + return 1 diff --git a/src/cuoti/models.py b/src/cuoti/models.py index 5b5f63b..5a8fa64 100644 --- a/src/cuoti/models.py +++ b/src/cuoti/models.py @@ -1,11 +1,108 @@ from __future__ import annotations +import re from datetime import datetime, timezone from typing import Literal from pydantic import BaseModel, Field, field_validator from .config import SUBJECTS +from .rich_text import normalize_rich_text_spacing + + +OPTION_PREFIX_RE = re.compile( + r"^\s*(?:[((]\s*(?P[A-Z])\s*[))]|(?P[A-Z])\s*[..、::))])\s*", + re.IGNORECASE, +) +CORRECT_ANSWER_OPTION_RE = re.compile( + r"^\s*[((]\s*(?P