This commit is contained in:
Your Name
2026-09-21 10:26:14 +08:00
parent bbe3e870a1
commit dbf474ddd7
67 changed files with 10636 additions and 279 deletions
@@ -0,0 +1,478 @@
"""The composed pages must show the saved batch exactly, including what is missing from it."""
from __future__ import annotations
from typing import Any
import pytest
from PySide6.QtWidgets import QApplication, QLabel
from doctor_workstation.ui.dialogs import issued_prescription_ai_pages as pages
@pytest.fixture(scope="module")
def application() -> QApplication:
return QApplication.instance() or QApplication([])
def comparison_rows(model: str) -> list[dict[str, Any]]:
doses = {"qwen": {"生地黄": 15, "生麦冬": 12, "麸炒白术": 12}, "openai": {"生地黄": 12, "茯苓": 8}}[model]
contributions = {"qwen": {"生地黄": 0.94, "生麦冬": 1.0}, "openai": {"生地黄": 0.75}}[model]
doctor = {"生地黄": 16, "生麦冬": 12, "红参片": 6, "茯苓": 10}
rows = []
for name in sorted(set(doses) | set(doctor)):
rows.append({
"name": name, "unit": "g",
"doctor_dosage": doctor.get(name),
"candidate_dosage": doses.get(name),
"contribution": contributions.get(name),
})
return rows
def batch(**overrides: Any) -> dict[str, Any]:
value = {
"id": 13, "status": "success", "coverage_status": "partial", "comparison_type": "latest_context",
"source_summary": {"attachment_count": 4, "diagnoses_count": 1, "video_calls_count": 3,
"source_record_count": 9},
"missing": [{"code": "TRANSCRIPT_NOT_VERIFIED_COMPLETE", "critical": True},
{"code": "TRANSCRIPT_NOT_VERIFIED_COMPLETE", "critical": True},
{"code": "ARCHIVE_SYNC_WATERMARK_UNAVAILABLE", "critical": False}],
"models": {
"qwen": {"status": "success", "algorithm_version": "prescription-soft-dice-v1.1.0",
"prompt_version": "manual-prescription-required-candidate-v4",
"comparison": {"status": "comparable", "rows": comparison_rows("qwen")},
"coverage": {"files": [{"status": "processed"}, {"status": "processed"},
{"status": "processed"}, {"status": "restricted"}]},
"progress": {"stage_label": "处理完成", "elapsed_seconds": 135, "attempt": 1},
"usage": {"total_calls": 2, "calls": [
{"stage": "text:0", "ok": True, "latency_ms": 3400, "file_count": 0,
"usage": {"completion_tokens": 1020}, "error_code": ""},
{"stage": "final", "ok": False, "latency_ms": 14400, "file_count": 0,
"usage": {"completion_tokens": 5617}, "error_code": "INVALID_REPORT_OUTPUT"}]}},
"openai": {"status": "success", "comparison": {"status": "comparable", "rows": comparison_rows("openai")},
"progress": {"stage_label": "处理完成", "elapsed_seconds": 593, "attempt": 1},
"usage": {"calls": []}},
},
}
value.update(overrides)
return value
# ---------------------------------------------------------------- candidates page
@pytest.fixture
def per_herb(application: QApplication) -> pages.CandidatesPage:
widget = pages.CandidatesPage()
widget.resize(1200, 700)
widget.show()
application.processEvents()
yield widget
widget.close()
def _dose_cells(page: pages.CandidatesPage, row: int) -> tuple[str, str]:
return (page.table.cellWidget(row, 1).dose, page.table.cellWidget(row, 2).dose)
def _verdict(page: pages.CandidatesPage, row: int) -> str:
return page.table.cellWidget(row, 4).findChildren(QLabel)[0].text()
def test_per_herb_merges_both_models_into_one_row(per_herb: pages.CandidatesPage) -> None:
per_herb.set_batch(batch())
names = [per_herb.table.item(row, 0).text() for row in range(per_herb.table.rowCount())]
assert names.count("生地黄") == 1
row = names.index("生地黄")
assert _dose_cells(per_herb, row) == ("15 g", "12 g")
assert per_herb.table.item(row, 3).text() == "16 g"
assert per_herb.table.cellWidget(row, 1).contribution == 0.94
assert per_herb.table.cellWidget(row, 2).contribution == 0.75
def test_per_herb_marks_absence_without_inventing_a_dose(per_herb: pages.CandidatesPage) -> None:
per_herb.set_batch(batch())
names = [per_herb.table.item(row, 0).text() for row in range(per_herb.table.rowCount())]
only_doctor = names.index("红参片")
assert _dose_cells(per_herb, only_doctor) == ("—", "—")
assert per_herb.table.cellWidget(only_doctor, 1).contribution is None
assert _verdict(per_herb, only_doctor) == "仅医方使用"
added = names.index("麸炒白术")
assert per_herb.table.item(added, 3).text() == "未收录"
assert _verdict(per_herb, added) == "仅 千问 收录"
def test_per_herb_conclusion_states_a_dose_gap_over_the_threshold(per_herb: pages.CandidatesPage) -> None:
per_herb.set_batch(batch())
names = [per_herb.table.item(row, 0).text() for row in range(per_herb.table.rowCount())]
row = names.index("生地黄")
assert _verdict(per_herb, row) == "两模型均收录" # 医方 16,两侧差 1 与 4,未过 5 克阈值
source = batch()
for model in source["models"].values():
for entry in model["comparison"]["rows"]:
if entry["name"] == "生地黄":
entry["candidate_dosage"] = 9
per_herb.set_batch(source)
names = [per_herb.table.item(index, 0).text() for index in range(per_herb.table.rowCount())]
assert _verdict(per_herb, names.index("生地黄")) == "剂量分歧 7 g"
def test_per_herb_filters_count_and_narrow_the_table(
per_herb: pages.CandidatesPage, application: QApplication) -> None:
per_herb.set_batch(batch())
total = per_herb.table.rowCount()
assert per_herb.filter_buttons["all"].text() == f"全部 {total}"
per_herb.search.setText("生地黄")
application.processEvents()
assert per_herb.table.rowCount() == 1
per_herb.search.clear()
per_herb.filter_buttons["doctor"].click()
application.processEvents()
assert 0 < per_herb.table.rowCount() < total
assert all(_verdict(per_herb, row) in {"仅医方使用", "两模型均未收录"}
for row in range(per_herb.table.rowCount()))
per_herb.filter_buttons["single"].click()
application.processEvents()
assert all("仅 " in _verdict(per_herb, row) for row in range(per_herb.table.rowCount()))
def test_per_herb_cards_show_each_saved_prescription(per_herb: pages.CandidatesPage) -> None:
per_herb.set_batch(batch(doctor_snapshot={"prescription": {
"herbs": [{"name": "生地黄", "dosage": 16, "unit": "g"}, {"name": "红参片", "dosage": 6, "unit": "g"}],
"prescription_type": "浓缩水丸", "dose_count": 1, "usage_instruction": "每日1剂,水煎分服。"}}))
doctor = per_herb.cards["doctor"]
assert doctor.summary.text() == "2 味 · 浓缩水丸"
assert "生地黄" in doctor.herbs.text() and "16 g" in doctor.herbs.text()
assert "每日1剂" in doctor.usage.text()
assert per_herb.cards["qwen"].summary.text() == "尚无候选方"
def test_per_herb_card_grows_for_a_long_usage_note(per_herb: pages.CandidatesPage,
application: QApplication) -> None:
"""A card must not cap itself and cut the herb list or the 方义 in half."""
note = "方中生黄芪益气固表,生地黄、生麦冬滋阴清热。" * 8
per_herb.set_batch(batch(doctor_snapshot={"prescription": {
"herbs": [{"name": f"药{index}", "dosage": 15, "unit": "g"} for index in range(20)],
"prescription_type": "浓缩水丸", "dose_count": 1, "usage_instruction": note}}))
per_herb.resize(1100, 700)
application.processEvents()
card = per_herb.cards["doctor"]
assert "另有 15 味" in card.herbs.text()
assert card.usage.height() >= card.usage.heightForWidth(card.usage.width())
assert card.height() >= card.herbs.height() + card.usage.height()
def test_per_herb_says_when_no_prescription_was_saved(per_herb: pages.CandidatesPage) -> None:
per_herb.set_batch({})
assert per_herb.cards["doctor"].summary.text() == "未保存原方"
assert per_herb.cards["doctor"].herbs.text() == "尚未保存药味"
assert per_herb.table.rowCount() == 0
# ---------------------------------------------------------------- sources page
@pytest.fixture
def sources(application: QApplication) -> pages.SourcesPage:
widget = pages.SourcesPage()
widget.resize(900, 600)
widget.show()
application.processEvents()
yield widget
widget.close()
def test_sources_groups_gaps_by_type_and_keeps_criticality(sources: pages.SourcesPage) -> None:
sources.set_batch(batch())
rows = {sources.gaps.item(row, 0).text(): (sources.gaps.item(row, 1).text(), sources.gaps.item(row, 2).text())
for row in range(sources.gaps.rowCount())}
transcript = next(key for key in rows if "转写" in key)
assert rows[transcript] == ("关键", "2")
archive = next(key for key in rows if "归档" in key)
assert rows[archive][0] == "一般"
def _composition(sources: pages.SourcesPage) -> dict[str, str]:
rows = {}
for index in range(sources.composition_layout.count()):
widget = sources.composition_layout.itemAt(index).widget()
labels = widget.findChildren(QLabel)
rows[labels[0].text()] = labels[-1].text()
return rows
def test_sources_reports_attachment_reading_and_composition(sources: pages.SourcesPage) -> None:
sources.set_batch(batch())
assert "模型实际读取 3 个" in sources.attachment_note.text()
assert sources.waffle.accessibleDescription() == "附件 4 个:已读 3,受限或不支持 1"
assert _composition(sources)["问诊通话"] == "3"
assert "soft-dice" in sources.meta.text()
def test_sources_reports_what_each_model_managed_to_read(sources: pages.SourcesPage) -> None:
sources.set_batch(batch())
rows = {sources.per_model.item(row, 0).text(): (sources.per_model.item(row, 1).text(),
sources.per_model.item(row, 3).text())
for row in range(sources.per_model.rowCount())}
assert rows["千问"] == ("3", "75%")
assert "已读取 3" in sources.attachment_legend.text()
def test_sources_meta_states_the_cutoff_and_reads_codes_in_chinese(sources: pages.SourcesPage) -> None:
sources.set_batch(batch(cutoff_at="2026-09-10 15:29"))
text = sources.meta.text()
assert "资料截止:2026-09-10 15:29" in text
assert "覆盖状态:部分资料缺失" in text # partial 在覆盖语境里说的是资料,不是进度
assert "对照类型:最新资料对照" in text
sources.set_batch(batch())
assert "资料截止:—" in sources.meta.text()
def test_sources_stays_empty_without_a_batch(sources: pages.SourcesPage) -> None:
sources.set_batch({})
assert _composition(sources) == {}
assert sources.gaps.rowCount() == 0
assert sources.attachment_note.text() == "本次没有附件"
assert not sources.gap_note.isVisible()
def test_sources_puts_critical_gaps_first_and_counts_them(sources: pages.SourcesPage) -> None:
sources.set_batch(batch())
assert "关键" in sources.gaps.item(0, 1).text()
assert "3 项 · 关键 2" in sources.gap_title.findChildren(QLabel)[-1].text()
# ---------------------------------------------------------------- progress page
@pytest.fixture
def progress(application: QApplication) -> pages.ProgressPage:
widget = pages.ProgressPage()
widget.resize(900, 600)
widget.show()
application.processEvents()
yield widget
widget.close()
def test_progress_lists_every_call_with_its_outcome(progress: pages.ProgressPage) -> None:
progress.set_batch(batch())
assert progress.calls.rowCount() == 2
assert progress.calls.item(0, 0).text() == "千问"
assert progress.calls.item(0, 1).text() == "文字资料 1" # 阶段键不直接露出
assert progress.calls.item(1, 1).text() == "生成候选与报告"
assert progress.calls.item(0, 3).text() == "3.4 s"
assert progress.calls.item(0, 6).text() == "通过"
assert progress.calls.item(1, 5).text() == "5617"
# 最慢的一次调用占满耗时分布条,其余按比例
assert progress.calls.cellWidget(1, 2)._fraction == 1.0
assert progress.calls.cellWidget(0, 2)._fraction < 0.3
assert "INVALID_REPORT_OUTPUT" not in progress.calls.item(1, 5).text()
def test_progress_shows_each_model_stage_and_attempt(progress: pages.ProgressPage) -> None:
progress.set_batch(batch())
assert "处理完成" in progress.stage_labels["qwen"].text()
assert "第 1 次尝试" in progress.stage_labels["qwen"].text()
assert "已用时 2 分 15 秒" in progress.stage_labels["qwen"].text()
assert "已用时 9 分 53 秒" in progress.stage_labels["openai"].text()
# ---------------------------------------------------------------- history page
@pytest.fixture
def history(application: QApplication) -> pages.HistoryPage:
widget = pages.HistoryPage()
widget.resize(900, 600)
widget.show()
application.processEvents()
yield widget
widget.close()
def test_history_orders_by_time_and_keeps_uncomparable_slots_empty(history: pages.HistoryPage) -> None:
history.set_history([
{"id": 13, "created_at": 300, "status": "success", "comparison_type": "latest_context",
"models": {"qwen": {"score": 18.9, "comparison_status": "comparable",
"algorithm_version": "prescription-soft-dice-v1.1.0"},
"openai": {"score": 16.7, "comparison_status": "comparable"}}},
{"id": 11, "created_at": 100, "status": "failed",
"models": {"qwen": {"score": None, "comparison_status": "not_comparable"},
"openai": {"score": None, "comparison_status": "not_comparable"}}},
])
description = history.chart.accessibleDescription()
assert description.index("11") < description.index("13") # oldest first on the chart
assert "11 千问 — OpenAI —" in description
assert history.table.item(0, 0).text() == "#13" # newest first in the list
assert history.table.item(0, 3).text() == "18.9%"
assert history.table.item(1, 3).text() == "—"
assert history.table.item(0, 5).text() == "v1.1.0"
assert history.chart.has_data()
def test_history_reads_the_score_from_either_payload_shape(history: pages.HistoryPage) -> None:
"""列表接口把分数摊平,详情接口留在 comparison 里,两种都要认。"""
history.set_history([
{"id": 40, "created_at": "2026-09-10 15:29:00", "status": "success", "comparison_type": "non_independent",
"models": {"qwen": {"comparison": {"status": "comparable", "score": 94.44}},
"openai": {"comparison": {"status": "not_comparable", "score": 93.3}}}},
{"id": 39, "created_at": "2026-09-09 09:00:00", "status": "success",
"models": {"qwen": {"score": 42.9, "comparison_status": "comparable"}}},
])
assert history.table.item(0, 0).text() == "#40"
assert history.table.item(0, 3).text() == "94.4%"
assert history.table.item(0, 4).text() == "—" # 不可比就不给分,哪怕载荷里带着数字
assert "分层" in history.strata.text() or "同一套" in history.strata.text()
assert history.table.item(1, 3).text() == "42.9%"
description = history.chart.accessibleDescription()
assert description.index("39") < description.index("40")
def test_history_without_any_comparable_batch_draws_nothing(history: pages.HistoryPage) -> None:
history.set_history([{"id": 1, "created_at": 1, "models": {"qwen": {"comparison_status": "not_comparable"}}}])
assert not history.chart.has_data()
assert history.table.rowCount() == 1
# ---------------------------------------------------------------- statistics panel
@pytest.fixture
def statistics(application: QApplication) -> pages.StatisticsPanel:
widget = pages.StatisticsPanel()
widget.resize(1000, 600)
widget.show()
application.processEvents()
yield widget
widget.close()
def statistics_payload() -> dict[str, Any]:
return {
"total_count": 10, "patient_count": 8,
"doctors": [
{"doctor_id": 26, "doctor_name": "何医生", "total_count": 6, "patient_count": 5, "paired_count": 3,
"models": {"qwen": {"eligible_count": 4, "mean": 18.4, "median": 16.9,
"excluded_reasons": {"SOURCE_HISTORY_VERSIONS_UNAVAILABLE": 2}},
"openai": {"eligible_count": 3, "mean": 21.0, "median": 19.6,
"excluded_reasons": {"SOURCE_HISTORY_VERSIONS_UNAVAILABLE": 3}}},
"review": {"evaluated_count": 0, "qualified_count": 0, "qualified_rate": None}},
{"doctor_id": 31, "doctor_name": "李医生", "total_count": 4, "patient_count": 3, "paired_count": 1,
"models": {"qwen": {"eligible_count": 2, "mean": 25.0, "excluded_reasons": {"incomplete_coverage": 2}},
"openai": {"eligible_count": 0, "mean": None, "excluded_reasons": {"incomplete_coverage": 4}}},
"review": {"evaluated_count": 2, "qualified_count": 1, "qualified_rate": 50.0}},
],
}
def test_statistics_headline_counts_and_coverage(statistics: pages.StatisticsPanel) -> None:
statistics.set_statistics(statistics_payload())
assert statistics.kpi_values["events"].text() == "10"
assert "涉及患者 8 人" in statistics.kpi_notes["events"].text()
assert statistics.kpi_values["qwen"].text() == "6"
assert "覆盖率 60.0%" in statistics.kpi_notes["qwen"].text()
assert statistics.kpi_values["openai"].text() == "3"
def test_statistics_review_rate_needs_a_review_sample(statistics: pages.StatisticsPanel) -> None:
payload = statistics_payload()
for doctor in payload["doctors"]:
doctor["review"] = {"evaluated_count": 0, "qualified_count": 0, "qualified_rate": None}
statistics.set_statistics(payload)
assert statistics.kpi_values["review"].text() == "—"
assert "尚未建立复核样本" in statistics.kpi_notes["review"].text()
statistics.set_statistics(statistics_payload())
assert statistics.kpi_values["review"].text() == "50.0%"
def _bars(layout) -> list[str]:
return [layout.itemAt(index).widget().accessibleDescription() for index in range(layout.count())]
def test_statistics_funnel_and_exclusions_are_aggregated(statistics: pages.StatisticsPanel) -> None:
statistics.set_statistics(statistics_payload())
funnel = _bars(statistics.funnel_layout)
assert "范围内开方事件 10" in funnel
assert "千问 有效基线比较 6" in funnel
assert "两模型配对共同样本 4" in funnel
reasons = _bars(statistics.exclusion_layout)
assert any(text.endswith(" 5") for text in reasons) # 来源历史版本无法重建 2 + 3
assert all("SOURCE_HISTORY" not in text for text in reasons)
def test_statistics_distribution_sums_the_saved_bins(statistics: pages.StatisticsPanel) -> None:
payload = statistics_payload()
payload["doctors"][0]["models"]["qwen"]["distribution"] = {"[0,20)": 3, "[20,40)": 1}
payload["doctors"][1]["models"]["qwen"]["distribution"] = {"[0,20)": 2}
statistics.set_statistics(payload)
assert statistics.distribution.has_data()
assert "千问 5/1/0/0/0" in statistics.distribution.accessibleDescription()
assert statistics.distribution.isVisible()
def test_statistics_draws_no_distribution_without_samples(statistics: pages.StatisticsPanel) -> None:
statistics.set_statistics(statistics_payload())
assert not statistics.distribution.has_data()
assert statistics.chart_empty.isVisible()
def test_statistics_summary_row_pairs_mean_with_median(statistics: pages.StatisticsPanel) -> None:
statistics.set_statistics(statistics_payload())
assert statistics.summary_values["qwen"].text() == "21.7% / 16.9%"
assert statistics.summary_values["paired"].text() == "4 例"
def test_statistics_lists_each_doctor_without_ranking(statistics: pages.StatisticsPanel) -> None:
statistics.set_statistics(statistics_payload())
assert statistics.doctors.rowCount() == 2
assert statistics.doctors.item(0, 0).text() == "何医生"
assert statistics.doctors.item(0, 3).text() == "4 / 18.4%"
assert statistics.doctors.item(1, 4).text() == "0 / —"
assert statistics.doctors.item(1, 6).text() == "50.0%"
assert "不是医生准确率" in statistics.footnote.text()
def test_progress_counts_the_batch_in_the_stat_row(progress: pages.ProgressPage) -> None:
progress.set_batch(batch())
assert progress.stat_values["calls"].text() == "2 次"
assert progress.stat_values["failures"].text() == "1 次"
assert progress.stat_values["repairs"].text() == "0 次"
assert progress.stat_values["elapsed"].text() == "9 分 53 秒"
def test_progress_derives_its_stages_from_the_saved_calls(progress: pages.ProgressPage) -> None:
progress.set_batch(batch())
names = []
layout = progress.stage_lists["qwen"]
for index in range(layout.count()):
widget = layout.itemAt(index).widget()
names.append(widget.findChildren(QLabel)[1].text())
assert names == ["文字资料分析", "生成候选与报告"]
assert progress.stage_lists["openai"].count() == 0 # 没有调用记录就不编造阶段
def test_history_names_the_version_change_between_two_batches(history: pages.HistoryPage) -> None:
history.set_history([
{"id": 5, "created_at": "2026-09-10 15:29:00", "status": "success",
"models": {"qwen": {"comparison": {"status": "comparable", "score": 15.3},
"algorithm_version": "prescription-soft-dice-v1.1.0",
"prompt_version": "v4"}}},
{"id": 4, "created_at": "2026-09-10 15:18:00", "status": "success",
"models": {"qwen": {"comparison": {"status": "comparable", "score": 18.9},
"algorithm_version": "prescription-soft-dice-v1.0.1",
"prompt_version": "v3"}}},
])
text = history.strata.text()
assert "比较算法 v1.0.1 → v1.1.0" in text
assert "提示词 v3 → v4" in text
assert "不能直接相减" in text
def test_history_says_when_every_batch_shares_one_version(history: pages.HistoryPage) -> None:
history.set_history([
{"id": 2, "created_at": "2026-09-10 15:29:00",
"models": {"qwen": {"algorithm_version": "prescription-soft-dice-v1.1.0"}}},
{"id": 1, "created_at": "2026-09-10 14:29:00",
"models": {"qwen": {"algorithm_version": "prescription-soft-dice-v1.1.0"}}},
])
assert "全部批次使用同一套算法" in history.strata.text()
def test_history_shows_why_a_batch_has_no_score(history: pages.HistoryPage) -> None:
history.set_history([{"id": 2, "created_at": "2026-09-10 13:25:00", "status": "failed",
"models": {"qwen": {"error_message": "模型返回未通过校验"}}}])
assert "模型返回未通过校验" in history.table.item(0, 2).text()
assert history.table.item(0, 3).text() == "—"