"""The composed pages must show the saved batch exactly, including what is missing from it.""" from __future__ import annotations from typing import Any import pytest from PySide6.QtWidgets import QApplication, QLabel from doctor_workstation.ui.dialogs import issued_prescription_ai_pages as pages @pytest.fixture(scope="module") def application() -> QApplication: return QApplication.instance() or QApplication([]) def comparison_rows(model: str) -> list[dict[str, Any]]: doses = {"qwen": {"生地黄": 15, "生麦冬": 12, "麸炒白术": 12}, "openai": {"生地黄": 12, "茯苓": 8}}[model] contributions = {"qwen": {"生地黄": 0.94, "生麦冬": 1.0}, "openai": {"生地黄": 0.75}}[model] doctor = {"生地黄": 16, "生麦冬": 12, "红参片": 6, "茯苓": 10} rows = [] for name in sorted(set(doses) | set(doctor)): rows.append({ "name": name, "unit": "g", "doctor_dosage": doctor.get(name), "candidate_dosage": doses.get(name), "contribution": contributions.get(name), }) return rows def batch(**overrides: Any) -> dict[str, Any]: value = { "id": 13, "status": "success", "coverage_status": "partial", "comparison_type": "latest_context", "source_summary": {"attachment_count": 4, "diagnoses_count": 1, "video_calls_count": 3, "source_record_count": 9}, "missing": [{"code": "TRANSCRIPT_NOT_VERIFIED_COMPLETE", "critical": True}, {"code": "TRANSCRIPT_NOT_VERIFIED_COMPLETE", "critical": True}, {"code": "ARCHIVE_SYNC_WATERMARK_UNAVAILABLE", "critical": False}], "models": { "qwen": {"status": "success", "algorithm_version": "prescription-soft-dice-v1.1.0", "prompt_version": "manual-prescription-required-candidate-v4", "comparison": {"status": "comparable", "rows": comparison_rows("qwen")}, "coverage": {"files": [{"status": "processed"}, {"status": "processed"}, {"status": "processed"}, {"status": "restricted"}]}, "progress": {"stage_label": "处理完成", "elapsed_seconds": 135, "attempt": 1}, "usage": {"total_calls": 2, "calls": [ {"stage": "text:0", "ok": True, "latency_ms": 3400, "file_count": 0, "usage": {"completion_tokens": 1020}, "error_code": ""}, {"stage": "final", "ok": False, "latency_ms": 14400, "file_count": 0, "usage": {"completion_tokens": 5617}, "error_code": "INVALID_REPORT_OUTPUT"}]}}, "openai": {"status": "success", "comparison": {"status": "comparable", "rows": comparison_rows("openai")}, "progress": {"stage_label": "处理完成", "elapsed_seconds": 593, "attempt": 1}, "usage": {"calls": []}}, }, } value.update(overrides) return value # ---------------------------------------------------------------- candidates page @pytest.fixture def per_herb(application: QApplication) -> pages.CandidatesPage: widget = pages.CandidatesPage() widget.resize(1200, 700) widget.show() application.processEvents() yield widget widget.close() def _dose_cells(page: pages.CandidatesPage, row: int) -> tuple[str, str]: return (page.table.cellWidget(row, 1).dose, page.table.cellWidget(row, 2).dose) def _verdict(page: pages.CandidatesPage, row: int) -> str: return page.table.cellWidget(row, 4).findChildren(QLabel)[0].text() def test_per_herb_merges_both_models_into_one_row(per_herb: pages.CandidatesPage) -> None: per_herb.set_batch(batch()) names = [per_herb.table.item(row, 0).text() for row in range(per_herb.table.rowCount())] assert names.count("生地黄") == 1 row = names.index("生地黄") assert _dose_cells(per_herb, row) == ("15 g", "12 g") assert per_herb.table.item(row, 3).text() == "16 g" assert per_herb.table.cellWidget(row, 1).contribution == 0.94 assert per_herb.table.cellWidget(row, 2).contribution == 0.75 def test_per_herb_marks_absence_without_inventing_a_dose(per_herb: pages.CandidatesPage) -> None: per_herb.set_batch(batch()) names = [per_herb.table.item(row, 0).text() for row in range(per_herb.table.rowCount())] only_doctor = names.index("红参片") assert _dose_cells(per_herb, only_doctor) == ("—", "—") assert per_herb.table.cellWidget(only_doctor, 1).contribution is None assert _verdict(per_herb, only_doctor) == "仅医方使用" added = names.index("麸炒白术") assert per_herb.table.item(added, 3).text() == "未收录" assert _verdict(per_herb, added) == "仅 千问 收录" def test_per_herb_conclusion_states_a_dose_gap_over_the_threshold(per_herb: pages.CandidatesPage) -> None: per_herb.set_batch(batch()) names = [per_herb.table.item(row, 0).text() for row in range(per_herb.table.rowCount())] row = names.index("生地黄") assert _verdict(per_herb, row) == "两模型均收录" # 医方 16,两侧差 1 与 4,未过 5 克阈值 source = batch() for model in source["models"].values(): for entry in model["comparison"]["rows"]: if entry["name"] == "生地黄": entry["candidate_dosage"] = 9 per_herb.set_batch(source) names = [per_herb.table.item(index, 0).text() for index in range(per_herb.table.rowCount())] assert _verdict(per_herb, names.index("生地黄")) == "剂量分歧 7 g" def test_per_herb_filters_count_and_narrow_the_table( per_herb: pages.CandidatesPage, application: QApplication) -> None: per_herb.set_batch(batch()) total = per_herb.table.rowCount() assert per_herb.filter_buttons["all"].text() == f"全部 {total}" per_herb.search.setText("生地黄") application.processEvents() assert per_herb.table.rowCount() == 1 per_herb.search.clear() per_herb.filter_buttons["doctor"].click() application.processEvents() assert 0 < per_herb.table.rowCount() < total assert all(_verdict(per_herb, row) in {"仅医方使用", "两模型均未收录"} for row in range(per_herb.table.rowCount())) per_herb.filter_buttons["single"].click() application.processEvents() assert all("仅 " in _verdict(per_herb, row) for row in range(per_herb.table.rowCount())) def test_per_herb_cards_show_each_saved_prescription(per_herb: pages.CandidatesPage) -> None: per_herb.set_batch(batch(doctor_snapshot={"prescription": { "herbs": [{"name": "生地黄", "dosage": 16, "unit": "g"}, {"name": "红参片", "dosage": 6, "unit": "g"}], "prescription_type": "浓缩水丸", "dose_count": 1, "usage_instruction": "每日1剂,水煎分服。"}})) doctor = per_herb.cards["doctor"] assert doctor.summary.text() == "2 味 · 浓缩水丸" assert "生地黄" in doctor.herbs.text() and "16 g" in doctor.herbs.text() assert "每日1剂" in doctor.usage.text() assert per_herb.cards["qwen"].summary.text() == "尚无候选方" def test_per_herb_card_grows_for_a_long_usage_note(per_herb: pages.CandidatesPage, application: QApplication) -> None: """A card must not cap itself and cut the herb list or the 方义 in half.""" note = "方中生黄芪益气固表,生地黄、生麦冬滋阴清热。" * 8 per_herb.set_batch(batch(doctor_snapshot={"prescription": { "herbs": [{"name": f"药{index}", "dosage": 15, "unit": "g"} for index in range(20)], "prescription_type": "浓缩水丸", "dose_count": 1, "usage_instruction": note}})) per_herb.resize(1100, 700) application.processEvents() card = per_herb.cards["doctor"] assert "另有 15 味" in card.herbs.text() assert card.usage.height() >= card.usage.heightForWidth(card.usage.width()) assert card.height() >= card.herbs.height() + card.usage.height() def test_per_herb_says_when_no_prescription_was_saved(per_herb: pages.CandidatesPage) -> None: per_herb.set_batch({}) assert per_herb.cards["doctor"].summary.text() == "未保存原方" assert per_herb.cards["doctor"].herbs.text() == "尚未保存药味" assert per_herb.table.rowCount() == 0 # ---------------------------------------------------------------- sources page @pytest.fixture def sources(application: QApplication) -> pages.SourcesPage: widget = pages.SourcesPage() widget.resize(900, 600) widget.show() application.processEvents() yield widget widget.close() def test_sources_groups_gaps_by_type_and_keeps_criticality(sources: pages.SourcesPage) -> None: sources.set_batch(batch()) rows = {sources.gaps.item(row, 0).text(): (sources.gaps.item(row, 1).text(), sources.gaps.item(row, 2).text()) for row in range(sources.gaps.rowCount())} transcript = next(key for key in rows if "转写" in key) assert rows[transcript] == ("关键", "2") archive = next(key for key in rows if "归档" in key) assert rows[archive][0] == "一般" def _composition(sources: pages.SourcesPage) -> dict[str, str]: rows = {} for index in range(sources.composition_layout.count()): widget = sources.composition_layout.itemAt(index).widget() labels = widget.findChildren(QLabel) rows[labels[0].text()] = labels[-1].text() return rows def test_sources_reports_attachment_reading_and_composition(sources: pages.SourcesPage) -> None: sources.set_batch(batch()) assert "模型实际读取 3 个" in sources.attachment_note.text() assert sources.waffle.accessibleDescription() == "附件 4 个:已读 3,受限或不支持 1" assert _composition(sources)["问诊通话"] == "3" assert "soft-dice" in sources.meta.text() def test_sources_reports_what_each_model_managed_to_read(sources: pages.SourcesPage) -> None: sources.set_batch(batch()) rows = {sources.per_model.item(row, 0).text(): (sources.per_model.item(row, 1).text(), sources.per_model.item(row, 3).text()) for row in range(sources.per_model.rowCount())} assert rows["千问"] == ("3", "75%") assert "已读取 3" in sources.attachment_legend.text() def test_sources_meta_states_the_cutoff_and_reads_codes_in_chinese(sources: pages.SourcesPage) -> None: sources.set_batch(batch(cutoff_at="2026-09-10 15:29")) text = sources.meta.text() assert "资料截止:2026-09-10 15:29" in text assert "覆盖状态:部分资料缺失" in text # partial 在覆盖语境里说的是资料,不是进度 assert "对照类型:最新资料对照" in text sources.set_batch(batch()) assert "资料截止:—" in sources.meta.text() def test_sources_stays_empty_without_a_batch(sources: pages.SourcesPage) -> None: sources.set_batch({}) assert _composition(sources) == {} assert sources.gaps.rowCount() == 0 assert sources.attachment_note.text() == "本次没有附件" assert not sources.gap_note.isVisible() def test_sources_puts_critical_gaps_first_and_counts_them(sources: pages.SourcesPage) -> None: sources.set_batch(batch()) assert "关键" in sources.gaps.item(0, 1).text() assert "3 项 · 关键 2" in sources.gap_title.findChildren(QLabel)[-1].text() # ---------------------------------------------------------------- progress page @pytest.fixture def progress(application: QApplication) -> pages.ProgressPage: widget = pages.ProgressPage() widget.resize(900, 600) widget.show() application.processEvents() yield widget widget.close() def test_progress_lists_every_call_with_its_outcome(progress: pages.ProgressPage) -> None: progress.set_batch(batch()) assert progress.calls.rowCount() == 2 assert progress.calls.item(0, 0).text() == "千问" assert progress.calls.item(0, 1).text() == "文字资料 1" # 阶段键不直接露出 assert progress.calls.item(1, 1).text() == "生成候选与报告" assert progress.calls.item(0, 3).text() == "3.4 s" assert progress.calls.item(0, 6).text() == "通过" assert progress.calls.item(1, 5).text() == "5617" # 最慢的一次调用占满耗时分布条,其余按比例 assert progress.calls.cellWidget(1, 2)._fraction == 1.0 assert progress.calls.cellWidget(0, 2)._fraction < 0.3 assert "INVALID_REPORT_OUTPUT" not in progress.calls.item(1, 5).text() def test_progress_shows_each_model_stage_and_attempt(progress: pages.ProgressPage) -> None: progress.set_batch(batch()) assert "处理完成" in progress.stage_labels["qwen"].text() assert "第 1 次尝试" in progress.stage_labels["qwen"].text() assert "已用时 2 分 15 秒" in progress.stage_labels["qwen"].text() assert "已用时 9 分 53 秒" in progress.stage_labels["openai"].text() # ---------------------------------------------------------------- history page @pytest.fixture def history(application: QApplication) -> pages.HistoryPage: widget = pages.HistoryPage() widget.resize(900, 600) widget.show() application.processEvents() yield widget widget.close() def test_history_orders_by_time_and_keeps_uncomparable_slots_empty(history: pages.HistoryPage) -> None: history.set_history([ {"id": 13, "created_at": 300, "status": "success", "comparison_type": "latest_context", "models": {"qwen": {"score": 18.9, "comparison_status": "comparable", "algorithm_version": "prescription-soft-dice-v1.1.0"}, "openai": {"score": 16.7, "comparison_status": "comparable"}}}, {"id": 11, "created_at": 100, "status": "failed", "models": {"qwen": {"score": None, "comparison_status": "not_comparable"}, "openai": {"score": None, "comparison_status": "not_comparable"}}}, ]) description = history.chart.accessibleDescription() assert description.index("11") < description.index("13") # oldest first on the chart assert "11 千问 — OpenAI —" in description assert history.table.item(0, 0).text() == "#13" # newest first in the list assert history.table.item(0, 3).text() == "18.9%" assert history.table.item(1, 3).text() == "—" assert history.table.item(0, 5).text() == "v1.1.0" assert history.chart.has_data() def test_history_reads_the_score_from_either_payload_shape(history: pages.HistoryPage) -> None: """列表接口把分数摊平,详情接口留在 comparison 里,两种都要认。""" history.set_history([ {"id": 40, "created_at": "2026-09-10 15:29:00", "status": "success", "comparison_type": "non_independent", "models": {"qwen": {"comparison": {"status": "comparable", "score": 94.44}}, "openai": {"comparison": {"status": "not_comparable", "score": 93.3}}}}, {"id": 39, "created_at": "2026-09-09 09:00:00", "status": "success", "models": {"qwen": {"score": 42.9, "comparison_status": "comparable"}}}, ]) assert history.table.item(0, 0).text() == "#40" assert history.table.item(0, 3).text() == "94.4%" assert history.table.item(0, 4).text() == "—" # 不可比就不给分,哪怕载荷里带着数字 assert "分层" in history.strata.text() or "同一套" in history.strata.text() assert history.table.item(1, 3).text() == "42.9%" description = history.chart.accessibleDescription() assert description.index("39") < description.index("40") def test_history_without_any_comparable_batch_draws_nothing(history: pages.HistoryPage) -> None: history.set_history([{"id": 1, "created_at": 1, "models": {"qwen": {"comparison_status": "not_comparable"}}}]) assert not history.chart.has_data() assert history.table.rowCount() == 1 # ---------------------------------------------------------------- statistics panel @pytest.fixture def statistics(application: QApplication) -> pages.StatisticsPanel: widget = pages.StatisticsPanel() widget.resize(1000, 600) widget.show() application.processEvents() yield widget widget.close() def statistics_payload() -> dict[str, Any]: return { "total_count": 10, "patient_count": 8, "doctors": [ {"doctor_id": 26, "doctor_name": "何医生", "total_count": 6, "patient_count": 5, "paired_count": 3, "models": {"qwen": {"eligible_count": 4, "mean": 18.4, "median": 16.9, "excluded_reasons": {"SOURCE_HISTORY_VERSIONS_UNAVAILABLE": 2}}, "openai": {"eligible_count": 3, "mean": 21.0, "median": 19.6, "excluded_reasons": {"SOURCE_HISTORY_VERSIONS_UNAVAILABLE": 3}}}, "review": {"evaluated_count": 0, "qualified_count": 0, "qualified_rate": None}}, {"doctor_id": 31, "doctor_name": "李医生", "total_count": 4, "patient_count": 3, "paired_count": 1, "models": {"qwen": {"eligible_count": 2, "mean": 25.0, "excluded_reasons": {"incomplete_coverage": 2}}, "openai": {"eligible_count": 0, "mean": None, "excluded_reasons": {"incomplete_coverage": 4}}}, "review": {"evaluated_count": 2, "qualified_count": 1, "qualified_rate": 50.0}}, ], } def test_statistics_headline_counts_and_coverage(statistics: pages.StatisticsPanel) -> None: statistics.set_statistics(statistics_payload()) assert statistics.kpi_values["events"].text() == "10" assert "涉及患者 8 人" in statistics.kpi_notes["events"].text() assert statistics.kpi_values["qwen"].text() == "6" assert "覆盖率 60.0%" in statistics.kpi_notes["qwen"].text() assert statistics.kpi_values["openai"].text() == "3" def test_statistics_review_rate_needs_a_review_sample(statistics: pages.StatisticsPanel) -> None: payload = statistics_payload() for doctor in payload["doctors"]: doctor["review"] = {"evaluated_count": 0, "qualified_count": 0, "qualified_rate": None} statistics.set_statistics(payload) assert statistics.kpi_values["review"].text() == "—" assert "尚未建立复核样本" in statistics.kpi_notes["review"].text() statistics.set_statistics(statistics_payload()) assert statistics.kpi_values["review"].text() == "50.0%" def _bars(layout) -> list[str]: return [layout.itemAt(index).widget().accessibleDescription() for index in range(layout.count())] def test_statistics_funnel_and_exclusions_are_aggregated(statistics: pages.StatisticsPanel) -> None: statistics.set_statistics(statistics_payload()) funnel = _bars(statistics.funnel_layout) assert "范围内开方事件 10" in funnel assert "千问 有效基线比较 6" in funnel assert "两模型配对共同样本 4" in funnel reasons = _bars(statistics.exclusion_layout) assert any(text.endswith(" 5") for text in reasons) # 来源历史版本无法重建 2 + 3 assert all("SOURCE_HISTORY" not in text for text in reasons) def test_statistics_distribution_sums_the_saved_bins(statistics: pages.StatisticsPanel) -> None: payload = statistics_payload() payload["doctors"][0]["models"]["qwen"]["distribution"] = {"[0,20)": 3, "[20,40)": 1} payload["doctors"][1]["models"]["qwen"]["distribution"] = {"[0,20)": 2} statistics.set_statistics(payload) assert statistics.distribution.has_data() assert "千问 5/1/0/0/0" in statistics.distribution.accessibleDescription() assert statistics.distribution.isVisible() def test_statistics_draws_no_distribution_without_samples(statistics: pages.StatisticsPanel) -> None: statistics.set_statistics(statistics_payload()) assert not statistics.distribution.has_data() assert statistics.chart_empty.isVisible() def test_statistics_summary_row_pairs_mean_with_median(statistics: pages.StatisticsPanel) -> None: statistics.set_statistics(statistics_payload()) assert statistics.summary_values["qwen"].text() == "21.7% / 16.9%" assert statistics.summary_values["paired"].text() == "4 例" def test_statistics_lists_each_doctor_without_ranking(statistics: pages.StatisticsPanel) -> None: statistics.set_statistics(statistics_payload()) assert statistics.doctors.rowCount() == 2 assert statistics.doctors.item(0, 0).text() == "何医生" assert statistics.doctors.item(0, 3).text() == "4 / 18.4%" assert statistics.doctors.item(1, 4).text() == "0 / —" assert statistics.doctors.item(1, 6).text() == "50.0%" assert "不是医生准确率" in statistics.footnote.text() def test_progress_counts_the_batch_in_the_stat_row(progress: pages.ProgressPage) -> None: progress.set_batch(batch()) assert progress.stat_values["calls"].text() == "2 次" assert progress.stat_values["failures"].text() == "1 次" assert progress.stat_values["repairs"].text() == "0 次" assert progress.stat_values["elapsed"].text() == "9 分 53 秒" def test_progress_derives_its_stages_from_the_saved_calls(progress: pages.ProgressPage) -> None: progress.set_batch(batch()) names = [] layout = progress.stage_lists["qwen"] for index in range(layout.count()): widget = layout.itemAt(index).widget() names.append(widget.findChildren(QLabel)[1].text()) assert names == ["文字资料分析", "生成候选与报告"] assert progress.stage_lists["openai"].count() == 0 # 没有调用记录就不编造阶段 def test_history_names_the_version_change_between_two_batches(history: pages.HistoryPage) -> None: history.set_history([ {"id": 5, "created_at": "2026-09-10 15:29:00", "status": "success", "models": {"qwen": {"comparison": {"status": "comparable", "score": 15.3}, "algorithm_version": "prescription-soft-dice-v1.1.0", "prompt_version": "v4"}}}, {"id": 4, "created_at": "2026-09-10 15:18:00", "status": "success", "models": {"qwen": {"comparison": {"status": "comparable", "score": 18.9}, "algorithm_version": "prescription-soft-dice-v1.0.1", "prompt_version": "v3"}}}, ]) text = history.strata.text() assert "比较算法 v1.0.1 → v1.1.0" in text assert "提示词 v3 → v4" in text assert "不能直接相减" in text def test_history_says_when_every_batch_shares_one_version(history: pages.HistoryPage) -> None: history.set_history([ {"id": 2, "created_at": "2026-09-10 15:29:00", "models": {"qwen": {"algorithm_version": "prescription-soft-dice-v1.1.0"}}}, {"id": 1, "created_at": "2026-09-10 14:29:00", "models": {"qwen": {"algorithm_version": "prescription-soft-dice-v1.1.0"}}}, ]) assert "全部批次使用同一套算法" in history.strata.text() def test_history_shows_why_a_batch_has_no_score(history: pages.HistoryPage) -> None: history.set_history([{"id": 2, "created_at": "2026-09-10 13:25:00", "status": "failed", "models": {"qwen": {"error_message": "模型返回未通过校验"}}}]) assert "模型返回未通过校验" in history.table.item(0, 2).text() assert history.table.item(0, 3).text() == "—"