213 lines
9.7 KiB
Python
213 lines
9.7 KiB
Python
"""
|
||
Тесты автономного HTML-пакета для ручной разметки (`src/dxa/review_pack.py`).
|
||
|
||
Проверяется то, от чего зависит работа врача без сервиса: страница не тянет
|
||
ничего из сети, встроенные данные разбираются, выгрузка читается обучением, а
|
||
вердикты из чужого пакета не подмешиваются молча.
|
||
"""
|
||
import csv
|
||
import json
|
||
import re
|
||
from pathlib import Path
|
||
|
||
import pytest
|
||
|
||
from src.dxa.dataset import build_records
|
||
from src.dxa.excel_labels import load_labels_csv
|
||
from src.dxa.labels import QUALITY_BAD, QUALITY_GOOD, ImageRecord
|
||
from src.dxa.review_pack import (
|
||
REVIEW_FIELDS,
|
||
collect_items,
|
||
main,
|
||
merge_review,
|
||
pack_fingerprint,
|
||
render_html,
|
||
)
|
||
|
||
DATASET_ROOT = Path("dataset_hack")
|
||
EXCEL_PATH = DATASET_ROOT / "НД_для_обучения" / "разметка.xlsx"
|
||
LABELS_CSV = Path("labels/labels_images.csv")
|
||
HAVE_DATASET = (DATASET_ROOT / "НД_для_обучения" / "Исследования").is_dir()
|
||
|
||
needs_dataset = pytest.mark.skipif(not HAVE_DATASET, reason="dataset_hack is not available")
|
||
|
||
|
||
def _record(path, region="spine", label=QUALITY_GOOD, marker=None, study="S1"):
|
||
return ImageRecord(path=Path(path), study=study, region=region, label=label, marker=marker)
|
||
|
||
|
||
def _labels_table(**overrides):
|
||
row = {"quality_class": "0", "anatomical_region": "spine", "violation_type": ""}
|
||
row.update(overrides)
|
||
return {"/s/a.dcm": row}
|
||
|
||
|
||
class TestFingerprint:
|
||
def test_is_order_independent(self):
|
||
assert pack_fingerprint(["a", "b"]) == pack_fingerprint(["b", "a"])
|
||
|
||
def test_differs_for_different_sets(self):
|
||
assert pack_fingerprint(["a", "b"]) != pack_fingerprint(["a", "c"])
|
||
|
||
def test_short_and_stable(self):
|
||
value = pack_fingerprint(["a"])
|
||
assert len(value) == 10 and value == pack_fingerprint(["a"])
|
||
|
||
|
||
class TestCollectItems:
|
||
def test_conflicts_come_first(self):
|
||
records = [
|
||
_record("/s/a.dcm", label=QUALITY_GOOD),
|
||
_record("/s/b.dcm", label=QUALITY_BAD, marker="good"),
|
||
_record("/s/c.dcm", label=QUALITY_GOOD, marker="bad"),
|
||
]
|
||
items = collect_items(records, _labels_table(), expert_paths=["/s/a.dcm"])
|
||
assert [item["path"] for item in items] == ["/s/b.dcm", "/s/c.dcm", "/s/a.dcm"]
|
||
assert all(item["conflict"] for item in items[:2])
|
||
assert not items[2]["conflict"]
|
||
|
||
def test_marks_source_from_labels_table(self):
|
||
records = [_record("/s/a.dcm"), _record("/s/b.dcm")]
|
||
items = collect_items(records, _labels_table(), expert_paths=["/s/a.dcm"])
|
||
sources = {item["path"]: item["source"] for item in items}
|
||
assert sources == {"/s/a.dcm": "table", "/s/b.dcm": "filename"}
|
||
|
||
def test_carries_violation_from_built_labels(self):
|
||
records = [_record("/s/a.dcm", label=QUALITY_BAD)]
|
||
items = collect_items(records, _labels_table(violation_type="rotation;roi_incorrect"))
|
||
assert items[0]["violation"] == "rotation;roi_incorrect"
|
||
# Подпись берётся по первому критерию: в разметке их может быть два.
|
||
assert items[0]["violation_label"] == "Ротация / позиционирование"
|
||
|
||
def test_limit_and_conflicts_only(self):
|
||
records = [
|
||
_record("/s/a.dcm", label=QUALITY_GOOD),
|
||
_record("/s/b.dcm", label=QUALITY_BAD, marker="good"),
|
||
_record("/s/c.dcm", label=QUALITY_GOOD, marker="bad"),
|
||
]
|
||
assert len(collect_items(records, {}, limit=2)) == 2
|
||
only = collect_items(records, {}, only_conflicts=True)
|
||
assert [item["path"] for item in only] == ["/s/b.dcm", "/s/c.dcm"]
|
||
|
||
|
||
class TestRenderHtml:
|
||
def _items(self):
|
||
return [
|
||
{"path": "/s/a.dcm", "file": "a.dcm", "study": "S1", "region": "spine",
|
||
"label": 1, "violation": "artifact", "violation_label": "Артефакты",
|
||
"conflict": True, "marker": "bad", "source": "table", "image": "data:image/png;base64,AA"},
|
||
]
|
||
|
||
def test_all_placeholders_are_substituted(self):
|
||
html = render_html(self._items(), "Пакет", "abcdef1234", [])
|
||
assert "@@" not in html, "в странице остались неподставленные маркеры"
|
||
|
||
def test_does_not_reference_network(self):
|
||
"""Пакет открывается с диска: внешних ресурсов быть не должно."""
|
||
html = render_html(self._items(), "Пакет", "abcdef1234", [])
|
||
assert "http://" not in html and "https://" not in html
|
||
assert not re.search(r'(?:src|href)="/(?!/)', html), "есть ссылки на /static или CDN"
|
||
|
||
def test_payload_is_valid_json_with_items(self):
|
||
html = render_html(self._items(), "Пакет", "abcdef1234", [{"code": "artifact", "label": "A", "scope": "any"}])
|
||
raw = re.search(r'<script id="payload" type="application/json">(.*?)</script>', html, re.S)
|
||
assert raw, "не найден блок с данными"
|
||
payload = json.loads(raw.group(1))
|
||
assert payload["packId"] == "abcdef1234"
|
||
assert len(payload["items"]) == 1
|
||
assert payload["items"][0]["conflict"] is True
|
||
assert "image" in payload["items"][0]
|
||
assert payload["reviewFields"] == list(REVIEW_FIELDS)
|
||
|
||
def test_embedded_image_uri_is_not_broken(self):
|
||
"""`</` внутри JSON ломало бы разбор блока script."""
|
||
item = {**self._items()[0], "comment": "закрывающий тег </script> в тексте"}
|
||
html = render_html([item], "Пакет", "abcdef1234", [])
|
||
payload = re.search(r'<script id="payload" type="application/json">(.*?)</script>', html, re.S)
|
||
assert json.loads(payload.group(1))
|
||
|
||
|
||
@needs_dataset
|
||
class TestMergeReview:
|
||
def _write_review(self, tmp_path, rows):
|
||
path = tmp_path / "doctor.csv"
|
||
with path.open("w", newline="", encoding="utf-8") as fh:
|
||
writer = csv.DictWriter(fh, fieldnames=list(REVIEW_FIELDS))
|
||
writer.writeheader()
|
||
writer.writerows(rows)
|
||
return path
|
||
|
||
def test_overrides_built_labels(self, tmp_path):
|
||
records = build_records(
|
||
str(DATASET_ROOT), str(EXCEL_PATH), dedup=True, labels_csv=str(LABELS_CSV)
|
||
)
|
||
first = records[0].path.as_posix()
|
||
review = self._write_review(tmp_path, [{
|
||
"path_to_image": first, "anatomical_region": "spine", "quality_class": "1",
|
||
"violation_type": "artifact", "comment": "проверено врачом",
|
||
"reviewer": "Тестова", "reviewed_at": "2026-09-27T00:00:00+00:00", "pack_id": "x",
|
||
}])
|
||
|
||
out = tmp_path / "merged.csv"
|
||
summary = merge_review(review, out, labels_csv=LABELS_CSV)
|
||
|
||
assert summary["reviewed"] == 1 and summary["overridden"] == 1
|
||
assert not summary["unmatched"]
|
||
table = load_labels_csv(out)
|
||
assert len(table) == summary["rows"]
|
||
assert table[first]["quality_class"] == "1"
|
||
assert table[first]["label_rule"] == "manual"
|
||
assert table[first]["reviewer"] == "Тестова"
|
||
# Остальные снимки остаются с меткой построенной разметки.
|
||
assert any(row["label_rule"] == "table" for row in table.values())
|
||
|
||
def test_reports_paths_absent_from_dataset(self, tmp_path):
|
||
review = self._write_review(tmp_path, [{
|
||
"path_to_image": "/чужой/снимок.dcm", "anatomical_region": "spine",
|
||
"quality_class": "0", "violation_type": "", "comment": "", "reviewer": "",
|
||
"reviewed_at": "", "pack_id": "x",
|
||
}])
|
||
out = tmp_path / "merged.csv"
|
||
summary = merge_review(review, out, labels_csv=LABELS_CSV)
|
||
assert summary["unmatched"] == ["/чужой/снимок.dcm"]
|
||
assert summary["overridden"] == 0
|
||
|
||
def test_empty_review_is_rejected(self, tmp_path):
|
||
path = tmp_path / "empty.csv"
|
||
with path.open("w", newline="", encoding="utf-8") as fh:
|
||
csv.DictWriter(fh, fieldnames=list(REVIEW_FIELDS)).writeheader()
|
||
with pytest.raises(ValueError, match="нет ни одного разобранного вердикта"):
|
||
merge_review(path, tmp_path / "out.csv", labels_csv=LABELS_CSV)
|
||
|
||
|
||
@needs_dataset
|
||
class TestCli:
|
||
def test_build_writes_file_with_expected_shape(self, tmp_path, capsys):
|
||
out = tmp_path / "pack.html"
|
||
code = main(["--out", str(out), "--limit", "3"])
|
||
assert code == 0
|
||
assert out.exists() and out.stat().st_size > 10_000
|
||
|
||
html = out.read_text(encoding="utf-8")
|
||
payload = re.search(r'<script id="payload" type="application/json">(.*?)</script>', html, re.S)
|
||
data = json.loads(payload.group(1))
|
||
assert len(data["items"]) == 3
|
||
assert all(item["image"].startswith("data:image/png;base64,") for item in data["items"])
|
||
|
||
printed = capsys.readouterr().out
|
||
assert "Снимков в пакете: 3" in printed
|
||
assert "Отпечаток набора:" in printed
|
||
|
||
def test_only_conflicts_filter(self, tmp_path):
|
||
out = tmp_path / "pack.html"
|
||
main(["--out", str(out), "--only-conflicts", "--limit", "5"])
|
||
html = out.read_text(encoding="utf-8")
|
||
data = json.loads(
|
||
re.search(r'<script id="payload" type="application/json">(.*?)</script>', html, re.S).group(1)
|
||
)
|
||
assert data["items"] and all(item["conflict"] for item in data["items"])
|
||
|
||
def test_merge_requires_out(self, tmp_path, capsys):
|
||
with pytest.raises(SystemExit):
|
||
main(["--merge", str(tmp_path / "x.csv")])
|