From 8de2a459c10286b9dfaf8b973273d0489bfc84e9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Marek=20Such=C3=A1nek?= Date: Thu, 24 Sep 2026 08:34:08 +0200 Subject: [PATCH] chore(docworker): Use dsw-models for document context --- packages/dsw-document-worker/CHANGELOG.md | 1 + packages/dsw-document-worker/Dockerfile | 1 + .../dsw/document_worker/consts.py | 3 - .../dsw/document_worker/{model => }/http.py | 4 +- .../dsw/document_worker/model/__init__.py | 4 - .../dsw/document_worker/model/context.py | 2323 ----------------- .../dsw/document_worker/model/utils.py | 130 - .../dsw/document_worker/sanitizer.py | 94 - .../document_worker/templates/extraction.py | 2 +- .../dsw/document_worker/templates/filters.py | 7 +- .../templates/steps/template.py | 5 +- .../document_worker/templates/templates.py | 2 +- .../dsw/document_worker/utils.py | 17 - .../dsw/document_worker/worker.py | 3 +- .../dsw-document-worker/lambda.Dockerfile | 1 + packages/dsw-document-worker/pyproject.toml | 4 +- .../support/DocumentContext.md | 2 +- .../tests/test_context_document.py | 2 +- ...xt_parity.py => test_context_reference.py} | 127 +- .../dsw-document-worker/tests/test_http.py | 2 +- .../tests/test_metamodel_version.py | 25 + .../tests/test_sanitizer.py | 3 +- .../dsw/models/document_context/graph.py | 6 +- .../dsw/models/document_context/rendering.py | 4 +- pyproject.toml | 2 - uv.lock | 12 +- 26 files changed, 81 insertions(+), 2705 deletions(-) rename packages/dsw-document-worker/dsw/document_worker/{model => }/http.py (98%) delete mode 100644 packages/dsw-document-worker/dsw/document_worker/model/__init__.py delete mode 100644 packages/dsw-document-worker/dsw/document_worker/model/context.py delete mode 100644 packages/dsw-document-worker/dsw/document_worker/model/utils.py delete mode 100644 packages/dsw-document-worker/dsw/document_worker/sanitizer.py rename packages/dsw-document-worker/tests/{test_context_parity.py => test_context_reference.py} (54%) create mode 100644 packages/dsw-document-worker/tests/test_metamodel_version.py diff --git a/packages/dsw-document-worker/CHANGELOG.md b/packages/dsw-document-worker/CHANGELOG.md index 903d52e0..14e819f5 100644 --- a/packages/dsw-document-worker/CHANGELOG.md +++ b/packages/dsw-document-worker/CHANGELOG.md @@ -21,6 +21,7 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - Update to DT metamodel 18.3 - The `jinja2.ext.i18n` extension is always enabled for Jinja-powered steps - `extras.project` (and the deprecated `extras.questionnaire`) provide `knowledge_model_package_uuid` instead of `knowledge_package_uuid` +- The template-facing document context object model (`ctx|to_context_obj`) comes from `dsw-models` (`dsw.models.document_context.graph`) instead of a copy inside the worker; the object model, its attributes and its Markdown rendering are unchanged ### Fixed diff --git a/packages/dsw-document-worker/Dockerfile b/packages/dsw-document-worker/Dockerfile index bfba993f..e97bd4ef 100644 --- a/packages/dsw-document-worker/Dockerfile +++ b/packages/dsw-document-worker/Dockerfile @@ -28,6 +28,7 @@ RUN python -m pip wheel --no-deps --wheel-dir=/app/wheels \ /app/packages/dsw-command-queue \ /app/packages/dsw-config \ /app/packages/dsw-database \ + /app/packages/dsw-models \ /app/packages/dsw-storage \ /app/packages/dsw-document-worker/addons/* \ /app/packages/dsw-document-worker diff --git a/packages/dsw-document-worker/dsw/document_worker/consts.py b/packages/dsw-document-worker/dsw/document_worker/consts.py index bdd1ed68..ebed2a4c 100644 --- a/packages/dsw-document-worker/dsw/document_worker/consts.py +++ b/packages/dsw-document-worker/dsw/document_worker/consts.py @@ -29,9 +29,6 @@ LOCALE_STAMP_FILE_NAME = 'updated_at' LOCALES_CACHE_DIR = '.locales' -CURRENT_METAMODEL_MAJOR = 18 -CURRENT_METAMODEL_MINOR = 3 - try: __version__ = version(PACKAGE_NAME) except PackageNotFoundError: diff --git a/packages/dsw-document-worker/dsw/document_worker/model/http.py b/packages/dsw-document-worker/dsw/document_worker/http.py similarity index 98% rename from packages/dsw-document-worker/dsw/document_worker/model/http.py rename to packages/dsw-document-worker/dsw/document_worker/http.py index 7bf24121..66744ef5 100644 --- a/packages/dsw-document-worker/dsw/document_worker/model/http.py +++ b/packages/dsw-document-worker/dsw/document_worker/http.py @@ -8,8 +8,8 @@ if typing.TYPE_CHECKING: - from ..config import TemplateConfig - from ..urls import UrlPolicy + from .config import TemplateConfig + from .urls import UrlPolicy LOG = logging.getLogger(__name__) diff --git a/packages/dsw-document-worker/dsw/document_worker/model/__init__.py b/packages/dsw-document-worker/dsw/document_worker/model/__init__.py deleted file mode 100644 index 310d2518..00000000 --- a/packages/dsw-document-worker/dsw/document_worker/model/__init__.py +++ /dev/null @@ -1,4 +0,0 @@ -from .context import DocumentContext - - -__all__ = ['DocumentContext'] diff --git a/packages/dsw-document-worker/dsw/document_worker/model/context.py b/packages/dsw-document-worker/dsw/document_worker/model/context.py deleted file mode 100644 index d85044ef..00000000 --- a/packages/dsw-document-worker/dsw/document_worker/model/context.py +++ /dev/null @@ -1,2323 +0,0 @@ -from __future__ import annotations - -import abc -import re -import typing - -import dateutil.parser as dp - -from .. import consts -from ..utils import check_metamodel_version -from .utils import render_markdown, strip_markdown - - -if typing.TYPE_CHECKING: - from datetime import datetime - - -AnnotationsT = dict[str, str | list[str]] -TODO_LABEL_UUID = '615b9028-5e3f-414f-b245-12d2ae2eeb20' -DEFAULT_COLOR = '#0033aa' - - -def _datetime(timestamp: str) -> datetime: - return dp.isoparse(timestamp) - - -def _load_annotations(annotations: list[dict[str, str]]) -> AnnotationsT: - result: AnnotationsT = {} - semi_result: dict[str, list[str]] = {} - for item in annotations: - key = item.get('key', '') - value = item.get('value', '') - if key in semi_result: - semi_result[key].append(value) - else: - semi_result[key] = [value] - for key, value_list in semi_result.items(): - if len(value_list) == 1: - result[key] = value_list[0] - else: - result[key] = value_list - return result - - -class Color: - @staticmethod - def contrast_ratio(color1: Color, color2: Color) -> float: - l1 = color1.luminance + 0.05 - l2 = color2.luminance + 0.05 - if l1 > l2: - return l1 / l2 - return l2 / l1 - - def __init__(self, color_hex: str = DEFAULT_COLOR, default: str = DEFAULT_COLOR): - color_hex = self.parse_color_to_hex(color_hex) or default - h = color_hex.lstrip('#') - self.red, self.green, self.blue = tuple(int(h[i:i + 2], 16) for i in (0, 2, 4)) - - @staticmethod - def parse_color_to_hex(color: str) -> str | None: - color = color.strip() - if re.match(r'^#[0-9a-fA-F]{6}$', color): - return color - if re.match(r'^#[0-9a-fA-F]{3}$', color): - r = color[1] - g = color[2] - b = color[3] - return f'#{r}{r}{g}{g}{b}{b}' - return None - - @property - def hex(self): - return f'#{self.red:02x}{self.green:02x}{self.blue:02x}' - - @property - def luminance(self): - def _luminance_component(component: int): - c = component / 255 - if c <= 0.03928: - return c / 12.92 - return ((c + 0.055) / 1.055) ** 2.4 - - r = _luminance_component(self.red) - g = _luminance_component(self.green) - b = _luminance_component(self.blue) - return 0.2126 * r + 0.7152 * g + 0.0722 * b - - @property - def is_dark(self): - return self.luminance < 0.5 - - @property - def is_light(self): - return not self.is_dark - - @property - def contrast_color(self) -> Color: - if self.contrast_ratio(self, Color('#ffffff')) > 3: - return Color('#ffffff') - return Color('#000000') - - def __str__(self): - return self.hex - - -class SimpleAuthor: - - def __init__(self, *, uuid: str, first_name: str, last_name: str, - image_url: str | None, gravatar_hash: str | None): - self.uuid = uuid - self.first_name = first_name - self.last_name = last_name - self.image_url = image_url - self.gravatar_hash = gravatar_hash - - @staticmethod - def load(data: dict | None, **options): - if data is None: - return None - return SimpleAuthor( - uuid=data['uuid'], - first_name=data['firstName'], - last_name=data['lastName'], - image_url=data['imageUrl'], - gravatar_hash=data['gravatarHash'], - ) - - -class User: - - def __init__(self, *, uuid: str, first_name: str, last_name: str, email: str, - created_at: datetime, updated_at: datetime, - affiliation: str | None, image_url: str | None): - self.uuid = uuid - self.first_name = first_name - self.last_name = last_name - self.email = email - self.image_url = image_url - self.affiliation = affiliation - self.created_at = created_at - self.updated_at = updated_at - - @staticmethod - def load(data: dict, **options): - if data is None: - return None - return User( - uuid=data['uuid'], - first_name=data['firstName'], - last_name=data['lastName'], - email=data['email'], - image_url=data['imageUrl'], - affiliation=data['affiliation'], - created_at=_datetime(data['createdAt']), - updated_at=_datetime(data['updatedAt']), - ) - - -class Organization: - - def __init__(self, *, org_id: str, name: str, description: str | None, - affiliations: list[str]): - self.id = org_id - self.name = name - self.description = description - self.affiliations = affiliations - - @staticmethod - def load(data: dict, **options): - return Organization( - org_id=data['organizationId'], - name=data['name'], - description=data['description'], - affiliations=data['affiliations'], - ) - - -class Tag: - - def __init__(self, *, uuid: str, name: str, description: str | None, - color: str, annotations: AnnotationsT): - self.uuid = uuid - self.name = name - self.description = description - self.color = color - self.annotations = annotations - - @property - def a(self): - return self.annotations - - def __eq__(self, other): - if not isinstance(other, Tag): - return False - return other.uuid == self.uuid - - @staticmethod - def load(data: dict, **options): - return Tag( - uuid=data['uuid'], - name=data['name'], - description=data['description'], - color=data['color'], - annotations=_load_annotations(data['annotations']), - ) - - -class ResourceCollection: - - def __init__(self, *, uuid: str, title: str, page_uuids: list[str], - annotations: AnnotationsT): - self.uuid = uuid - self.title = title - self.page_uuids = page_uuids - self.annotations = annotations - self.pages: list[ResourcePage] = [] - - def resolve_links(self, ctx): - self.pages = [ctx.e.resource_pages[key] - for key in self.page_uuids - if key in ctx.e.resource_pages] - for page in self.pages: - page.collection = self - - @property - def a(self): - return self.annotations - - @staticmethod - def load(data: dict, **options): - return ResourceCollection( - uuid=data['uuid'], - title=data['title'], - page_uuids=data['resourcePageUuids'], - annotations=_load_annotations(data['annotations']), - ) - - -class ResourcePage: - - def __init__(self, *, uuid: str, title: str, content: str, - annotations: AnnotationsT): - self.uuid = uuid - self.title = title - self.content = content - self.annotations = annotations - - self.collection: ResourceCollection | None = None - - @property - def a(self): - return self.annotations - - @staticmethod - def load(data: dict, **options): - return ResourcePage( - uuid=data['uuid'], - title=data['title'], - content=data['content'], - annotations=_load_annotations(data['annotations']), - ) - - -class Integration(abc.ABC): - - def __init__(self, *, uuid: str, name: str, - integration_type: str, annotations: AnnotationsT): - self.uuid = uuid - self.name = name - self.type = integration_type - self.annotations = annotations - - @property - def a(self): - return self.annotations - - def __eq__(self, other): - if not isinstance(other, Integration): - return False - return other.uuid == self.uuid - - @staticmethod - @abc.abstractmethod - def load(data: dict, **options): - pass - - -class ApiIntegration(Integration): - - def __init__(self, *, uuid: str, name: str, variables: list[str], - allow_custom_reply: bool, request_method: str, - request_url: str, request_headers: dict[str, str], - request_body: str | None, request_allow_empty_search: bool, - response_list_field: str | None, response_item_template: str, - response_item_template_for_selection: str | None, - annotations: AnnotationsT): - super().__init__( - uuid=uuid, - name=name, - annotations=annotations, - integration_type='ApiIntegration', - ) - self.variables = variables - self.allow_custom_reply = allow_custom_reply - self.request_method = request_method - self.request_url = request_url - self.request_headers = request_headers - self.request_body = request_body - self.request_allow_empty_search = request_allow_empty_search - self.response_list_field = response_list_field - self.response_item_template = response_item_template - self.response_item_template_for_selection = response_item_template_for_selection - - @staticmethod - def default(): - return ApiIntegration( - uuid=consts.NULL_UUID, - name='', - variables=[], - allow_custom_reply=True, - request_method='GET', - request_url='', - request_headers={}, - request_body=None, - request_allow_empty_search=False, - response_list_field=None, - response_item_template='', - response_item_template_for_selection=None, - annotations={}, - ) - - @staticmethod - def load(data: dict, **options): - return ApiIntegration( - uuid=data['uuid'], - name=data['name'], - variables=data['variables'], - allow_custom_reply=data['allowCustomReply'], - request_method=data['requestMethod'], - request_url=data['requestUrl'], - request_headers=data['requestHeaders'], - request_body=data.get('requestBody'), - request_allow_empty_search=data.get('requestAllowEmptySearch', False), - response_list_field=data.get('responseListField'), - response_item_template=data['responseItemTemplate'], - response_item_template_for_selection=data.get('responseItemTemplateForSelection'), - annotations=_load_annotations(data['annotations']), - ) - - -class PluginIntegration(Integration): - - def __init__(self, *, uuid: str, name: str, - plugin_uuid: str, integration_id: str, settings: dict, - annotations: AnnotationsT): - super().__init__( - uuid=uuid, - name=name, - annotations=annotations, - integration_type='PluginIntegration', - ) - self.plugin_uuid = plugin_uuid - self.integration_id = integration_id - self.settings = settings - - @staticmethod - def load(data: dict, **options): - return PluginIntegration( - uuid=data['uuid'], - name=data['name'], - plugin_uuid=data['pluginUuid'], - integration_id=data['pluginIntegrationId'], - settings=data['pluginIntegrationSettings'], - annotations=_load_annotations(data['annotations']), - ) - - -class Phase: - - def __init__(self, *, uuid: str, title: str, description: str | None, - annotations: AnnotationsT, order: int = 0): - self.uuid = uuid - self.title = title - self.description = description - self.order = order - self.annotations = annotations - - @property - def a(self): - return self.annotations - - def __eq__(self, other): - if not isinstance(other, Phase): - return False - return other.uuid == self.uuid - - @staticmethod - def load(data: dict, **options): - return Phase( - uuid=data['uuid'], - title=data['title'], - description=data['description'], - annotations=_load_annotations(data['annotations']), - ) - - -PHASE_NEVER = Phase( - uuid=consts.NULL_UUID, - title='never', - description=None, - order=10000000, - annotations={}, -) - - -class Metric: - - def __init__(self, *, uuid: str, title: str, description: str | None, - abbreviation: str, annotations: AnnotationsT): - self.uuid = uuid - self.title = title - self.description = description - self.abbreviation = abbreviation - self.annotations = annotations - - @property - def a(self): - return self.annotations - - def __eq__(self, other): - if not isinstance(other, Metric): - return False - return other.uuid == self.uuid - - @staticmethod - def load(data: dict, **options): - return Metric( - uuid=data['uuid'], - title=data['title'], - description=data['description'], - abbreviation=data['abbreviation'], - annotations=_load_annotations(data['annotations']), - ) - - -class MetricMeasure: - - def __init__(self, *, measure: float, weight: float, metric_uuid: str): - self.measure = measure - self.weight = weight - self.metric_uuid = metric_uuid - - self.metric: Metric | None = None - - def resolve_links(self, ctx): - if self.metric_uuid in ctx.e.metrics: - self.metric = ctx.e.metrics[self.metric_uuid] - - @staticmethod - def load(data: dict, **options): - return MetricMeasure( - measure=float(data['measure']), - weight=float(data['weight']), - metric_uuid=data['metricUuid'], - ) - - -class Reference(abc.ABC): - - def __init__(self, *, uuid: str, ref_type: str, annotations: AnnotationsT): - self.uuid = uuid - self.type = ref_type - self.annotations = annotations - - @property - def a(self): - return self.annotations - - def __eq__(self, other): - if not isinstance(other, Reference): - return False - return other.uuid == self.uuid - - @abc.abstractmethod - def resolve_links(self, ctx): - pass - - @staticmethod - @abc.abstractmethod - def load(data: dict, **options): - pass - - -class CrossReference(Reference): - - def __init__(self, *, uuid: str, target_uuid: str, description: str, - annotations: AnnotationsT): - super().__init__( - uuid=uuid, - ref_type='CrossReference', - annotations=annotations, - ) - self.target_uuid = target_uuid - self.description = description - - def resolve_links(self, ctx): - pass - - @staticmethod - def load(data: dict, **options): - return CrossReference( - uuid=data['uuid'], - target_uuid=data['targetUuid'], - description=data['description'], - annotations=_load_annotations(data['annotations']), - ) - - -class URLReference(Reference): - - def __init__(self, *, uuid: str, label: str, url: str, - annotations: AnnotationsT): - super().__init__( - uuid=uuid, - ref_type='URLReference', - annotations=annotations, - ) - self.label = label - self.url = url - - def resolve_links(self, ctx): - pass - - @staticmethod - def load(data: dict, **options): - return URLReference( - uuid=data['uuid'], - label=data['label'], - url=data['url'], - annotations=_load_annotations(data['annotations']), - ) - - -class ResourcePageReference(Reference): - - def __init__(self, *, uuid: str, resource_page_uuid: str | None, - annotations: AnnotationsT): - super().__init__( - uuid=uuid, - ref_type='ResourcePageReference', - annotations=annotations, - ) - self.resource_page_uuid = resource_page_uuid - - self.resource_page: ResourcePage | None = None - - def resolve_links(self, ctx): - if self.resource_page_uuid in ctx.e.resource_pages: - self.resource_page = ctx.e.resource_pages[self.resource_page_uuid] - - @staticmethod - def load(data: dict, **options): - return ResourcePageReference( - uuid=data['uuid'], - resource_page_uuid=data['resourcePageUuid'], - annotations=_load_annotations(data['annotations']), - ) - - -class Expert: - - def __init__(self, *, uuid: str, name: str, email: str, - annotations: AnnotationsT): - self.uuid = uuid - self.name = name - self.email = email - self.annotations = annotations - - @property - def a(self): - return self.annotations - - def __eq__(self, other): - if not isinstance(other, Expert): - return False - return other.uuid == self.uuid - - @staticmethod - def load(data: dict, **options): - return Expert( - uuid=data['uuid'], - name=data['name'], - email=data['email'], - annotations=_load_annotations(data['annotations']), - ) - - -class Reply(abc.ABC): - - def __init__(self, *, path: str, created_at: datetime, - created_by: SimpleAuthor | None, reply_type: str): - self.path = path - self.created_at = created_at - self.created_by = created_by - self.type = reply_type - - self.question: Question | None = None - self.fragments: list[str] = path.split('.') - - def resolve_links_parent(self, ctx): - question_uuid = self.fragments[-1] - if question_uuid in ctx.e.questions: - self.question = ctx.e.questions.get(question_uuid) - if self.question is not None: - self.question.replies[self.path] = self - - @abc.abstractmethod - def resolve_links(self, ctx): - pass - - @property - def item_title(self) -> str: - """Title to be used if is a reply to first question inside a list item""" - return '' - - @property - def has_direct_item_title(self) -> bool: - return True - - @staticmethod - @abc.abstractmethod - def load(path: str, data: dict, **options): - pass - - -class AnswerReply(Reply): - - def __init__(self, *, path: str, created_at: datetime, - created_by: SimpleAuthor | None, answer_uuid: str): - super().__init__( - path=path, - created_at=created_at, - created_by=created_by, - reply_type='AnswerReply', - ) - self.answer_uuid = answer_uuid - - self.answer: Answer | None = None - - @property - def value(self) -> str | None: - return self.answer_uuid - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - self.answer = ctx.e.answers.get(self.answer_uuid) - - @property - def item_title(self) -> str: - if self.answer is not None: - return self.answer.label - return super().item_title - - @staticmethod - def load(path: str, data: dict, **options): - return AnswerReply( - path=path, - created_at=_datetime(data['createdAt']), - created_by=SimpleAuthor.load(data['createdBy'], **options), - answer_uuid=data['value']['value'], - ) - - -class StringReply(Reply): - - def __init__(self, *, path: str, created_at: datetime, - created_by: SimpleAuthor | None, value: str): - super().__init__( - path=path, - created_at=created_at, - created_by=created_by, - reply_type='StringReply', - ) - self.value = value - - @property - def markdown_html(self) -> str: - return render_markdown(self.value) - - @property - def markdown_plain(self) -> str: - return strip_markdown(self.value) - - @property - def as_number(self) -> float | None: - try: - return float(self.value) - except Exception: - return None - - @property - def as_datetime(self) -> datetime | None: - try: - return dp.parse(self.value) - except Exception: - return None - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - - @property - def item_title(self) -> str: - return self.value - - @staticmethod - def load(path: str, data: dict, **options): - return StringReply( - path=path, - created_at=_datetime(data['createdAt']), - created_by=SimpleAuthor.load(data['createdBy'], **options), - value=data['value']['value'], - ) - - -class ItemListReply(Reply): - - def __init__(self, *, path: str, created_at: datetime, - created_by: SimpleAuthor | None, items: list[str]): - super().__init__( - path=path, - created_at=created_at, - created_by=created_by, - reply_type='ItemListReply', - ) - self.items = items - - @property - def value(self) -> list[str]: - return self.items - - def __iter__(self): - return iter(self.items) - - def __len__(self): - return len(self.items) - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - - @property - def has_direct_item_title(self) -> bool: - return False - - @staticmethod - def load(path: str, data: dict, **options): - return ItemListReply( - path=path, - created_at=_datetime(data['createdAt']), - created_by=SimpleAuthor.load(data['createdBy'], **options), - items=data['value']['value'], - ) - - -class MultiChoiceReply(Reply): - - def __init__(self, *, path: str, created_at: datetime, - created_by: SimpleAuthor | None, choice_uuids: list[str]): - super().__init__( - path=path, - created_at=created_at, - created_by=created_by, - reply_type='MultiChoiceReply', - ) - self.choice_uuids = choice_uuids - - self.choices: list[Choice] = [] - - @property - def value(self) -> list[str]: - return self.choice_uuids - - def __iter__(self): - return iter(self.choices) - - def __len__(self): - return len(self.choices) - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - self.choices = [ctx.e.choices[key] - for key in self.choice_uuids - if key in ctx.e.choices] - - @property - def item_title(self) -> str: - return ', '.join(choice.label for choice in self.choices) - - @staticmethod - def load(path: str, data: dict, **options): - return MultiChoiceReply( - path=path, - created_at=_datetime(data['createdAt']), - created_by=SimpleAuthor.load(data['createdBy'], **options), - choice_uuids=data['value']['value'], - ) - - -class IntegrationReply(Reply): - - def __init__(self, *, path: str, created_at: datetime, - created_by: SimpleAuthor | None, value: str, value_type: str, - raw: typing.Any | None = None): - super().__init__( - path=path, - created_at=created_at, - created_by=created_by, - reply_type='IntegrationReply', - ) - self.value_type = value_type - self.raw = raw - self.value = value - - @property - def is_plain(self) -> bool: - return self.value_type == 'PlainType' - - @property - def is_integration(self) -> bool: - return self.value_type == 'IntegrationType' - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - - @property - def item_title(self) -> str: - non_empty_lines = list(filter( - lambda line: len(line) > 0, - strip_markdown(self.value).splitlines(), - )) - if len(non_empty_lines) > 0: - return non_empty_lines[0] - return super().item_title - - @staticmethod - def load(path: str, data: dict, **options): - return IntegrationReply( - path=path, - created_at=_datetime(data['createdAt']), - created_by=SimpleAuthor.load(data['createdBy'], **options), - value_type=data['value']['value']['type'], - value=data['value']['value'].get('value', ''), - raw=data['value']['value'].get('raw'), - ) - - -class ItemSelectReply(Reply): - - def __init__(self, *, path: str, created_at: datetime, - created_by: SimpleAuthor | None, item_uuid: str): - super().__init__( - path=path, - created_at=created_at, - created_by=created_by, - reply_type='ItemSelectReply', - ) - self.item_uuid = item_uuid - self._item_title: str = 'Item' - - @property - def value(self) -> str: - return self.item_uuid - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - - @property - def item_title(self) -> str: - return self._item_title - - @item_title.setter - def item_title(self, value: str): - self._item_title = value - - @property - def has_direct_item_title(self) -> bool: - return False - - @staticmethod - def load(path: str, data: dict, **options): - return ItemSelectReply( - path=path, - created_at=_datetime(data['createdAt']), - created_by=SimpleAuthor.load(data['createdBy'], **options), - item_uuid=data['value']['value'], - ) - - -class FileReply(Reply): - - def __init__(self, *, path: str, created_at: datetime, - created_by: SimpleAuthor | None, file_uuid: str): - super().__init__( - path=path, - created_at=created_at, - created_by=created_by, - reply_type='FileReply', - ) - self.file_uuid = file_uuid - self.file: ProjectFile | None = None - - @property - def value(self) -> str: - return self.file_uuid - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - self.file = ctx.project.files.get(self.file_uuid) - if self.file is not None: - self.file.reply = self - - @property - def item_title(self) -> str: - if self.file is not None: - return self.file.name - return super().item_title - - @staticmethod - def load(path: str, data: dict, **options): - return FileReply( - path=path, - created_at=_datetime(data['createdAt']), - created_by=SimpleAuthor.load(data['createdBy'], **options), - file_uuid=data['value']['value'], - ) - - -class Answer: - - def __init__(self, *, uuid: str, label: str, advice: str | None, - metric_measures: list[MetricMeasure], followup_uuids: list[str], - annotations: AnnotationsT): - self.uuid = uuid - self.label = label - self.advice = advice - self.metric_measures = metric_measures - self.followup_uuids = followup_uuids - self.annotations = annotations - - self.followups: list[Question] = [] - self.parent: OptionsQuestion | None = None - - @property - def a(self): - return self.annotations - - def __eq__(self, other): - if not isinstance(other, Answer): - return False - return other.uuid == self.uuid - - def resolve_links(self, ctx): - self.followups = [ctx.e.questions[key] - for key in self.followup_uuids - if key in ctx.e.questions] - for followup in self.followups: - followup.parent = self - followup.resolve_links(ctx) - for mm in self.metric_measures: - mm.resolve_links(ctx) - - @staticmethod - def load(data: dict, **options): - mm = [MetricMeasure.load(d, **options) - for d in data['metricMeasures']] - return Answer( - uuid=data['uuid'], - label=data['label'], - advice=data['advice'], - metric_measures=mm, - followup_uuids=data['followUpUuids'], - annotations=_load_annotations(data['annotations']), - ) - - -class Choice: - - def __init__(self, *, uuid: str, label: str, annotations: AnnotationsT): - self.uuid = uuid - self.label = label - self.annotations = annotations - - self.parent: MultiChoiceQuestion | None = None - - @property - def a(self): - return self.annotations - - def __eq__(self, other): - if not isinstance(other, Choice): - return False - return other.uuid == self.uuid - - @staticmethod - def load(data: dict, **options): - return Choice( - uuid=data['uuid'], - label=data['label'], - annotations=_load_annotations(data['annotations']), - ) - - -class Question(abc.ABC): - - def __init__(self, *, uuid: str, q_type: str, title: str, text: str | None, - tag_uuids: list[str], reference_uuids: list[str], - expert_uuids: list[str], required_phase_uuid: str | None, - annotations: AnnotationsT): - self.uuid = uuid - self.type = q_type - self.title = title - self.text = text - self.tag_uuids = tag_uuids - self.reference_uuids = reference_uuids - self.expert_uuids = expert_uuids - self.required_phase_uuid = required_phase_uuid - self.annotations = annotations - - self.is_required: bool | None = None - self.parent: Chapter | ListQuestion | Answer | None = None - self.replies: dict[str, Reply] = {} - self.tags: list[Tag] = [] - self.references: list[Reference] = [] - self.experts: list[Expert] = [] - self.required_phase: Phase = PHASE_NEVER - - @property - def a(self): - return self.annotations - - def __eq__(self, other): - if not isinstance(other, Question): - return False - return other.uuid == self.uuid - - def resolve_links_parent(self, ctx): - self.tags = [ctx.e.tags[key] - for key in self.tag_uuids - if key in ctx.e.tags] - self.experts = [ctx.e.experts[key] - for key in self.expert_uuids - if key in ctx.e.experts] - self.references = [ctx.e.references[key] - for key in self.reference_uuids - if key in ctx.e.references] - for ref in self.references: - ref.resolve_links(ctx) - if self.required_phase_uuid is None or ctx.current_phase is None: - self.is_required = False - else: - self.required_phase = ctx.e.phases.get(self.required_phase_uuid, PHASE_NEVER) - self.is_required = ctx.current_phase.order >= self.required_phase.order - - @abc.abstractmethod - def resolve_links(self, ctx): - pass - - @property - def url_references(self) -> list[URLReference]: - return [r for r in self.references if isinstance(r, URLReference)] - - @property - def resource_page_references(self) -> list[ResourcePageReference]: - return [r for r in self.references if isinstance(r, ResourcePageReference)] - - @property - def cross_references(self) -> list[CrossReference]: - return [r for r in self.references if isinstance(r, CrossReference)] - - @staticmethod - @abc.abstractmethod - def load(data: dict, **options): - pass - - -class ValueQuestionValidation: - SHORT_TYPE: dict[str, str] = { - 'MinLengthQuestionValidation': 'min-length', - 'MaxLengthQuestionValidation': 'max-length', - 'RegexQuestionValidation': 'regex', - 'OrcidQuestionValidation': 'orcid', - 'DoiQuestionValidation': 'doi', - 'MinNumberQuestionValidation': 'min', - 'MaxNumberQuestionValidation': 'max', - 'FromDateQuestionValidation': 'from-date', - 'ToDateQuestionValidation': 'to-date', - 'FromDateTimeQuestionValidation': 'from-datetime', - 'ToDateTimeQuestionValidation': 'to-datetime', - 'FromTimeQuestionValidation': 'from-time', - 'ToTimeQuestionValidation': 'to-time', - 'DomainQuestionValidation': 'domain', - } - VALUE_TYPE: dict[str, type | None] = { - 'MinLengthQuestionValidation': int, - 'MaxLengthQuestionValidation': int, - 'RegexQuestionValidation': str, - 'OrcidQuestionValidation': None, - 'DoiQuestionValidation': None, - 'MinNumberQuestionValidation': float, - 'MaxNumberQuestionValidation': float, - 'FromDateQuestionValidation': str, - 'ToDateQuestionValidation': str, - 'FromDateTimeQuestionValidation': str, - 'ToDateTimeQuestionValidation': str, - 'FromTimeQuestionValidation': str, - 'ToTimeQuestionValidation': str, - 'DomainQuestionValidation': str, - } - - def __init__(self, *, validation_type: str, value: str | float | None = None): - self.type = self.SHORT_TYPE.get(validation_type, 'unknown') - self.full_type = validation_type - self.value = value - - @staticmethod - def load(data: dict, **options): - return ValueQuestionValidation( - validation_type=data['type'], - value=data.get('value'), - ) - - -class ValueQuestion(Question): - - def __init__(self, *, uuid: str, title: str, text: str | None, - tag_uuids: list[str], reference_uuids: list[str], - expert_uuids: list[str], required_phase_uuid: str | None, - value_type: str, annotations: AnnotationsT): - super().__init__( - uuid=uuid, - q_type='ValueQuestion', - title=title, - text=text, - tag_uuids=tag_uuids, - reference_uuids=reference_uuids, - expert_uuids=expert_uuids, - required_phase_uuid=required_phase_uuid, - annotations=annotations, - ) - self.value_type = value_type - self.validations: list[ValueQuestionValidation] = [] - - @property - def a(self): - return self.annotations - - @property - def is_string(self): - return self.value_type == 'StringQuestionValueType' - - @property - def is_text(self): - return self.value_type == 'TextQuestionValueType' - - @property - def is_number(self): - return self.value_type == 'NumberQuestionValueType' - - @property - def is_email(self): - return self.value_type == 'EmailQuestionValueType' - - @property - def is_url(self): - return self.value_type == 'UrlQuestionValueType' - - @property - def is_color(self): - return self.value_type == 'ColorQuestionValueType' - - @property - def is_time(self): - return self.value_type == 'TimeQuestionValueType' - - @property - def is_datetime(self): - return self.value_type == 'DateTimeQuestionValueType' - - @property - def is_date(self): - return self.value_type == 'DateQuestionValueType' - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - - @staticmethod - def load(data: dict, **options): - question = ValueQuestion( - uuid=data['uuid'], - title=data['title'], - text=data['text'], - tag_uuids=data['tagUuids'], - reference_uuids=data['referenceUuids'], - expert_uuids=data['expertUuids'], - required_phase_uuid=data['requiredPhaseUuid'], - value_type=data['valueType'], - annotations=_load_annotations(data['annotations']), - ) - question.validations = [ValueQuestionValidation.load(d, **options) - for d in data['validations']] - return question - - -class OptionsQuestion(Question): - - def __init__(self, *, uuid: str, title: str, text: str | None, - tag_uuids: list[str], reference_uuids: list[str], - expert_uuids: list[str], required_phase_uuid: str | None, - answer_uuids: list[str], annotations: AnnotationsT): - super().__init__( - uuid=uuid, - q_type='OptionsQuestion', - title=title, - text=text, - tag_uuids=tag_uuids, - reference_uuids=reference_uuids, - expert_uuids=expert_uuids, - required_phase_uuid=required_phase_uuid, - annotations=annotations, - ) - self.answer_uuids = answer_uuids - - self.answers: list[Answer] = [] - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - self.answers = [ctx.e.answers[key] - for key in self.answer_uuids - if key in ctx.e.answers] - for answer in self.answers: - answer.parent = self - answer.resolve_links(ctx) - - @staticmethod - def load(data: dict, **options): - return OptionsQuestion( - uuid=data['uuid'], - title=data['title'], - text=data['text'], - tag_uuids=data['tagUuids'], - reference_uuids=data['referenceUuids'], - expert_uuids=data['expertUuids'], - required_phase_uuid=data['requiredPhaseUuid'], - answer_uuids=data['answerUuids'], - annotations=_load_annotations(data['annotations']), - ) - - -class MultiChoiceQuestion(Question): - - def __init__(self, *, uuid: str, title: str, text: str | None, - tag_uuids: list[str], reference_uuids: list[str], - expert_uuids: list[str], required_phase_uuid: str | None, - choice_uuids: list[str], annotations: AnnotationsT): - super().__init__( - uuid=uuid, - q_type='MultiChoiceQuestion', - title=title, - text=text, - tag_uuids=tag_uuids, - reference_uuids=reference_uuids, - expert_uuids=expert_uuids, - required_phase_uuid=required_phase_uuid, - annotations=annotations, - ) - self.choice_uuids = choice_uuids - - self.choices: list[Choice] = [] - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - self.choices = [ctx.e.choices[key] - for key in self.choice_uuids - if key in ctx.e.choices] - for choice in self.choices: - choice.parent = self - - @staticmethod - def load(data: dict, **options): - return MultiChoiceQuestion( - uuid=data['uuid'], - title=data['title'], - text=data['text'], - tag_uuids=data['tagUuids'], - reference_uuids=data['referenceUuids'], - expert_uuids=data['expertUuids'], - required_phase_uuid=data['requiredPhaseUuid'], - choice_uuids=data['choiceUuids'], - annotations=_load_annotations(data['annotations']), - ) - - -class ListQuestion(Question): - - def __init__(self, *, uuid: str, title: str, text: str, - tag_uuids: list[str], reference_uuids: list[str], - expert_uuids: list[str], required_phase_uuid: str | None, - followup_uuids: list[str], annotations: AnnotationsT): - super().__init__( - uuid=uuid, - q_type='ListQuestion', - title=title, - text=text, - tag_uuids=tag_uuids, - reference_uuids=reference_uuids, - expert_uuids=expert_uuids, - required_phase_uuid=required_phase_uuid, - annotations=annotations, - ) - self.followup_uuids = followup_uuids - - self.followups: list[Question] = [] - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - self.followups = [ctx.e.questions[key] - for key in self.followup_uuids - if key in ctx.e.questions] - for followup in self.followups: - followup.parent = self - followup.resolve_links(ctx) - - @staticmethod - def load(data: dict, **options): - return ListQuestion( - uuid=data['uuid'], - title=data['title'], - text=data['text'], - tag_uuids=data['tagUuids'], - reference_uuids=data['referenceUuids'], - expert_uuids=data['expertUuids'], - required_phase_uuid=data['requiredPhaseUuid'], - followup_uuids=data['itemTemplateQuestionUuids'], - annotations=_load_annotations(data['annotations']), - ) - - -class IntegrationQuestion(Question): - - def __init__(self, *, uuid: str, title: str, text: str | None, - tag_uuids: list[str], reference_uuids: list[str], - expert_uuids: list[str], required_phase_uuid: str | None, - integration_uuid: str | None, variables: dict[str, str], - annotations: AnnotationsT): - super().__init__( - uuid=uuid, - q_type='IntegrationQuestion', - title=title, - text=text, - tag_uuids=tag_uuids, - reference_uuids=reference_uuids, - expert_uuids=expert_uuids, - required_phase_uuid=required_phase_uuid, - annotations=annotations, - ) - self.variables = variables - self.integration_uuid = integration_uuid - - self.integration: Integration | None = None - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - self.integration = ctx.e.integrations.get( - self.integration_uuid, - None, - ) - - @staticmethod - def load(data: dict, **options): - return IntegrationQuestion( - uuid=data['uuid'], - title=data['title'], - text=data['text'], - tag_uuids=data['tagUuids'], - reference_uuids=data['referenceUuids'], - expert_uuids=data['expertUuids'], - required_phase_uuid=data['requiredPhaseUuid'], - integration_uuid=data['integrationUuid'], - variables=data['variables'], - annotations=_load_annotations(data['annotations']), - ) - - -class ItemSelectQuestion(Question): - - def __init__(self, *, uuid: str, title: str, text: str | None, - tag_uuids: list[str], reference_uuids: list[str], - expert_uuids: list[str], required_phase_uuid: str | None, - list_question_uuid: str | None, annotations: AnnotationsT): - super().__init__( - uuid=uuid, - q_type='ItemSelectQuestion', - title=title, - text=text, - tag_uuids=tag_uuids, - reference_uuids=reference_uuids, - expert_uuids=expert_uuids, - required_phase_uuid=required_phase_uuid, - annotations=annotations, - ) - self.list_question_uuid = list_question_uuid - self.list_question = None - - def resolve_links(self, ctx): - super().resolve_links_parent(ctx) - self.list_question = ctx.e.questions.get(self.list_question_uuid) - - @staticmethod - def load(data: dict, **options): - return ItemSelectQuestion( - uuid=data['uuid'], - title=data['title'], - text=data['text'], - tag_uuids=data['tagUuids'], - reference_uuids=data['referenceUuids'], - expert_uuids=data['expertUuids'], - required_phase_uuid=data['requiredPhaseUuid'], - list_question_uuid=data['listQuestionUuid'], - annotations=_load_annotations(data['annotations']), - ) - - -class FileQuestion(Question): - - def __init__(self, *, uuid, title, text, tag_uuids, reference_uuids, - expert_uuids, required_phase_uuid, max_size, file_types, - annotations): - super().__init__( - uuid=uuid, - q_type='FileQuestion', - title=title, - text=text, - tag_uuids=tag_uuids, - reference_uuids=reference_uuids, - expert_uuids=expert_uuids, - required_phase_uuid=required_phase_uuid, - annotations=annotations, - ) - self.max_size = max_size - self.file_types = file_types - - def resolve_links(self, ctx): - pass - - @staticmethod - def load(data: dict, **options): - return FileQuestion( - uuid=data['uuid'], - title=data['title'], - text=data['text'], - tag_uuids=data['tagUuids'], - reference_uuids=data['referenceUuids'], - expert_uuids=data['expertUuids'], - required_phase_uuid=data['requiredPhaseUuid'], - max_size=data['maxSize'], - file_types=data['fileTypes'], - annotations=_load_annotations(data['annotations']), - ) - - -class Chapter: - - def __init__(self, *, uuid: str, title: str, text: str | None, - question_uuids: list[str], annotations: AnnotationsT): - self.uuid = uuid - self.title = title - self.text = text - self.question_uuids = question_uuids - self.annotations = annotations - - self.questions: list[Question] = [] - self.reports: list[ReportItem] = [] - - @property - def a(self): - return self.annotations - - def __eq__(self, other): - if not isinstance(other, Chapter): - return False - return other.uuid == self.uuid - - def resolve_links(self, ctx): - self.questions = [ctx.e.questions[key] - for key in self.question_uuids - if key in ctx.e.questions] - for question in self.questions: - question.parent = self - question.resolve_links(ctx) - - @staticmethod - def load(data: dict, **options): - return Chapter( - uuid=data['uuid'], - title=data['title'], - text=data['text'], - question_uuids=data['questionUuids'], - annotations=_load_annotations(data['annotations']), - ) - - -_QUESTION_TYPES: dict[str, type[Question]] = { - 'OptionsQuestion': OptionsQuestion, - 'ListQuestion': ListQuestion, - 'ValueQuestion': ValueQuestion, - 'MultiChoiceQuestion': MultiChoiceQuestion, - 'IntegrationQuestion': IntegrationQuestion, - 'ItemSelectQuestion': ItemSelectQuestion, - 'FileQuestion': FileQuestion, -} - - -_REFERENCE_TYPES: dict[str, type[Reference]] = { - 'URLReference': URLReference, - 'ResourcePageReference': ResourcePageReference, - 'CrossReference': CrossReference, -} - - -_INTEGRATION_TYPES: dict[str, type[Integration]] = { - 'ApiIntegration': ApiIntegration, - 'PluginIntegration': PluginIntegration, -} - - -_REPLY_TYPES: dict[str, type[Reply]] = { - 'AnswerReply': AnswerReply, - 'StringReply': StringReply, - 'ItemListReply': ItemListReply, - 'MultiChoiceReply': MultiChoiceReply, - 'IntegrationReply': IntegrationReply, - 'ItemSelectReply': ItemSelectReply, - 'FileReply': FileReply, -} - - -def _load_question(data: dict, **options): - question_type = data['questionType'] - question_class = _QUESTION_TYPES.get(question_type) - if question_class is None: - raise ValueError(f'Unknown question type: {question_type}') - return question_class.load(data, **options) - - -def _load_reference(data: dict, **options): - reference_type = data['referenceType'] - reference_class = _REFERENCE_TYPES.get(reference_type) - if reference_class is None: - raise ValueError(f'Unknown reference type: {reference_type}') - return reference_class.load(data, **options) - - -def _load_integration(data: dict, **options): - integration_type = data['integrationType'] - integration_class = _INTEGRATION_TYPES.get(integration_type) - if integration_class is None: - raise ValueError(f'Unknown integration type: {integration_type}') - return integration_class.load(data, **options) - - -def _load_reply(path: str, data: dict, **options): - reply_type = data['value']['type'] - reply_class = _REPLY_TYPES.get(reply_type) - if reply_class is None: - raise ValueError(f'Unknown reply type: {reply_type}') - return reply_class.load(path, data, **options) - - -class KnowledgeModelEntities: - - def __init__(self): - self.chapters: dict[str, Chapter] = {} - self.questions: dict[str, Question] = {} - self.answers: dict[str, Answer] = {} - self.choices: dict[str, Choice] = {} - self.resource_collections: dict[str, ResourceCollection] = {} - self.resource_pages: dict[str, ResourcePage] = {} - self.references: dict[str, Reference] = {} - self.experts: dict[str, Expert] = {} - self.tags: dict[str, Tag] = {} - self.metrics: dict[str, Metric] = {} - self.phases: dict[str, Phase] = {} - self.integrations: dict[str, Integration] = {} - - @staticmethod - def load(data: dict, **options): - e = KnowledgeModelEntities() - e.chapters = {key: Chapter.load(d, **options) - for key, d in data['chapters'].items()} - e.questions = {key: _load_question(d, **options) - for key, d in data['questions'].items()} - e.answers = {key: Answer.load(d, **options) - for key, d in data['answers'].items()} - e.choices = {key: Choice.load(d, **options) - for key, d in data['choices'].items()} - e.resource_collections = {key: ResourceCollection.load(d, **options) - for key, d in data['resourceCollections'].items()} - e.resource_pages = {key: ResourcePage.load(d, **options) - for key, d in data['resourcePages'].items()} - e.references = {key: _load_reference(d, **options) - for key, d in data['references'].items()} - e.experts = {key: Expert.load(d, **options) - for key, d in data['experts'].items()} - e.tags = {key: Tag.load(d, **options) - for key, d in data['tags'].items()} - e.metrics = {key: Metric.load(d, **options) - for key, d in data['metrics'].items()} - e.phases = {key: Phase.load(d, **options) - for key, d in data['phases'].items()} - e.integrations = {key: _load_integration(d, **options) - for key, d in data['integrations'].items()} - return e - - -class KnowledgeModel: - - def __init__(self, *, uuid: str, chapter_uuids: list[str], tag_uuids: list[str], - metric_uuids: list[str], phase_uuids: list[str], integration_uuids: list[str], - resource_collection_uuids: list[str], entities: KnowledgeModelEntities, - annotations: AnnotationsT): - self.uuid = uuid - self.entities = entities - self.chapter_uuids = chapter_uuids - self.tag_uuids = tag_uuids - self.metric_uuids = metric_uuids - self.phase_uuids = phase_uuids - self.resource_collection_uuids = resource_collection_uuids - self.integration_uuids = integration_uuids - self.annotations = annotations - - self.chapters: list[Chapter] = [] - self.tags: list[Tag] = [] - self.metrics: list[Metric] = [] - self.phases: list[Phase] = [] - self.resource_collections: list[ResourceCollection] = [] - self.integrations: list[Integration] = [] - - @property - def a(self): - return self.annotations - - @property - def e(self): - return self.entities - - def resolve_links(self, ctx): - self.chapters = [ctx.e.chapters[key] - for key in self.chapter_uuids - if key in ctx.e.chapters] - self.tags = [ctx.e.tags[key] - for key in self.tag_uuids - if key in ctx.e.tags] - self.metrics = [ctx.e.metrics[key] - for key in self.metric_uuids - if key in ctx.e.metrics] - self.phases = [ctx.e.phases[key] - for key in self.phase_uuids - if key in ctx.e.phases] - self.resource_collections = [ctx.e.resource_collections[key] - for key in self.resource_collection_uuids - if key in ctx.e.resource_collections] - self.integrations = [ctx.e.integrations[key] - for key in self.integration_uuids - if key in ctx.e.integrations] - for index, phase in enumerate(self.phases, start=1): - phase.order = index - for chapter in self.chapters: - chapter.resolve_links(ctx) - for resource_collection in self.resource_collections: - resource_collection.resolve_links(ctx) - - @staticmethod - def load(data: dict, **options): - return KnowledgeModel( - uuid=data['uuid'], - chapter_uuids=data['chapterUuids'], - tag_uuids=data['tagUuids'], - metric_uuids=data['metricUuids'], - phase_uuids=data['phaseUuids'], - integration_uuids=data['integrationUuids'], - resource_collection_uuids=data['resourceCollectionUuids'], - entities=KnowledgeModelEntities.load(data['entities'], **options), - annotations=_load_annotations(data['annotations']), - ) - - -class ContextConfig: - - def __init__(self, *, app_title: str, app_title_short: str, client_url: str, - primary_color: str, illustrations_color: str, logo_url: str, - service_name: str, service_name_short: str, service_url: str, - service_domain_name: str): - self.app_title = app_title - self.app_title_short = app_title_short - self.client_url = client_url.rstrip('/') - self.primary_color = Color(primary_color) - self.illustrations_color = Color(illustrations_color) - self.logo_url = logo_url - self.service_name = service_name - self.service_name_short = service_name_short - self.service_url = service_url.rstrip('/') - self.service_domain_name = service_domain_name - - @staticmethod - def load(data: dict, **options): - return ContextConfig( - app_title=data.get('appTitle', ''), - app_title_short=data.get('appTitleShort', ''), - client_url=data.get('clientUrl', ''), - primary_color=data.get('primaryColor', DEFAULT_COLOR), - illustrations_color=data.get('illustrationsColor', DEFAULT_COLOR), - logo_url=data.get('logoUrl', ''), - service_name=data.get('serviceName', ''), - service_name_short=data.get('serviceNameShort', ''), - service_url=data.get('serviceUrl', ''), - service_domain_name=data.get('serviceDomainName', ''), - ) - - -class DocumentTemplateLocale: - - def __init__(self, *, uuid: str, name: str, code: str, - created_at: datetime, updated_at: datetime): - self.uuid = uuid - self.name = name - self.code = code - self.created_at = created_at - self.updated_at = updated_at - - @staticmethod - def load(data: dict, **options): - return DocumentTemplateLocale( - uuid=data['uuid'], - name=data['name'], - code=data['code'], - created_at=_datetime(data['createdAt']), - updated_at=_datetime(data['updatedAt']), - ) - - -class Document: - - def __init__(self, *, uuid: str, name: str, document_template_uuid: str, format_uuid: str, - created_by: User | None, created_at: datetime, language: str | None, - locale: DocumentTemplateLocale | None): - self.uuid = uuid - self.name = name - self.document_template_uuid = document_template_uuid - self.format_uuid = format_uuid - self.created_by = created_by - self.created_at = created_at - self.language = language - self.locale = locale - - @staticmethod - def load(data: dict, **options): - locale_data = data.get('locale') - return Document( - uuid=data['uuid'], - name=data['name'], - document_template_uuid=data['documentTemplateUuid'], - format_uuid=data['formatUuid'], - created_by=User.load(data['createdBy'], **options), - created_at=_datetime(data['createdAt']), - language=data.get('language'), - locale=(DocumentTemplateLocale.load(locale_data, **options) - if locale_data is not None else None), - ) - - -class ProjectVersion: - - def __init__(self, *, uuid: str, event_uuid: str, name: str, description: str | None, - created_at: datetime, updated_at: datetime, - created_by: SimpleAuthor | None): - self.uuid = uuid - self.event_uuid = event_uuid - self.name = name - self.description = description - self.created_at = created_at - self.updated_at = updated_at - self.created_by = created_by - - @staticmethod - def load(data: dict, **options): - return ProjectVersion( - uuid=data['uuid'], - event_uuid=data['eventUuid'], - name=data['name'], - description=data['description'] or '', - created_at=_datetime(data['createdAt']), - updated_at=_datetime(data['updatedAt']), - created_by=SimpleAuthor.load(data['createdBy'], **options), - ) - - -class RepliesContainer: - - def __init__(self, *, replies: dict[str, Reply]): - self.replies = replies - - def __getitem__(self, path: str) -> Reply | None: - return self.get(path) - - def __len__(self) -> int: - return len(self.replies) - - def get(self, path: str, default=None) -> Reply | None: - return self.replies.get(path, default) - - def iterate_by_prefix(self, path_prefix: str) -> typing.Iterable[Reply]: - return (r for path, r in self.replies.items() if path.startswith(path_prefix)) - - def iterate_by_suffix(self, path_suffix: str) -> typing.Iterable[Reply]: - return (r for path, r in self.replies.items() if path.endswith(path_suffix)) - - def values(self) -> typing.Iterable[Reply]: - return self.replies.values() - - def keys(self) -> typing.Iterable[str]: - return self.replies - - def items(self) -> typing.ItemsView[str, Reply]: - return self.replies.items() - - -class ProjectFile: - - def __init__(self, *, uuid: str, file_name: str, file_size: int, - content_type: str): - self.uuid = uuid - self.name = file_name - self.size = file_size - self.content_type = content_type - - self.reply: FileReply | None = None - self.download_url: str = '' - self.project_uuid: str | None = None - - def resolve_links(self, ctx): - self.project_uuid = ctx.project.uuid - client_url = ctx.config.client_url - self.download_url = (f'{client_url}/projects/{self.project_uuid}' - f'/files/{self.uuid}/download') - - @staticmethod - def load(data: dict, **options): - return ProjectFile( - uuid=data['uuid'], - file_name=data['fileName'], - file_size=data['fileSize'], - content_type=data['contentType'], - ) - - -class Project: - - def __init__(self, *, uuid: str, name: str, description: str | None, - created_by: User, phase_uuid: str | None, language: str | None, - created_at: datetime, updated_at: datetime): - self.uuid = uuid - self.name = name - self.description = description - self.created_by = created_by - self.phase_uuid = phase_uuid - self.language: str | None = language - self.created_at = created_at - self.updated_at = updated_at - - self.version: ProjectVersion | None = None - self.versions: list[ProjectVersion] = [] - self.files: dict[str, ProjectFile] = {} - self.todos: list[str] = [] - self.project_tags: list[str] = [] - self.phase: Phase = PHASE_NEVER - - self.replies: RepliesContainer = RepliesContainer(replies={}) - - def resolve_links(self, ctx): - for reply in self.replies.values(): - reply.resolve_links(ctx) - for file in self.files.values(): - file.resolve_links(ctx) - - @staticmethod - def load(data: dict, **options): - entity_uuid = data['uuid'] - versions = [ProjectVersion.load(d, **options) - for d in data['versions']] - version = None - replies = {p: _load_reply(p, d, **options) - for p, d in data['replies'].items()} - files = {d['uuid']: ProjectFile.load(d, **options) - for d in data.get('files', [])} - for v in versions: - if v.uuid == data['versionUuid']: - version = v - entity = Project( - uuid=entity_uuid, - name=data['name'], - description=data['description'] or '', - created_by=User.load(data['createdBy'], **options), - phase_uuid=data['phaseUuid'], - language=data.get('language'), - created_at=_datetime(data['createdAt']), - updated_at=_datetime(data['updatedAt']), - ) - entity.version = version - entity.versions = versions - entity.files = files - entity.project_tags = data.get('projectTags', []) - entity.replies.replies = replies - entity.todos = [k for k, v in data.get('labels', {}).items() if TODO_LABEL_UUID in v] - return entity - - -class KnowledgeModelPackage: - - def __init__(self, *, org_id: str, km_id: str, version: str, versions: list[str], - name: str, description: str, language: str, created_at: datetime): - self.organization_id = org_id - self.km_id = km_id - self.version = version - self.versions = versions - self.name = name - self.description = description - self.language = language - self.created_at = created_at - - self.id: str = f'{org_id}:{km_id}:{version}' - - @property - def org_id(self): - return self.organization_id - - @staticmethod - def load(data: dict, **options): - return KnowledgeModelPackage( - org_id=data['organizationId'], - km_id=data['kmId'], - version=data['version'], - versions=data['versions'], - name=data['name'], - description=data.get('description', ''), - language=data.get('language', 'en'), - created_at=_datetime(data['createdAt']), - ) - - -class ReportIndication: - - def __init__(self, *, indication_type: str, answered: int, unanswered: int): - self.indication_type = indication_type - self.answered = answered - self.unanswered = unanswered - - @property - def total(self) -> int: - return self.answered + self.unanswered - - @property - def percentage(self) -> float: - if self.total == 0: - return 0 - return self.answered / self.total - - @property - def is_for_phase(self): - return self.indication_type == 'PhasesAnsweredIndication' - - @property - def is_overall(self): - return self.indication_type == 'AnsweredIndication' - - @staticmethod - def load(data: dict, **options): - return ReportIndication( - indication_type=data['indicationType'], - answered=int(data['answeredQuestions']), - unanswered=int(data['unansweredQuestions']), - ) - - -class ReportMetric: - - def __init__(self, *, measure: float | None, metric_uuid: str): - self.measure = measure - self.metric_uuid = metric_uuid - - self.metric: Metric | None = None - - def resolve_links(self, ctx): - if self.metric_uuid in ctx.e.metrics: - self.metric = ctx.e.metrics[self.metric_uuid] - - @staticmethod - def load(data: dict, **options): - measure = data['measure'] - return ReportMetric( - # null when nothing answered in the chapter contributes to the metric - measure=None if measure is None else float(measure), - metric_uuid=data['metricUuid'], - ) - - -class ReportItem: - - def __init__(self, *, indications: list[ReportIndication], metrics: list[ReportMetric], - chapter_uuid: str | None): - self.indications = indications - self.metrics = metrics - self.chapter_uuid = chapter_uuid - - self.chapter: Chapter | None = None - - def resolve_links(self, ctx): - for m in self.metrics: - m.resolve_links(ctx) - if self.chapter_uuid is not None and self.chapter_uuid in ctx.e.chapters: - self.chapter = ctx.e.chapters[self.chapter_uuid] - if self.chapter is not None: - self.chapter.reports.append(self) - - @staticmethod - def load(data: dict, **options): - return ReportItem( - indications=[ReportIndication.load(d, **options) - for d in data['indications']], - metrics=[ReportMetric.load(d, **options) - for d in data['metrics']], - chapter_uuid=data.get('chapterUuid'), - ) - - -class Report: - - def __init__(self, *, uuid: str, created_at: datetime, - updated_at: datetime, chapter_reports: list[ReportItem], - total_report: ReportItem): - self.uuid = uuid - self.created_at = created_at - self.updated_at = updated_at - self.total_report = total_report - self.chapter_reports = chapter_reports - - def resolve_links(self, ctx): - self.total_report.resolve_links(ctx) - for report in self.chapter_reports: - report.resolve_links(ctx) - - @staticmethod - def load(data: dict, **options): - return Report( - uuid=data['uuid'], - created_at=_datetime(data['createdAt']), - updated_at=_datetime(data['updatedAt']), - total_report=ReportItem.load(data['totalReport'], **options), - chapter_reports=[ReportItem.load(d, **options) - for d in data['chapterReports']], - ) - - -class UserGroup: - - def __init__(self, *, uuid: str, name: str, description: str | None, private: bool, - created_at: datetime, updated_at: datetime): - self.uuid = uuid - self.name = name - self.description = description - self.private = private - self.created_at = created_at - self.updated_at = updated_at - - self.members: list[UserGroupMember] = [] - - @staticmethod - def load(data: dict, **options): - ug = UserGroup( - uuid=data['uuid'], - name=data['name'], - description=data['description'], - private=data['private'], - created_at=_datetime(data['createdAt']), - updated_at=_datetime(data['updatedAt']), - ) - ug.members = [UserGroupMember.load(d, **options) - for d in data.get('users', [])] - return ug - - -class UserGroupMember: - - def __init__(self, *, uuid: str, first_name: str, last_name: str, gravatar_hash: str, - image_url: str | None, membership_type: str): - self.uuid = uuid - self.first_name = first_name - self.last_name = last_name - self.gravatar_hash = gravatar_hash - self.image_url = image_url - self.membership_type = membership_type - - @staticmethod - def load(data: dict, **options): - membership = 'member' - if 'owner' in data['membershipType'].lower(): - membership = 'owner' - return UserGroupMember( - uuid=data['uuid'], - first_name=data['firstName'], - last_name=data['lastName'], - gravatar_hash=data['gravatarHash'], - image_url=data['imageUrl'], - membership_type=membership, - ) - - -class DocumentContextUserPermission: - - def __init__(self, *, user: User | None, permissions: list[str]): - self.user = user - self.permissions = permissions - - @property - def is_viewer(self): - return 'VIEW' in self.permissions - - @property - def is_commenter(self): - return 'COMMENT' in self.permissions - - @property - def is_editor(self): - return 'EDIT' in self.permissions - - @property - def is_owner(self): - return 'ADMIN' in self.permissions - - @staticmethod - def load(data: dict, **options): - return DocumentContextUserPermission( - user=User.load(data['user'], **options), - permissions=data['perms'], - ) - - -class DocumentContextUserGroupPermission: - - def __init__(self, *, group: UserGroup | None, permissions: list[str]): - self.group = group - self.permissions = permissions - - @property - def is_viewer(self): - return 'VIEW' in self.permissions - - @property - def is_commenter(self): - return 'COMMENT' in self.permissions - - @property - def is_editor(self): - return 'EDIT' in self.permissions - - @property - def is_owner(self): - return 'ADMIN' in self.permissions - - @staticmethod - def load(data: dict, **options): - return DocumentContextUserGroupPermission( - group=UserGroup.load(data['group'], **options), - permissions=data['perms'], - ) - - -class DocumentContext: - """Document Context smart representation""" - - def __init__(self, *, ctx, **options): - check_metamodel_version( - metamodel_version=str(ctx.get('metamodelVersion', '0')), - ) - self.config = ContextConfig.load(ctx['config'], **options) - self.km = KnowledgeModel.load(ctx['knowledgeModel'], **options) - self.project = Project.load(ctx['project'], **options) - self.report = Report.load(ctx['report'], **options) - self.document = Document.load(ctx['document'], **options) - self.km_package = KnowledgeModelPackage.load(ctx['knowledgeModelPackage'], **options) - self.organization = Organization.load(ctx['organization'], **options) - self.current_phase: Phase = PHASE_NEVER - - self.users = [DocumentContextUserPermission.load(d, **options) - for d in ctx['users']] - self.groups = [DocumentContextUserGroupPermission.load(d, **options) - for d in ctx['groups']] - - @property - def e(self) -> KnowledgeModelEntities: - return self.km.entities - - @property - def cfg(self) -> ContextConfig: - return self.config - - @property - def org(self) -> Organization: - return self.organization - - @property - def pkg(self) -> KnowledgeModelPackage: - return self.km_package - - @property - def doc(self) -> Document: - return self.document - - @property - def replies(self) -> RepliesContainer: - return self.project.replies - - def resolve_links(self): - phase_uuid = self.project.phase_uuid - if phase_uuid is not None and phase_uuid in self.e.phases: - self.current_phase = self.e.phases[phase_uuid] - self.project.phase = self.current_phase - self.km.resolve_links(self) - self.report.resolve_links(self) - self.project.resolve_links(self) - - rv = ReplyVisitor(context=self) - rv.visit() - for reply in self.replies.values(): - if isinstance(reply, ItemSelectReply): - reply.item_title = rv.item_titles.get(reply.item_uuid, 'Item') - - -class ReplyVisitor: - - def __init__(self, *, context: DocumentContext): - self.item_titles: dict[str, str] = {} - self.item_same_as: dict[str, str] = {} - self._set_also: dict[str, list[str]] = {} - self.context = context - - def visit(self): - for chapter in self.context.km.chapters: - self._visit_chapter(chapter) - self._post_process_item_titles() - - def _visit_chapter(self, chapter: Chapter): - for question in chapter.questions: - self._visit_question(question, path=chapter.uuid) - - def _visit_question(self, question: Question, path: str): - new_path = f'{path}.{question.uuid}' - if isinstance(question, ListQuestion): - self._visit_list_question(question, new_path) - elif isinstance(question, OptionsQuestion): - self._visit_options_question(new_path) - - def _visit_list_question(self, question: ListQuestion, path: str): - reply = self.context.replies.get(path) - if reply is None or not isinstance(reply, ItemListReply): - return - for n, item_uuid in enumerate(reply.items, start=1): - item_path = f'{path}.{item_uuid}' - self.item_titles[item_uuid] = f'Item {n}' - self._prepare_item_title(question, item_uuid, item_path) - - for followup in question.followups: - self._visit_question(followup, path=item_path) - - def _visit_options_question(self, path: str): - reply = self.context.replies.get(path) - if reply is None or not isinstance(reply, AnswerReply) or reply.answer is None: - return - - new_path = f'{path}.{reply.answer_uuid}' - for followup in reply.answer.followups: - self._visit_question(followup, path=new_path) - - def _prepare_item_title(self, question: ListQuestion, item_uuid: str, item_path: str): - if len(question.followups) == 0: - return - followup = question.followups[0] - followup_path = f'{item_path}.{followup.uuid}' - - reply = self.context.replies.get(followup_path) - if reply is None: - return - - if isinstance(reply, ItemListReply) and len(reply.items) > 0: - self.item_same_as[item_uuid] = reply.items[0] - elif isinstance(reply, ItemSelectReply) and reply.item_uuid: - self.item_same_as[item_uuid] = reply.item_uuid - elif reply.has_direct_item_title: - self.item_titles[item_uuid] = reply.item_title - - def _post_process_item_titles(self): - to_process = list(self.item_same_as.keys()) - while len(to_process) > 0: - item_uuid = to_process.pop() - visited = set() - target_item_uuid = item_uuid - visited.add(target_item_uuid) - while target_item_uuid in self.item_same_as: - same_as = self.item_same_as[target_item_uuid] - if same_as in visited: - break - visited.add(same_as) - target_item_uuid = same_as - if target_item_uuid in self.item_titles: - self.item_titles[item_uuid] = self.item_titles[target_item_uuid] diff --git a/packages/dsw-document-worker/dsw/document_worker/model/utils.py b/packages/dsw-document-worker/dsw/document_worker/model/utils.py deleted file mode 100644 index a2b07d9a..00000000 --- a/packages/dsw-document-worker/dsw/document_worker/model/utils.py +++ /dev/null @@ -1,130 +0,0 @@ -from __future__ import annotations - -import io -import re -import typing - -import markdown -import markdown.preprocessors -import markupsafe - -from ..sanitizer import sanitize_html - - -def unmark_element(element, stream=None): - if stream is None: - stream = io.StringIO() - if element.text: - stream.write(element.text) - for sub in element: - unmark_element(sub, stream) - if element.tail: - stream.write(element.tail) - return stream.getvalue() - - -# patching Markdown -markdown.Markdown.output_formats['plain'] = unmark_element -__md = markdown.Markdown(output_format='plain') -__md.stripTopLevelTags = False - - -def strip_markdown(text): - return __md.convert(text) - - -class DSWMarkdownExt(markdown.extensions.Extension): - - @typing.override - def extendMarkdown(self, md): - md.preprocessors.register(DSWMarkdownProcessor(md), 'dsw_markdown', 27) - md.registerExtension(self) - - -class DSWMarkdownProcessor(markdown.preprocessors.Preprocessor): - LI_RE = re.compile(r'^[ ]*((\d+\.)|[*+-])[ ]+.*') - # Opening of a fenced code block, mirroring the `fenced_code` extension: - # a run of at least three backticks or tildes at the start of the line. - FENCE_RE = re.compile(r'^(?P`{3,}|~{3,})') - - def __init__(self, md): - super().__init__(md) - - def _find_fence_close(self, lines, start, fence): - # `fenced_code` closes on the exact same marker (same character and - # length), optionally followed by trailing spaces. - for index in range(start, len(lines)): - if lines[index].rstrip(' ') == fence: - return index - return None - - def run(self, lines): - prev_li = False - new_lines = [] - index = 0 - - while index < len(lines): - line = lines[index] - - # Copy complete fenced code blocks verbatim so that list-like or - # backslash-terminated code lines are not rewritten. A block counts - # only when a matching closing fence exists later, exactly as the - # `fenced_code` extension requires; an unterminated fence is treated - # as ordinary text. - fence_match = self.FENCE_RE.match(line) - if fence_match is not None: - fence = fence_match.group('fence') - close = self._find_fence_close(lines, index + 1, fence) - if close is not None: - new_lines.extend(lines[index:close + 1]) - prev_li = False - index = close + 1 - continue - - # Add line break before the first list item - if self.LI_RE.match(line): - if not prev_li: - new_lines.append('') - prev_li = True - elif line == '': - prev_li = False - - # Replace trailing un-escaped backslash with (supported) two spaces - _line = line.rstrip('\\') - if line[-1:] == '\\' and (len(line) - len(_line)) % 2 == 1: - new_lines.append(f'{line[:-1]} ') - index += 1 - continue - - new_lines.append(line) - index += 1 - - return new_lines - - -def render_markdown(md_text: str, sanitize: bool = True): - """Render Markdown to HTML. - - The result is sanitized by default as it may contain raw HTML coming from - end users (e.g. project replies) that would otherwise be passed through - verbatim. Use ``sanitize=False`` only for content fully controlled by the - document template itself. - """ - if md_text is None: - return '' - html = markdown.markdown( - text=md_text, - extensions=[ - DSWMarkdownExt(), - 'fenced_code', - 'pymdownx.tilde', - ], - extension_configs={ - # only enable ~~strikethrough~~, keep single ~tilde~ literal - 'pymdownx.tilde': {'subscript': False}, - }, - ) - if not sanitize: - # explicitly requested raw HTML pass-through (template-controlled content) - return markupsafe.Markup(html) # noqa: S704 - return markupsafe.Markup(sanitize_html(html)) diff --git a/packages/dsw-document-worker/dsw/document_worker/sanitizer.py b/packages/dsw-document-worker/dsw/document_worker/sanitizer.py deleted file mode 100644 index 2c8a223b..00000000 --- a/packages/dsw-document-worker/dsw/document_worker/sanitizer.py +++ /dev/null @@ -1,94 +0,0 @@ -from __future__ import annotations - -import re - -import nh3 - - -# Attributes that hold a URL and therefore need scheme checking -URL_ATTRIBUTES = frozenset({ - ('a', 'href'), - ('area', 'href'), - ('blockquote', 'cite'), - ('del', 'cite'), - ('img', 'src'), - ('ins', 'cite'), - ('q', 'cite'), -}) - -# Tags allowed in the sanitized output (nh3 defaults + table footer) -ALLOWED_TAGS = nh3.ALLOWED_TAGS | {'tfoot'} - -# Attributes allowed per tag ('*' applies to all tags) -ALLOWED_ATTRIBUTES: dict[str, set[str]] = { - tag: set(attrs) for tag, attrs in nh3.ALLOWED_ATTRIBUTES.items() -} -ALLOWED_ATTRIBUTES['*'] = {'class', 'dir', 'id', 'lang', 'title'} -ALLOWED_ATTRIBUTES.setdefault('a', set()).update({'href', 'hreflang', 'target'}) -ALLOWED_ATTRIBUTES.setdefault('ol', set()).update({'start', 'type'}) -ALLOWED_ATTRIBUTES.setdefault('time', set()).add('datetime') -for _tag in ('div', 'img', 'p', 'span', 'table', 'td', 'th', 'tr'): - ALLOWED_ATTRIBUTES.setdefault(_tag, set()).add('style') - -# URL schemes allowed at all (further restricted per attribute below) -ALLOWED_URL_SCHEMES = frozenset({'data', 'http', 'https', 'mailto'}) - -# CSS properties allowed in the "style" attribute; anything that can reference -# an external resource (e.g. background-image with url(...)) is left out -ALLOWED_STYLE_PROPERTIES = frozenset({ - 'background-color', 'border', 'border-bottom', 'border-collapse', - 'border-color', 'border-left', 'border-right', 'border-style', - 'border-top', 'border-width', 'color', 'font-family', 'font-size', - 'font-style', 'font-variant', 'font-weight', 'height', 'letter-spacing', - 'line-height', 'margin', 'margin-bottom', 'margin-left', 'margin-right', - 'margin-top', 'padding', 'padding-bottom', 'padding-left', 'padding-right', - 'padding-top', 'text-align', 'text-decoration', 'text-indent', - 'text-transform', 'vertical-align', 'white-space', 'width', 'word-break', -}) - -_SCHEME_PATTERN = re.compile(r'^([a-zA-Z][a-zA-Z0-9+.\-]*):') -_IMG_SCHEMES = frozenset({'http', 'https'}) -_LINK_SCHEMES = frozenset({'http', 'https', 'mailto'}) - - -def _url_scheme(value: str) -> str | None: - # Strip whitespace (incl. embedded tabs/newlines used to obfuscate schemes) - url = ''.join(value.split()).lower() - match = _SCHEME_PATTERN.match(url) - if match is None: - return None # relative URL - return match.group(1) - - -def _attribute_filter(tag: str, attr: str, value: str) -> str | None: - if (tag, attr) not in URL_ATTRIBUTES: - return value - scheme = _url_scheme(value) - if scheme is None: - return value # relative URLs are resolved against the document base - if tag == 'img': - if scheme in _IMG_SCHEMES: - return value - if ''.join(value.split()).lower().startswith('data:image/'): - return value - return None - if scheme in _LINK_SCHEMES: - return value - return None - - -def sanitize_html(html: str) -> str: - """Sanitize an HTML fragment using a strict allow-list. - - Removes scripting (tags, event handlers), embedded content (iframe, - object, embed, svg, ...), stylesheets, and URLs with unexpected schemes - such as ``file:``, ``javascript:`` or ``data:text/html``. - """ - return nh3.clean( - html, - tags=ALLOWED_TAGS, - attributes=ALLOWED_ATTRIBUTES, - url_schemes=set(ALLOWED_URL_SCHEMES), - attribute_filter=_attribute_filter, - filter_style_properties=set(ALLOWED_STYLE_PROPERTIES), - ) diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/extraction.py b/packages/dsw-document-worker/dsw/document_worker/templates/extraction.py index 34ced936..ecbf1e3a 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/extraction.py +++ b/packages/dsw-document-worker/dsw/document_worker/templates/extraction.py @@ -2,7 +2,7 @@ import abc -from ..model import context as dc +from dsw.models.document_context import graph as dc DEFAULT_CONFIG = { diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/filters.py b/packages/dsw-document-worker/dsw/document_worker/templates/filters.py index 6e3183d3..c56433b2 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/filters.py +++ b/packages/dsw-document-worker/dsw/document_worker/templates/filters.py @@ -8,12 +8,11 @@ import dateutil.parser as dp import jinja2 -from dsw.document_worker.utils import byte_size_format +from dsw.models.document_context.graph import DocumentContext +from dsw.models.document_context.rendering import render_markdown from ..exceptions import JobError -from ..model import DocumentContext -from ..model.utils import render_markdown -from ..utils import JinjaEnvironment +from ..utils import JinjaEnvironment, byte_size_format from .extraction import extract_replies from .tests import tests diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/steps/template.py b/packages/dsw-document-worker/dsw/document_worker/templates/steps/template.py index 2f94e2b0..59849a7b 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/steps/template.py +++ b/packages/dsw-document-worker/dsw/document_worker/templates/steps/template.py @@ -9,11 +9,12 @@ import jinja2.exceptions import rdflib +from dsw.models.document_context.graph import ProjectFile + from ...consts import DEFAULT_ENCODING, JINJA_EXTENSIONS, JINJA_I18N_TRIMMED from ...context import Context from ...documents import DocumentFile, FileFormat, FileFormats -from ...model.context import ProjectFile -from ...model.http import RequestsWrapper +from ...http import RequestsWrapper from ...urls import UrlPolicy from ...utils import JinjaEnvironment from ..filters import filters diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/templates.py b/packages/dsw-document-worker/dsw/document_worker/templates/templates.py index 58daa5f8..d683feb2 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/templates.py +++ b/packages/dsw-document-worker/dsw/document_worker/templates/templates.py @@ -22,9 +22,9 @@ DBDocumentTemplateAsset, DBDocumentTemplateFile, ) + from dsw.models.document_context.graph import ProjectFile from ..documents import DocumentFile - from ..model.context import ProjectFile LOG = logging.getLogger(__name__) diff --git a/packages/dsw-document-worker/dsw/document_worker/utils.py b/packages/dsw-document-worker/dsw/document_worker/utils.py index 916cc9db..55864d04 100644 --- a/packages/dsw-document-worker/dsw/document_worker/utils.py +++ b/packages/dsw-document-worker/dsw/document_worker/utils.py @@ -4,8 +4,6 @@ import jinja2.sandbox -from . import consts - _BYTE_SIZES = ['B', 'kB', 'MB', 'GB', 'TB', 'PB', 'EB', 'ZB'] @@ -22,21 +20,6 @@ def byte_size_format(num: float): return f'{_round_size(num)} YB' -def check_metamodel_version(metamodel_version: str): - version_parts = metamodel_version.split('.') - try: - major = int(version_parts[0]) if len(version_parts) > 0 else 0 - minor = int(version_parts[1]) if len(version_parts) > 1 else 0 - except ValueError as e: - raise ValueError(f'Invalid metamodel version format: {metamodel_version}') from e - if major != consts.CURRENT_METAMODEL_MAJOR: - raise ValueError(f'Unsupported metamodel version: {metamodel_version} ' - f'(expected major version {consts.CURRENT_METAMODEL_MAJOR})') - if minor < consts.CURRENT_METAMODEL_MINOR: - raise ValueError(f'Unsupported metamodel version: {metamodel_version} ' - f'(expected at least {consts.CURRENT_METAMODEL_MINOR} minor version)') - - class JinjaEnvironment(jinja2.sandbox.SandboxedEnvironment): def is_safe_attribute(self, obj: typing.Any, attr: str, value: typing.Any) -> bool: diff --git a/packages/dsw-document-worker/dsw/document_worker/worker.py b/packages/dsw-document-worker/dsw/document_worker/worker.py index 6707d990..db9ed471 100644 --- a/packages/dsw-document-worker/dsw/document_worker/worker.py +++ b/packages/dsw-document-worker/dsw/document_worker/worker.py @@ -10,6 +10,7 @@ from dsw.command_queue import CommandQueue, CommandWorker from dsw.config.sentry import SentryReporter from dsw.database.database import Database +from dsw.models.document_context.graph import check_metamodel_version from dsw.storage import S3Storage from . import consts @@ -21,7 +22,7 @@ from .pot import PotFileJob from .templates import Format, Template, TemplateRegistry from .templates.locales import TemplateLocale -from .utils import byte_size_format, check_metamodel_version +from .utils import byte_size_format if typing.TYPE_CHECKING: diff --git a/packages/dsw-document-worker/lambda.Dockerfile b/packages/dsw-document-worker/lambda.Dockerfile index d041a324..f7c68961 100644 --- a/packages/dsw-document-worker/lambda.Dockerfile +++ b/packages/dsw-document-worker/lambda.Dockerfile @@ -28,6 +28,7 @@ RUN python -m pip wheel --no-deps --wheel-dir=/app/wheels \ /app/packages/dsw-command-queue \ /app/packages/dsw-config \ /app/packages/dsw-database \ + /app/packages/dsw-models \ /app/packages/dsw-storage \ /app/packages/dsw-document-worker/addons/* \ /app/packages/dsw-document-worker diff --git a/packages/dsw-document-worker/pyproject.toml b/packages/dsw-document-worker/pyproject.toml index dfee7ca2..c324b002 100644 --- a/packages/dsw-document-worker/pyproject.toml +++ b/packages/dsw-document-worker/pyproject.toml @@ -43,8 +43,8 @@ artifacts = ["dsw/*/build_info.py"] artifacts = ["dsw/*/build_info.py"] [tool.hatch.metadata.hooks.uv-dynamic-versioning] -dependencies = ["Babel", "click", "Jinja2", "Markdown", "MarkupSafe", "nh3", "panflute", "pathvalidate", "pluggy", "polib", "pymdown-extensions", "python-dateutil", "python-slugify", "rdflib", "rdflib-jsonld", "requests", "sentry-sdk", "tenacity", "weasyprint", "XlsxWriter", "dsw-command-queue=={{ version }}", "dsw-config=={{ version }}", "dsw-database=={{ version }}", "dsw-storage=={{ version }}"] -optional-dependencies = { test = ["pytest", "dsw-models[rendering]=={{ version }}"] } +dependencies = ["Babel", "click", "Jinja2", "panflute", "pathvalidate", "pluggy", "polib", "python-dateutil", "python-slugify", "rdflib", "rdflib-jsonld", "requests", "sentry-sdk", "tenacity", "weasyprint", "XlsxWriter", "dsw-command-queue=={{ version }}", "dsw-config=={{ version }}", "dsw-database=={{ version }}", "dsw-models[rendering]=={{ version }}", "dsw-storage=={{ version }}"] +optional-dependencies = { test = ["pytest"] } [tool.uv-dynamic-versioning] vcs = "git" diff --git a/packages/dsw-document-worker/support/DocumentContext.md b/packages/dsw-document-worker/support/DocumentContext.md index 68d8a265..16082282 100644 --- a/packages/dsw-document-worker/support/DocumentContext.md +++ b/packages/dsw-document-worker/support/DocumentContext.md @@ -9,7 +9,7 @@ This document describes the structure of document context provided by the Docume * All data types are using Python, e.g., `str` is textual string, `Optional[str]` is a string or `None`, `list[str]` is a list of strings. * We use `snake_case` for naming of attributes and variables, `PascalCase` is used for class names. * `datetime` is the standard [`datetime.datetime`](https://docs.python.org/3/library/datetime.html#datetime-objects). -* You can investigate the [`context`](../document_worker/model/context.py) module; however, constructs that are not documented here may change in any version without an explicit notice. +* You can investigate the [`graph`](../../dsw-models/dsw/models/document_context/graph.py) module of `dsw-models`; however, constructs that are not documented here may change in any version without an explicit notice. ## Diagram diff --git a/packages/dsw-document-worker/tests/test_context_document.py b/packages/dsw-document-worker/tests/test_context_document.py index 87d2cd9b..e898c139 100644 --- a/packages/dsw-document-worker/tests/test_context_document.py +++ b/packages/dsw-document-worker/tests/test_context_document.py @@ -1,4 +1,4 @@ -from dsw.document_worker.model.context import Document +from dsw.models.document_context.graph import Document BASE_DATA = { diff --git a/packages/dsw-document-worker/tests/test_context_parity.py b/packages/dsw-document-worker/tests/test_context_reference.py similarity index 54% rename from packages/dsw-document-worker/tests/test_context_parity.py rename to packages/dsw-document-worker/tests/test_context_reference.py index 56faf9b5..59d038f6 100644 --- a/packages/dsw-document-worker/tests/test_context_parity.py +++ b/packages/dsw-document-worker/tests/test_context_reference.py @@ -1,26 +1,23 @@ -"""Parity of the template-facing document context between the worker and dsw-models. +"""The template-facing document context, built from real contexts by the worker's own filter. -``dsw.models.document_context.graph`` must expose exactly what ``ctx|to_context_obj`` exposes -today. Both object models are built from the same contexts and compared attribute by attribute -(properties included), recursively. +The object model itself lives in ``dsw.models.document_context.graph``; what is asserted here is +that ``ctx|to_context_obj`` keeps producing it for the contexts the worker actually receives — +the synthetic one and one generated over a released knowledge model with replies to (almost) +every question. """ import copy -import datetime import gzip import json import pathlib import uuid -import pytest - -from dsw.document_worker.model import context as worker_context -from dsw.models.document_context import graph as models_context -from dsw.models.document_context.wire import DocumentContext as WireDocumentContext +from dsw.document_worker.templates.filters import to_context_obj from dsw.models.knowledge_model import graph as km_graph from dsw.models.knowledge_model.bundle import compile_bundle from dsw.models.knowledge_model.package import KnowledgeModelBundle from dsw.models.project.report import generate_report from dsw.models.strictness import load +from dsw.models.document_context.wire import DocumentContext as WireDocumentContext MODELS_FIXTURES = pathlib.Path(__file__).parents[2] / 'dsw-models' / 'tests' / 'fixtures' @@ -106,104 +103,34 @@ def visit(question, prefix: str, depth: int) -> None: return ctx -_SCALARS = (str, int, float, bool, type(None), datetime.datetime, uuid.UUID) - - -def assert_same_surface(old, new, path='dc', seen=None): - seen = set() if seen is None else seen - if isinstance(old, _SCALARS) or isinstance(new, _SCALARS): - assert type(old) is type(new) and old == new, f'{path}: {old!r} != {new!r}' - return - if isinstance(old, (list, tuple)): - assert isinstance(new, type(old)) and len(old) == len(new), f'{path}: {old!r} vs {new!r}' - for index, (left, right) in enumerate(zip(old, new, strict=True)): - assert_same_surface(left, right, f'{path}[{index}]', seen) - return - if isinstance(old, dict): - assert isinstance(new, dict) and list(old) == list(new), f'{path}: keys differ' - for key in old: - assert_same_surface(old[key], new[key], f'{path}[{key!r}]', seen) - return - assert type(old).__name__ == type(new).__name__, f'{path}: {type(old)} vs {type(new)}' - if (id(old), id(new)) in seen: - return - seen.add((id(old), id(new))) - names = sorted(name for name in dir(old) if not name.startswith('_')) - assert names == sorted(name for name in dir(new) if not name.startswith('_')), f'{path}: attributes differ' - for name in names: - left, right = _read(old, name), _read(new, name) - if callable(left) and not isinstance(left, type): - assert callable(right), f'{path}.{name}: callable differs' - continue - assert_same_surface(left, right, f'{path}.{name}', seen) - - -def _read(obj, name): - try: - return getattr(obj, name) - except Exception as error: # noqa: BLE001 (compare failing properties too) - return ('raised', type(error).__name__, str(error)) - - -def build_both(ctx: dict): - old = worker_context.DocumentContext(ctx=copy.deepcopy(ctx)) - old.resolve_links() - new = models_context.DocumentContext(ctx=copy.deepcopy(ctx)) - new.resolve_links() - return old, new - - -def test_synthetic_context_parity(): - old, new = build_both(enrich(json.loads(SYNTHETIC_CONTEXT.read_text(encoding='utf-8')))) - assert_same_surface(old, new) +def test_synthetic_context(): + dc = to_context_obj(enrich(json.loads(SYNTHETIC_CONTEXT.read_text(encoding='utf-8')))) + assert dc.km.chapters + assert dc.current_phase.title def test_metric_without_measure(): ctx = enrich(json.loads(SYNTHETIC_CONTEXT.read_text(encoding='utf-8'))) ctx['report']['chapterReports'][0]['metrics'][0]['measure'] = None - old, new = build_both(ctx) - assert old.report.chapter_reports[0].metrics[0].measure is None - assert_same_surface(old, new) + dc = to_context_obj(ctx) + assert dc.report.chapter_reports[0].metrics[0].measure is None -def test_generated_reference_context_parity(): - ctx = enrich(generated_context()) - old, new = build_both(ctx) - assert len(old.replies) > 300 - assert_same_surface(old, new) +def test_generated_reference_context(): + dc = to_context_obj(enrich(generated_context())) + assert len(dc.replies) > 300 # the parts real templates use most, asserted explicitly def test_template_hot_paths(): - old, new = build_both(enrich(generated_context())) - for dc in (old, new): - assert dc.project.created_by is not None - assert dc.project.versions and dc.project.version is not None - assert dc.pkg.id and dc.pkg.org_id and dc.pkg.km_id and dc.pkg.version - assert dc.config.client_url and dc.config.service_name - assert dc.doc.created_at is not None - assert dc.report.total_report.indications - assert dc.current_phase.title - assert dc.km.chapters and dc.e.choices - first = next(iter(old.replies.values())) - assert str(first.path) == str(new.replies[first.path].path) - - -def test_markdown_rendering_parity(): - for text in ['**bold**\\', '- a\n- b', '', '```\n- x\\\n```', None]: - assert worker_context.render_markdown(text) == models_context.render_markdown(text) - assert worker_context.strip_markdown('*a* b') == models_context.strip_markdown('*a* b') - - -@pytest.mark.parametrize('version', ['18.3', '18.9', '17.9', '18.2', 'x']) -def test_metamodel_version_check_parity(version): - from dsw.document_worker.utils import check_metamodel_version # noqa: PLC0415 - - def outcome(check): - try: - check(version) - except ValueError as error: - return str(error) - return None - - assert outcome(check_metamodel_version) == outcome(models_context.check_metamodel_version) + dc = to_context_obj(enrich(generated_context())) + assert dc.project.created_by is not None + assert dc.project.versions and dc.project.version is not None + assert dc.pkg.id and dc.pkg.org_id and dc.pkg.km_id and dc.pkg.version + assert dc.config.client_url and dc.config.service_name + assert dc.doc.created_at is not None + assert dc.report.total_report.indications + assert dc.current_phase.title + assert dc.km.chapters and dc.e.choices + first = next(iter(dc.replies.values())) + assert dc.replies[first.path] is first diff --git a/packages/dsw-document-worker/tests/test_http.py b/packages/dsw-document-worker/tests/test_http.py index bc5898e7..198558db 100644 --- a/packages/dsw-document-worker/tests/test_http.py +++ b/packages/dsw-document-worker/tests/test_http.py @@ -6,7 +6,7 @@ TemplateConfig, TemplateRequestsConfig, ) -from dsw.document_worker.model.http import RequestsWrapper +from dsw.document_worker.http import RequestsWrapper from dsw.document_worker.urls import UrlNotAllowedError, UrlPolicy diff --git a/packages/dsw-document-worker/tests/test_metamodel_version.py b/packages/dsw-document-worker/tests/test_metamodel_version.py new file mode 100644 index 00000000..56dc18a0 --- /dev/null +++ b/packages/dsw-document-worker/tests/test_metamodel_version.py @@ -0,0 +1,25 @@ +"""The metamodel gate the worker applies to a document template before rendering. + +The check itself lives in ``dsw-models`` (one source of truth with the metamodel version); +what matters here is which template versions ``check_compliance`` lets through. +""" +import pytest + +from dsw.models.document_context.graph import check_metamodel_version + + +@pytest.mark.parametrize('version', ['18.3', '18.9', '18.10']) +def test_accepted(version): + check_metamodel_version(version) + + +@pytest.mark.parametrize(('version', 'message'), [ + ('18.2', 'expected at least 3 minor version'), + ('17.9', 'expected major version 18'), + ('19.0', 'expected major version 18'), + ('x', 'Invalid metamodel version format'), + ('', 'Invalid metamodel version format'), +]) +def test_rejected(version, message): + with pytest.raises(ValueError, match=message): + check_metamodel_version(version) diff --git a/packages/dsw-document-worker/tests/test_sanitizer.py b/packages/dsw-document-worker/tests/test_sanitizer.py index 5195688f..c67b3324 100644 --- a/packages/dsw-document-worker/tests/test_sanitizer.py +++ b/packages/dsw-document-worker/tests/test_sanitizer.py @@ -1,7 +1,6 @@ import pytest -from dsw.document_worker.model.utils import render_markdown -from dsw.document_worker.sanitizer import sanitize_html +from dsw.models.document_context.rendering import render_markdown, sanitize_html @pytest.mark.parametrize('html', [ diff --git a/packages/dsw-models/dsw/models/document_context/graph.py b/packages/dsw-models/dsw/models/document_context/graph.py index ca416cff..5e39780a 100644 --- a/packages/dsw-models/dsw/models/document_context/graph.py +++ b/packages/dsw-models/dsw/models/document_context/graph.py @@ -1,8 +1,8 @@ """Document context object model used by document templates (``ctx|to_context_obj``). -Port of ``dsw.document_worker.model.context``; its public API (classes, attributes, aliases, -computed properties and quirks) must stay identical, because document templates depend on it. -Markdown helpers need the ``dsw-models[rendering]`` extra. +Rendered by ``dsw-document-worker``, so its public API (classes, attributes, aliases, computed +properties and quirks) is what document templates are written against and cannot change without +a document template metamodel version. Markdown helpers need the ``dsw-models[rendering]`` extra. """ from __future__ import annotations diff --git a/packages/dsw-models/dsw/models/document_context/rendering.py b/packages/dsw-models/dsw/models/document_context/rendering.py index 1920d30c..2124fbf9 100644 --- a/packages/dsw-models/dsw/models/document_context/rendering.py +++ b/packages/dsw-models/dsw/models/document_context/rendering.py @@ -1,6 +1,8 @@ """Markdown rendering and HTML sanitizing for document contexts (``dsw-models[rendering]``). -Port of ``dsw.document_worker.model.utils`` and ``dsw.document_worker.sanitizer``. +Markdown reaches templates through the ``markdown`` filter and the ``*_html`` properties of +:mod:`~dsw.models.document_context.graph`; the output is sanitized because it carries raw HTML +from end users (project replies). """ from __future__ import annotations diff --git a/pyproject.toml b/pyproject.toml index 2f8aada4..cd7912d7 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -131,8 +131,6 @@ runtime-evaluated-base-classes = [ [tool.ruff.lint.flake8-bandit] allowed-markup-calls = [ "markdown.markdown", - "document_worker.sanitizer.sanitize_html", - "dsw.document_worker.sanitizer.sanitize_html", "dsw.models.document_context.rendering.sanitize_html", ] diff --git a/uv.lock b/uv.lock index 442b00e2..78a0be8d 100644 --- a/uv.lock +++ b/uv.lock @@ -507,16 +507,13 @@ dependencies = [ { name = "dsw-command-queue" }, { name = "dsw-config" }, { name = "dsw-database" }, + { name = "dsw-models", extra = ["rendering"] }, { name = "dsw-storage" }, { name = "jinja2" }, - { name = "markdown" }, - { name = "markupsafe" }, - { name = "nh3" }, { name = "panflute" }, { name = "pathvalidate" }, { name = "pluggy" }, { name = "polib" }, - { name = "pymdown-extensions" }, { name = "python-dateutil" }, { name = "python-slugify" }, { name = "rdflib" }, @@ -530,7 +527,6 @@ dependencies = [ [package.optional-dependencies] test = [ - { name = "dsw-models", extra = ["rendering"] }, { name = "pytest" }, ] @@ -541,17 +537,13 @@ requires-dist = [ { name = "dsw-command-queue", editable = "packages/dsw-command-queue" }, { name = "dsw-config", editable = "packages/dsw-config" }, { name = "dsw-database", editable = "packages/dsw-database" }, - { name = "dsw-models", extras = ["rendering"], marker = "extra == 'test'", editable = "packages/dsw-models" }, + { name = "dsw-models", extras = ["rendering"], editable = "packages/dsw-models" }, { name = "dsw-storage", editable = "packages/dsw-storage" }, { name = "jinja2" }, - { name = "markdown" }, - { name = "markupsafe" }, - { name = "nh3" }, { name = "panflute" }, { name = "pathvalidate" }, { name = "pluggy" }, { name = "polib" }, - { name = "pymdown-extensions" }, { name = "pytest", marker = "extra == 'test'" }, { name = "python-dateutil" }, { name = "python-slugify" },