diff --git a/.cspell/dictionary.txt b/.cspell/dictionary.txt index b53c1366..16ed9421 100644 --- a/.cspell/dictionary.txt +++ b/.cspell/dictionary.txt @@ -5,6 +5,8 @@ appconfigdata Stubber autouse caplog +capfd +keepends delenv SIGABRT pgconn @@ -25,6 +27,8 @@ pagebreak pagebreaks pagebreakpy panflute +Pango +OOXML idstring privkey searchpath diff --git a/.github/workflows/pipeline.yml b/.github/workflows/pipeline.yml index 8217e934..b197bbb1 100644 --- a/.github/workflows/pipeline.yml +++ b/.github/workflows/pipeline.yml @@ -134,6 +134,12 @@ jobs: run: | uv run --no-sync bash scripts/test-package.sh dsw-tdk + - name: "Test: templating" + id: test-templating + continue-on-error: true + run: | + uv run --no-sync bash scripts/test-package.sh dsw-templating + - name: "Evaluate tests" if: always() shell: bash @@ -150,8 +156,9 @@ jobs: dsw-models=${{ steps.test-models.outcome }} dsw-storage=${{ steps.test-storage.outcome }} dsw-tdk=${{ steps.test-tdk.outcome }} + dsw-templating=${{ steps.test-templating.outcome }} run: | - EXPECTED=9 + EXPECTED=10 SEEN=0 FAILED="" while IFS='=' read -r pkg outcome; do @@ -175,12 +182,13 @@ jobs: fi echo "## All $EXPECTED packages passed" - # dsw-tdk and its only workspace dependency dsw-models declare requires-python - # >=3.12 (unlike the rest of the workspace, which is >=3.14) because the TDK is - # a CLI that end users pip install. Both are installed standalone from this - # checkout here rather than via `uv sync`, which is bound to the root >=3.14 - # floor. Python 3.14 is already covered across all three OSes by the - # test-worker job above. + # dsw-tdk, its only workspace dependency dsw-models and dsw-templating (the + # rendering library the TDK is meant to use) declare requires-python >=3.12 + # (unlike the rest of the workspace, which is >=3.14) because the TDK is a CLI + # that end users pip install. They are installed standalone from this checkout + # here rather than via `uv sync`, which is bound to the root >=3.14 floor. + # Python 3.14 is already covered across all three OSes by the test-worker job + # above. test-tdk-oldest: name: Test TDK # Avoid double runs: in-repo branches are covered by the push event. @@ -216,8 +224,8 @@ jobs: - name: Install dsw-tdk with test dependencies run: | - # dsw-models from this checkout, not PyPI: dsw-tdk pins the same version - uv pip install './packages/dsw-models[test]' './packages/dsw-tdk[test]' + # dsw-models and dsw-templating from this checkout, not PyPI: dsw-tdk pins the same version + uv pip install './packages/dsw-models[test]' './packages/dsw-templating[all,test]' './packages/dsw-tdk[test]' .venv/bin/python --version - name: "Verify: TDK" @@ -229,6 +237,11 @@ jobs: run: | ../../.venv/bin/pytest -s tests + - name: "Test: Templating" + working-directory: packages/dsw-templating + run: | + ../../.venv/bin/pytest -s tests + - name: "Test: TDK" working-directory: packages/dsw-tdk run: | @@ -255,6 +268,7 @@ jobs: - dsw-models - dsw-storage - dsw-tdk + - dsw-templating python-version: - "3.14" diff --git a/.github/workflows/release-package.yml b/.github/workflows/release-package.yml index 6b833085..d2be1e22 100644 --- a/.github/workflows/release-package.yml +++ b/.github/workflows/release-package.yml @@ -108,6 +108,7 @@ jobs: - dsw-models - dsw-storage - dsw-tdk + - dsw-templating steps: - name: Check out repository diff --git a/README.md b/README.md index 745c0e50..6de6cb2c 100644 --- a/README.md +++ b/README.md @@ -19,8 +19,10 @@ In this monorepo, we manage the following Python packages (each has its own subd * [Config (dsw-config)](packages/dsw-config) * [Database (dsw-database)](packages/dsw-database) * [Storage (dsw-storage)](packages/dsw-storage) +* [Templating (dsw-templating)](packages/dsw-templating) -Libraries are currently kept compatible with Python 3.14 and higher. +Libraries are currently kept compatible with Python 3.14 and higher, except for +`dsw-templating` that renders document templates also for the TDK (Python 3.12 and higher). ### Utilities diff --git a/packages/dsw-data-seeder/Dockerfile b/packages/dsw-data-seeder/Dockerfile index de856ca9..4779d014 100644 --- a/packages/dsw-data-seeder/Dockerfile +++ b/packages/dsw-data-seeder/Dockerfile @@ -67,6 +67,7 @@ COPY packages/dsw-mailer/pyproject.toml /app/packages/dsw-mailer/ COPY packages/dsw-models/pyproject.toml /app/packages/dsw-models/ COPY packages/dsw-storage/pyproject.toml /app/packages/dsw-storage/ COPY packages/dsw-tdk/pyproject.toml /app/packages/dsw-tdk/ +COPY packages/dsw-templating/pyproject.toml /app/packages/dsw-templating/ # project.dependencies is dynamic, so `uv export` must build each member's # metadata rather than read it, and hatchling validates project.readme while @@ -80,6 +81,7 @@ COPY packages/dsw-mailer/README.md /app/packages/dsw-mailer/ COPY packages/dsw-models/README.md /app/packages/dsw-models/ COPY packages/dsw-storage/README.md /app/packages/dsw-storage/ COPY packages/dsw-tdk/README.md /app/packages/dsw-tdk/ +COPY packages/dsw-templating/README.md /app/packages/dsw-templating/ RUN uv export --locked --no-dev --no-emit-workspace --no-hashes --package dsw-data-seeder -o /app/requirements.txt \ && python -m pip wheel --no-cache-dir --wheel-dir=/app/wheels -r /app/requirements.txt diff --git a/packages/dsw-document-worker/CHANGELOG.md b/packages/dsw-document-worker/CHANGELOG.md index 14e819f5..7445f628 100644 --- a/packages/dsw-document-worker/CHANGELOG.md +++ b/packages/dsw-document-worker/CHANGELOG.md @@ -12,7 +12,7 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - Document template locales: a `.po` file attached to a document template version is applied at render time, so language is a parameter of document generation instead of a property of the format - Generation of the POT file with translatable strings (new `generatePotFile` command function), stored in S3 and flagged by `document_template.pot_file_ready` - `document.language` and `document.locale` in the document context -- Translations are available to all steps via `Step.before_render` and the `gettext` / `ngettext` / `pgettext` helpers (see [Translations](./support/Translations.md)) +- Translations are available to all steps via `Step.before_render` and the `gettext` / `ngettext` / `pgettext` helpers (see [Translations](../dsw-templating/support/Translations.md)) - Lambda handler reads its configuration from AWS AppConfig when `AWS_APP_CONFIG` is set (see `dsw-config`) - `docx-landscape.lua` Pandoc filter (a `landscape` div, or `\landscape` / `\portrait` paragraphs, switch DOCX page orientation) @@ -22,6 +22,9 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - The `jinja2.ext.i18n` extension is always enabled for Jinja-powered steps - `extras.project` (and the deprecated `extras.questionnaire`) provide `knowledge_model_package_uuid` instead of `knowledge_package_uuid` - The template-facing document context object model (`ctx|to_context_obj`) comes from `dsw-models` (`dsw.models.document_context.graph`) instead of a copy inside the worker; the object model, its attributes and its Markdown rendering are unchanged +- The rendering engine (formats, steps, Jinja filters and tests, conversions, POT extraction) moved to the new `dsw-templating` package, together with the Pandoc filters, the `pandoc-docx-pagebreakpy` filter (now a command of `dsw-templating[docx]` instead of a separate addon) and the template development documentation; rendering is unchanged +- Plugins are loaded from the `dsw_templating_plugins` entry point and implement the hooks of `dsw.templating.plugins` (pluggy project `dsw-templating`) instead of `dsw_document_worker_plugins` / `dsw.document_worker.plugins` +- The Pandoc filters are shipped inside `dsw-templating`; the images no longer copy them to `/pandoc/filters`, which (or `PANDOC_FILTERS`) is still searched first for filters added by a deployment ### Fixed diff --git a/packages/dsw-document-worker/Dockerfile b/packages/dsw-document-worker/Dockerfile index e97bd4ef..78a3b157 100644 --- a/packages/dsw-document-worker/Dockerfile +++ b/packages/dsw-document-worker/Dockerfile @@ -30,7 +30,7 @@ RUN python -m pip wheel --no-deps --wheel-dir=/app/wheels \ /app/packages/dsw-database \ /app/packages/dsw-models \ /app/packages/dsw-storage \ - /app/packages/dsw-document-worker/addons/* \ + /app/packages/dsw-templating \ /app/packages/dsw-document-worker @@ -71,6 +71,7 @@ COPY packages/dsw-mailer/pyproject.toml /app/packages/dsw-mailer/ COPY packages/dsw-models/pyproject.toml /app/packages/dsw-models/ COPY packages/dsw-storage/pyproject.toml /app/packages/dsw-storage/ COPY packages/dsw-tdk/pyproject.toml /app/packages/dsw-tdk/ +COPY packages/dsw-templating/pyproject.toml /app/packages/dsw-templating/ # project.dependencies is dynamic, so `uv export` must build each member's # metadata rather than read it, and hatchling validates project.readme while @@ -84,6 +85,7 @@ COPY packages/dsw-mailer/README.md /app/packages/dsw-mailer/ COPY packages/dsw-models/README.md /app/packages/dsw-models/ COPY packages/dsw-storage/README.md /app/packages/dsw-storage/ COPY packages/dsw-tdk/README.md /app/packages/dsw-tdk/ +COPY packages/dsw-templating/README.md /app/packages/dsw-templating/ RUN uv export --locked --no-dev --no-emit-workspace --no-hashes --package dsw-document-worker -o /app/requirements.txt \ && python -m pip wheel --wheel-dir=/app/wheels -r /app/requirements.txt @@ -99,10 +101,6 @@ ENV APPLICATION_CONFIG_PATH=/app/config/application.yml \ COPY packages/dsw-document-worker/resources/fonts /usr/share/fonts/truetype/custom RUN fc-cache -# Add Pandoc filters -COPY packages/dsw-document-worker/resources/pandoc/filters /pandoc/filters -RUN fc-cache - # Use non-root user USER user diff --git a/packages/dsw-document-worker/README.md b/packages/dsw-document-worker/README.md index 5b6a3ef5..17de298e 100644 --- a/packages/dsw-document-worker/README.md +++ b/packages/dsw-document-worker/README.md @@ -21,12 +21,13 @@ For more information, see [deployment example](https://github.com/ds-wizard/dsw- For general information, please visit our [User Guide](https://guide.ds-wizard.org). -DSW Document Worker technical documentation for template development: +Documents are rendered by [dsw-templating](../dsw-templating), which also holds the technical +documentation for template development: -* [Document Context](./support/DocumentContext.md) -* [Jinja Filters](./support/JinjaFilters.md) -* [Jinja Tests](./support/JinjaTests.md) -* [Translations](./support/Translations.md) +* [Document Context](../dsw-templating/support/DocumentContext.md) +* [Jinja Filters](../dsw-templating/support/JinjaFilters.md) +* [Jinja Tests](../dsw-templating/support/JinjaTests.md) +* [Translations](../dsw-templating/support/Translations.md) ## Docker diff --git a/packages/dsw-document-worker/addons/pandoc-docx-pagebreak-py/LICENSE b/packages/dsw-document-worker/addons/pandoc-docx-pagebreak-py/LICENSE deleted file mode 100644 index c3c2dff7..00000000 --- a/packages/dsw-document-worker/addons/pandoc-docx-pagebreak-py/LICENSE +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) 2018 pandocker - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. diff --git a/packages/dsw-document-worker/addons/pandoc-docx-pagebreak-py/docx_pagebreak.py b/packages/dsw-document-worker/addons/pandoc-docx-pagebreak-py/docx_pagebreak.py deleted file mode 100644 index 8e1fb3db..00000000 --- a/packages/dsw-document-worker/addons/pandoc-docx-pagebreak-py/docx_pagebreak.py +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env python3 -# -*- coding: utf-8 -*- - -""" pandoc-docx-pagebreakpy -Pandoc filter to insert pagebreak as openxml RawBlock -Only for docx output - -- https://github.com/UoA-eResearch/pandoc-docx-pagebreak-py -""" -import panflute as pf - - -class DocxPagebreak(object): - - @staticmethod - def _make_pagebreak(): - return pf.RawBlock('', format='openxml') - - @staticmethod - def _make_toc(instr: str): - toc_lines = [ - r'', - r'', - r'', - r'', - ] - - if instr == r"\toc1": - toc_lines.append(r'TOC \o "1-1" \h \z \u') - elif instr == r"\toc2": - toc_lines.append(r'TOC \o "1-2" \h \z \u') - elif instr == r"\toc4": - toc_lines.append(r'TOC \o "1-4" \h \z \u') - elif instr == r"\toc5": - toc_lines.append(r'TOC \o "1-5" \h \z \u') - elif instr == r"\toc6": - toc_lines.append(r'TOC \o "1-6" \h \z \u') - else: - toc_lines.append(r'TOC \o "1-3" \h \z \u') - - toc_lines.append(r'') - toc_lines.append(r'') - toc_lines.append(r'') - toc_lines.append(r'') - toc_lines.append(r'') - return pf.RawBlock('\n'.join(toc_lines), format='openxml') - - def action(self, elem, doc): - if doc.format != 'docx': - return elem - if isinstance(elem, (pf.Para, pf.Plain)): - for child in elem.content: - if isinstance(child, pf.Str) and child.text == r'\newpage': - elem = self._make_pagebreak() - elif isinstance(child, pf.Str) and child.text.startswith(r'\toc'): - elem = self._make_toc(child.text) - if isinstance(elem, pf.RawBlock): - if elem.text == r'\newpage': - elem = self._make_pagebreak() - elif elem.text.startswith(r'\toc'): - elem = self._make_toc(elem.text) - return elem - - -def main(doc=None): - dp = DocxPagebreak() - return pf.run_filter(dp.action, doc=doc) - - -if __name__ == '__main__': - main() diff --git a/packages/dsw-document-worker/addons/pandoc-docx-pagebreak-py/setup.py b/packages/dsw-document-worker/addons/pandoc-docx-pagebreak-py/setup.py deleted file mode 100644 index c24bf82b..00000000 --- a/packages/dsw-document-worker/addons/pandoc-docx-pagebreak-py/setup.py +++ /dev/null @@ -1,30 +0,0 @@ -from setuptools import setup -from os import path - -here = path.abspath(path.dirname(__file__)) - -setup( - name="pandoc-docx-pagebreak", - description="Pandoc filter for docx output to insert pagebreak at will", # Required - url="https://github.com/pandocker/pandoc-docx-pagebreak-py", # Original repository - author="Kazuki Yamamoto, pandocker", - author_email="k.yamamoto.08136891@gmail.com", - classifiers=[ - "Development Status :: 3 - Alpha", - "Intended Audience :: Developers", - "Topic :: Software Development :: Build Tools", - "License :: OSI Approved :: MIT License", - "Programming Language :: Python :: 3", - "Programming Language :: Python :: 3.11", - ], - keywords="pandoc filter docx", - py_modules=["docx_pagebreak"], - install_requires=[ - "panflute==2.3.1", - ], - entry_points={ - "console_scripts": [ - "pandoc-docx-pagebreakpy=docx_pagebreak:main", - ], - }, -) diff --git a/packages/dsw-document-worker/dsw/document_worker/cli.py b/packages/dsw-document-worker/dsw/document_worker/cli.py index a6aee985..a4c406b7 100644 --- a/packages/dsw-document-worker/dsw/document_worker/cli.py +++ b/packages/dsw-document-worker/dsw/document_worker/cli.py @@ -72,8 +72,8 @@ def run(config: DocumentWorkerConfig, workdir: str): @main.command() def list_plugins(): - from .plugins.manager import create_manager + from dsw.templating.plugins import create_manager pm = create_manager() - for plugin in pm.list_name_plugin(): - click.echo(f'{plugin[0]}: {plugin[1].__name__}') + for name, module in pm.list_name_plugin(): + click.echo(f'{name}: {getattr(module, "__name__", module)}') diff --git a/packages/dsw-document-worker/dsw/document_worker/consts.py b/packages/dsw-document-worker/dsw/document_worker/consts.py index ebed2a4c..19d5650b 100644 --- a/packages/dsw-document-worker/dsw/document_worker/consts.py +++ b/packages/dsw-document-worker/dsw/document_worker/consts.py @@ -8,22 +8,10 @@ CMD_FUNCTION_GENERATE_POT_FILE = 'generatePotFile' COMPONENT_NAME = 'Document Worker' DEFAULT_ENCODING = 'utf-8' -EXIT_SUCCESS = 0 NULL_UUID = '00000000-0000-0000-0000-000000000000' PACKAGE_NAME = 'dsw-document-worker' -PLUGINS_ENTRYPOINT = 'dsw_document_worker_plugins' PROG_NAME = 'docworker' -JINJA_EXTENSIONS = ('jinja2.ext.do', 'jinja2.ext.loopcontrols') -JINJA_FILE_EXTENSIONS = ('.j2', '.jinja', '.jinja2', '.jnj') - -# Rendering and POT extraction must agree on this: the msgid of a {% trans %} -# block depends on it, and the POT file is per document template while a step -# option would be per format. -JINJA_I18N_TRIMMED = True - -DEFAULT_LANGUAGE = 'en' -DEFAULT_LOCALE_DOMAIN = 'default' LOCALE_PO_FILE_NAME = 'translation.po' LOCALE_MO_FILE_NAME = 'translation.mo' LOCALE_STAMP_FILE_NAME = 'updated_at' @@ -46,23 +34,6 @@ class DocumentState: FINISHED = 'DoneDocumentState' -class TemplateAssetField: - UUID = 'uuid' - FILENAME = 'fileName' - CONTENT_TYPE = 'contentType' - - -class FormatField: - UUID = 'uuid' - NAME = 'name' - STEPS = 'steps' - - -class StepField: - NAME = 'name' - OPTIONS = 'options' - - class DocumentNamingStrategy: UUID = 'uuid' SANITIZE = 'sanitize' diff --git a/packages/dsw-document-worker/dsw/document_worker/context.py b/packages/dsw-document-worker/dsw/document_worker/context.py index ab51c3e3..6d4ab5b5 100644 --- a/packages/dsw-document-worker/dsw/document_worker/context.py +++ b/packages/dsw-document-worker/dsw/document_worker/context.py @@ -66,7 +66,7 @@ def get(cls) -> _Context: @classmethod def initialize(cls, db, s3, config, workdir): - from .plugins.manager import create_manager + from dsw.templating.plugins import create_manager cls._instance = _Context( app=AppContext( pm=create_manager(), diff --git a/packages/dsw-document-worker/dsw/document_worker/documents.py b/packages/dsw-document-worker/dsw/document_worker/documents.py index d5a0db38..93985b1a 100644 --- a/packages/dsw-document-worker/dsw/document_worker/documents.py +++ b/packages/dsw-document-worker/dsw/document_worker/documents.py @@ -1,6 +1,5 @@ from __future__ import annotations -import pathlib import typing import pathvalidate @@ -12,148 +11,7 @@ if typing.TYPE_CHECKING: from dsw.database.model import DBDocument - - -class FileFormat: - - def __init__(self, name: str, content_type: str, file_extension: str): - self.name = name - self.content_type = content_type - self.file_extension = file_extension - - def __eq__(self, other): - return isinstance(other, FileFormat) and other.name == self.name - - def __hash__(self): - return hash(self.name) - - def __str__(self): - return self.name - - def __repr__(self): - return f'Format[{self.name}]' - - -class FileFormats: - JSON = FileFormat('json', 'application/json', 'json') - HTML = FileFormat('html', 'text/html', 'html') - PDF = FileFormat('pdf', 'application/pdf', 'pdf') - DOCX = FileFormat( - 'docx', - 'application/vnd.openxmlformats-officedocument.wordprocessingml.document', - 'docx', - ) - Markdown = FileFormat('markdown', 'text/markdown', 'md') - ODT = FileFormat('odt', 'application/vnd.oasis.opendocument.text', 'odt') - RST = FileFormat('rst', 'text/x-rst', 'rst') - LaTeX = FileFormat('latex', 'application/x-tex', 'tex') - EPUB = FileFormat('epub', 'application/epub+zip', 'epub') - DocBook4 = FileFormat('docbook4', 'application/docbook+xml', 'dbk') - DocBook5 = FileFormat('docbook5', 'application/docbook+xml', 'dbk') - PPTX = FileFormat( - 'pptx', - 'application/vnd.openxmlformats-officedocument.presentationml.presentation', - 'pptx', - ) - RTF = FileFormat('rtf', 'application/rtf', 'rtf') - ADoc = FileFormat('asciidoc', 'text/asciidoc', 'adoc') - RDF_XML = FileFormat('rdf', 'application/rdf+xml', 'rdf') - N3 = FileFormat('n3', 'text/n3', 'n3') - NTRIPLES = FileFormat('nt', 'application/n-triples', 'nt') - TURTLE = FileFormat('ttl', 'text/turtle', 'ttl') - TRIG = FileFormat('trig', 'application/trig', 'trig') - JSONLD = FileFormat('jsonld', 'application/ld+json', 'jsonld') - ZIP = FileFormat('zip', 'application/zip', 'zip') - TAR = FileFormat('tar', 'application/x-tar', 'tar') - TAR_GZIP = FileFormat('gzip', 'application/gzip', 'tar.gz') - TAR_BZIP2 = FileFormat('bzip2', 'application/x-bzip2', 'tar.bz2') - TAR_LZMA = FileFormat('lzma', 'application/x-lzma', 'tar.xz') - XLSX = FileFormat( - 'xlsx', - 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet', - 'xlsx', - ) - XLSM = FileFormat( - 'xlsm', - 'application/vnd.ms-excel.sheet.macroEnabled.12', - 'xlsm', - ) - - @staticmethod - def get(name: str): - known_formats = { - 'html': FileFormats.HTML, - 'pdf': FileFormats.PDF, - 'docx': FileFormats.DOCX, - 'markdown': FileFormats.Markdown, - 'odt': FileFormats.ODT, - 'rst': FileFormats.RST, - 'latex': FileFormats.LaTeX, - 'json': FileFormats.JSON, - 'epub': FileFormats.EPUB, - 'docbook4': FileFormats.DocBook4, - 'docbook5': FileFormats.DocBook5, - 'pptx': FileFormats.PPTX, - 'rtf': FileFormats.RTF, - 'asciidoc': FileFormats.ADoc, - 'rdf': FileFormats.RDF_XML, - 'rdf/xml': FileFormats.RDF_XML, - 'turtle': FileFormats.TURTLE, - 'ttl': FileFormats.TURTLE, - 'n3': FileFormats.N3, - 'ntriples': FileFormats.NTRIPLES, - 'n-triples': FileFormats.NTRIPLES, - 'trig': FileFormats.TRIG, - 'json-ld': FileFormats.JSONLD, - 'jsonld': FileFormats.JSONLD, - 'zip': FileFormats.ZIP, - 'tar': FileFormats.TAR, - 'gzip': FileFormats.TAR_GZIP, - 'bzip2': FileFormats.TAR_BZIP2, - 'lzma': FileFormats.TAR_LZMA, - 'xlsx': FileFormats.XLSX, - 'xlsm': FileFormats.XLSM, - } - return known_formats.get(name) - - -class DocumentFile: - - def __init__(self, file_format: FileFormat, content: bytes, - encoding: str | None = None): - self.file_format = file_format - self._content = content - self.byte_size = len(content) - self.encoding = encoding - - @property - def content_type(self) -> str: - return self.file_format.content_type - - @property - def safe_encoding(self) -> str: - return self.encoding or consts.DEFAULT_ENCODING - - @property - def content(self) -> bytes: - return self._content - - @content.setter - def content(self, content: bytes): - self._content = content - self.byte_size = len(content) - - def filename(self, name: str) -> str: - return f'{name}.{self.file_format.file_extension}' - - def store(self, name: str): - pathlib.Path(self.filename(name)).write_bytes(self.content) - - @property - def object_content_type(self) -> str: - if self.encoding is not None: - return f'{self.content_type}; charset={self.encoding}' - return self.content_type + from dsw.templating import DocumentFile def _name_uuid(document: DBDocument) -> str: diff --git a/packages/dsw-document-worker/dsw/document_worker/exceptions.py b/packages/dsw-document-worker/dsw/document_worker/exceptions.py index 1af7d0fc..e86a93fd 100644 --- a/packages/dsw-document-worker/dsw/document_worker/exceptions.py +++ b/packages/dsw-document-worker/dsw/document_worker/exceptions.py @@ -1,5 +1,7 @@ from __future__ import annotations +from dsw.templating import TemplateTriggeredError + class JobError(Exception): @@ -41,6 +43,15 @@ def create_job_error(job_id: str, message: str, document_found=True, exc=None): if isinstance(exc, JobError): return exc + if isinstance(exc, TemplateTriggeredError): + # raised by a template to report a problem to its user, not a system failure + return JobError( + job_id=job_id, + msg=exc.msg, + exc=None, + skip_reporting=True, + ) + return JobError( job_id=job_id, msg=message, diff --git a/packages/dsw-document-worker/dsw/document_worker/limits.py b/packages/dsw-document-worker/dsw/document_worker/limits.py index 918b8c26..9a8122e9 100644 --- a/packages/dsw-document-worker/dsw/document_worker/limits.py +++ b/packages/dsw-document-worker/dsw/document_worker/limits.py @@ -1,8 +1,9 @@ from __future__ import annotations +from dsw.templating.utils import byte_size_format + from .context import Context from .exceptions import JobError -from .utils import byte_size_format class LimitsEnforcer: diff --git a/packages/dsw-document-worker/dsw/document_worker/plugins/__init__.py b/packages/dsw-document-worker/dsw/document_worker/plugins/__init__.py deleted file mode 100644 index 7ca58c1b..00000000 --- a/packages/dsw-document-worker/dsw/document_worker/plugins/__init__.py +++ /dev/null @@ -1,4 +0,0 @@ -from .specs import hookimpl, hookspec - - -__all__ = ['hookimpl', 'hookspec'] diff --git a/packages/dsw-document-worker/dsw/document_worker/plugins/manager.py b/packages/dsw-document-worker/dsw/document_worker/plugins/manager.py deleted file mode 100644 index 99684373..00000000 --- a/packages/dsw-document-worker/dsw/document_worker/plugins/manager.py +++ /dev/null @@ -1,14 +0,0 @@ -from __future__ import annotations - -import pluggy - -from .. import consts - - -def create_manager(): - import dsw.document_worker.plugins.specs as hookspecs - - pm = pluggy.PluginManager(consts.PACKAGE_NAME) - pm.load_setuptools_entrypoints(consts.PLUGINS_ENTRYPOINT) - pm.add_hookspecs(hookspecs) - return pm diff --git a/packages/dsw-document-worker/dsw/document_worker/pot.py b/packages/dsw-document-worker/dsw/document_worker/pot.py index b94feb49..4cfbaa37 100644 --- a/packages/dsw-document-worker/dsw/document_worker/pot.py +++ b/packages/dsw-document-worker/dsw/document_worker/pot.py @@ -1,22 +1,15 @@ from __future__ import annotations import dataclasses -import io import logging import re import typing import uuid -import babel -import jinja2.exceptions -import jinja2.ext -from babel.messages.catalog import Catalog -from babel.messages.extract import DEFAULT_KEYWORDS, extract -from babel.messages.pofile import write_po - from dsw.command_queue import CommandJobError +from dsw.templating import consts as templating_consts +from dsw.templating.pot import extract_catalog, render_pot_file -from . import consts from .context import Context @@ -26,17 +19,7 @@ LOG = logging.getLogger(__name__) -COMMENT_TAGS = ('TRANSLATORS:',) -EXTRACT_METHOD = typing.cast('typing.Any', jinja2.ext.babel_extract) COORDINATE_PATTERN = re.compile(r'^[A-Za-z0-9._-]+$') -NO_WRAP = 0 -EXTRACT_OPTIONS = { - 'encoding': consts.DEFAULT_ENCODING, - 'extensions': ','.join(consts.JINJA_EXTENSIONS), - 'silent': 'false', - 'newstyle_gettext': 'true', - 'trimmed': str(consts.JINJA_I18N_TRIMMED).lower(), -} @dataclasses.dataclass(frozen=True) @@ -89,75 +72,10 @@ def load(command: PersistentCommand) -> PotFileRequest: organization_id=coordinates['organizationId'], template_id=coordinates['templateId'], version=coordinates['version'], - language=str(body.get('language') or consts.DEFAULT_LANGUAGE), + language=str(body.get('language') or templating_consts.DEFAULT_LANGUAGE), ) -@dataclasses.dataclass -class ExtractionResult: - catalog: Catalog - failed_files: list[str] - - -def extract_messages(content: str) -> list[tuple]: - return list(extract( - method=EXTRACT_METHOD, - fileobj=io.BytesIO(content.encode(consts.DEFAULT_ENCODING)), - keywords=DEFAULT_KEYWORDS, - comment_tags=COMMENT_TAGS, - options=EXTRACT_OPTIONS, - )) - - -def make_catalog(*, project: str, version: str, language: str) -> Catalog: - locale: babel.Locale | None = None - try: - locale = babel.Locale.parse(language.replace('-', '_')) - except (ValueError, babel.UnknownLocaleError): - LOG.warning('Cannot parse language "%s" - POT file without locale info', language) - return Catalog( - locale=locale, - domain=consts.DEFAULT_LOCALE_DOMAIN, - project=project, - version=version, - charset=consts.DEFAULT_ENCODING, - fuzzy=False, - ) - - -def extract_catalog(files: list[DBDocumentTemplateFile], *, project: str, - version: str, language: str) -> ExtractionResult: - catalog = make_catalog(project=project, version=version, language=language) - failed_files = [] - for file in sorted(files, key=lambda f: f.file_name): - try: - messages = extract_messages(file.content) - except jinja2.exceptions.TemplateSyntaxError as e: - LOG.warning('Skipping file "%s" that cannot be parsed: %s', file.file_name, str(e)) - failed_files.append(file.file_name) - continue - for lineno, message, comments, context in messages: - catalog.add( - message, - None, - [(file.file_name, lineno)], - auto_comments=comments, - context=context, - ) - return ExtractionResult(catalog=catalog, failed_files=failed_files) - - -def render_pot_file(result: ExtractionResult) -> bytes: - lines = [f'# Translations template for {result.catalog.project}.'] - if result.failed_files: - lines.append(f'# Skipped files that could not be parsed: ' - f'{", ".join(result.failed_files)}') - result.catalog.header_comment = '\n'.join(lines) + '\n' - buffer = io.BytesIO() - write_po(buffer, result.catalog, width=NO_WRAP, omit_header=False, sort_output=True) - return buffer.getvalue() - - class PotFileJob: def __init__(self, command: PersistentCommand): @@ -197,7 +115,7 @@ def _fetch_template_files(self) -> list[DBDocumentTemplateFile]: tenant_uuid=self.rq.tenant_uuid, ) return [f for f in db_files - if f.file_name.endswith(consts.JINJA_FILE_EXTENSIONS)] + if f.file_name.endswith(templating_consts.JINJA_FILE_EXTENSIONS)] def _store_pot_file(self, data: bytes): try: diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/__init__.py b/packages/dsw-document-worker/dsw/document_worker/templates/__init__.py index d08d7af4..f5792f2e 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/__init__.py +++ b/packages/dsw-document-worker/dsw/document_worker/templates/__init__.py @@ -1,4 +1,5 @@ -from .formats import Format +from dsw.templating import Format + from .templates import Template, TemplateRegistry diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/locales.py b/packages/dsw-document-worker/dsw/document_worker/templates/locales.py index e2dfe784..94beae37 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/locales.py +++ b/packages/dsw-document-worker/dsw/document_worker/templates/locales.py @@ -1,10 +1,8 @@ from __future__ import annotations -import dataclasses import gettext import logging import typing -import uuid import polib @@ -15,46 +13,10 @@ if typing.TYPE_CHECKING: from pathlib import Path + from dsw.templating import TemplateLocale -LOG = logging.getLogger(__name__) - - -@dataclasses.dataclass(frozen=True) -class TemplateLocale: - uuid: str - name: str - code: str - updated_at: str - - @staticmethod - def load(data: dict | None) -> TemplateLocale | None: - if not isinstance(data, dict): - return None - try: - locale_uuid = str(uuid.UUID(str(data['uuid']))) - except (KeyError, ValueError): - LOG.warning('Ignoring locale without a valid UUID') - return None - return TemplateLocale( - uuid=locale_uuid, - name=str(data.get('name', '')), - code=str(data.get('code', '')), - updated_at=str(data.get('updatedAt', '')), - ) - -@dataclasses.dataclass -class RenderContext: - translations: gettext.NullTranslations - language: str | None = None - locale: TemplateLocale | None = None - - @staticmethod - def null(language: str | None = None) -> RenderContext: - return RenderContext( - translations=gettext.NullTranslations(), - language=language, - ) +LOG = logging.getLogger(__name__) class LocaleLoader: diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/templates.py b/packages/dsw-document-worker/dsw/document_worker/templates/templates.py index d683feb2..9606301c 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/templates.py +++ b/packages/dsw-document-worker/dsw/document_worker/templates/templates.py @@ -1,71 +1,105 @@ from __future__ import annotations -import base64 import dataclasses import datetime import logging +import os +import pathlib import shutil import typing +from dsw.templating import ( + PandocSettings, + RenderContext, + RenderSettings, + RequestsSettings, + SecuritySettings, + TemplateSettings, + register_plugin_steps, +) +from dsw.templating import Template as TemplateEngine +from dsw.templating.settings import bundled_pandoc_filters + from .. import consts from ..context import Context -from .formats import Format -from .locales import LocaleLoader, RenderContext, TemplateLocale -from .steps.base import Step, register_step +from .locales import LocaleLoader if typing.TYPE_CHECKING: - from pathlib import Path - from dsw.database.model import ( DBDocumentTemplate, DBDocumentTemplateAsset, DBDocumentTemplateFile, ) - from dsw.models.document_context.graph import ProjectFile + from dsw.templating import DocumentFile, Format, TemplateLocale - from ..documents import DocumentFile + from ..config import DocumentWorkerConfig LOG = logging.getLogger(__name__) -class TemplateError(Exception): - - def __init__(self, template_uuid: str, message: str): - self.template_uuid = template_uuid - self.message = message - - def __str__(self): - return f'Error in template "{self.template_uuid}"\n' \ - f'- {self.message}' - - -class Asset: - - def __init__(self, *, uuid: str, name: str, content_type: str, - data: bytes, path: Path): - self.uuid = uuid - self.name = name - self.content_type = content_type - self.data = data - self.path = path +def render_settings(cfg: DocumentWorkerConfig, coordinates: str) -> RenderSettings: + """Map the worker configuration onto what the rendering of a template needs.""" + template_cfg = cfg.templates.get_config(coordinates) + template_settings = None + if template_cfg is not None: + template_settings = TemplateSettings( + secrets=template_cfg.secrets, + requests=RequestsSettings( + enabled=template_cfg.requests.enabled, + limit=template_cfg.requests.limit, + timeout=template_cfg.requests.timeout, + ), + ) + return RenderSettings( + security=SecuritySettings( + allow_external_resources=cfg.security.allow_external_resources, + allow_private_network=cfg.security.allow_private_network, + allowed_hosts=cfg.security.allowed_hosts, + allowed_paths=cfg.security.allowed_paths, + max_redirects=cfg.security.max_redirects, + ), + pandoc=PandocSettings( + command=cfg.pandoc.command, + timeout=cfg.pandoc.timeout, + # deployments may add their own filters and templates there + filter_dirs=[ + pathlib.Path(os.getenv('PANDOC_FILTERS', '/pandoc/filters')), + bundled_pandoc_filters(), + ], + templates_dir=pathlib.Path(os.getenv('PANDOC_TEMPLATES', '/pandoc/templates')), + ), + template=template_settings, + ) - @property - def is_image(self) -> bool: - return self.content_type.startswith('image/') - @property - def data_base64(self) -> str: - return base64.b64encode(self.data).decode('ascii') +class S3ProjectFiles: + """Project files of one project, downloaded from S3 into the template directory.""" - @property - def data_url(self) -> str: - return f'data:{self.content_type};base64,{self.data_base64}' + def __init__(self, *, tenant_uuid: str, project_uuid: str | None, cache_dir: pathlib.Path): + self.tenant_uuid = tenant_uuid + self.project_uuid = project_uuid + self.cache_dir = cache_dir - @property - def src_value(self): - return self.data_url + def resolve(self, file_uuid: str, name: str, + content_type: str) -> pathlib.Path | None: + if self.project_uuid is None: + LOG.warning('Project UUID is not set, cannot fetch project file') + return None + file_path = self.cache_dir / file_uuid + if not file_path.parent.exists(): + file_path.parent.mkdir(parents=True, exist_ok=True) + if not file_path.exists(): + result = Context.get().app.s3.download_project_file( + tenant_uuid=self.tenant_uuid, + project_uuid=self.project_uuid, + file_uuid=file_uuid, + target_path=file_path, + ) + if not result: + return None + return file_path @dataclasses.dataclass @@ -76,9 +110,11 @@ class TemplateComposite: class Template: + """Local copy of a document template from the database, rendered by dsw-templating.""" - def __init__(self, tenant_uuid: str, template_dir: Path, + def __init__(self, tenant_uuid: str, template_dir: pathlib.Path, db_template: TemplateComposite): + ctx = Context.get() self.tenant_uuid = tenant_uuid self.template_dir = template_dir self.last_used = datetime.datetime.now(tz=datetime.UTC) @@ -86,84 +122,24 @@ def __init__(self, tenant_uuid: str, template_dir: Path, self.template_uuid = self.db_template.template.uuid self.coordinates = self.db_template.template.coordinates - self.formats: dict[str, Format] = {} - self.project_uuid: str | None = None + self.engine = TemplateEngine( + template_dir=template_dir, + formats=self.db_template.template.formats, + assets=self.db_template.assets.values(), + template_uuid=self.template_uuid, + coordinates=self.coordinates, + settings=render_settings(ctx.app.cfg, self.coordinates), + plugins=ctx.app.pm, + ) self.render_ctx = RenderContext.null() self._locale_loader = LocaleLoader( cache_dir=template_dir / consts.LOCALES_CACHE_DIR, tenant_uuid=tenant_uuid, ) - def raise_exc(self, message: str): - raise TemplateError(self.template_uuid, message) - - def fetch_asset(self, file_name: str) -> Asset | None: - LOG.info('Fetching asset "%s"', file_name) - file_path = self.template_dir / file_name - asset = None - for a in self.db_template.assets.values(): - if a.file_name == file_name: - asset = a - break - if asset is None or not file_path.exists(): - LOG.error('Asset "%s" not found', file_name) - return None - return Asset( - uuid=asset.uuid, - name=file_name, - content_type=asset.content_type, - data=file_path.read_bytes(), - path=file_path, - ) - - def fetch_project_file(self, file: ProjectFile) -> Asset | None: - return self._fetch_project_file( - file_uuid=file.uuid, - name=file.name, - content_type=file.content_type, - ) - - def fetch_project_file_dict(self, file: dict) -> Asset | None: - file_uuid = file.get('uuid') - name = file.get('fileName') - content_type = file.get('contentType') - if isinstance(file_uuid, str) and isinstance(name, str) and isinstance(content_type, str): - return self._fetch_project_file( - file_uuid=file_uuid, - name=name, - content_type=content_type, - ) - return None - - def _fetch_project_file(self, file_uuid: str, name: str, - content_type: str) -> Asset | None: - LOG.info('Fetching project file "%s"', file_uuid) - if self.project_uuid is None: - LOG.warning('Project UUID is not set, cannot fetch project file') - return None - file_path = self.template_dir / 'project-files' / file_uuid - if not file_path.parent.exists(): - file_path.parent.mkdir(parents=True, exist_ok=True) - if not file_path.exists(): - result = Context.get().app.s3.download_project_file( - tenant_uuid=self.tenant_uuid, - project_uuid=self.project_uuid, - file_uuid=file_uuid, - target_path=file_path, - ) - if not result: - LOG.error('Project file "%s" cannot be retrieved', file_uuid) - return None - return Asset( - uuid=file_uuid, - name=name, - content_type=content_type, - data=file_path.read_bytes(), - path=file_path, - ) - - def asset_path(self, filename: str) -> str: - return str(self.template_dir / filename) + @property + def formats(self) -> dict[str, Format]: + return self.engine.formats def _store_asset(self, asset: DBDocumentTemplateAsset): LOG.debug('Storing asset %s (%s)', asset.uuid, asset.file_name) @@ -267,9 +243,11 @@ def update_template_assets(self, db_assets: dict[str, DBDocumentTemplateAsset]): for asset_uuid in to_chk: self._update_asset(db_assets[asset_uuid]) self.db_template.assets = db_assets + self.engine.assets = list(db_assets.values()) def update_template(self, db_template: TemplateComposite): self.db_template.template = db_template.template + self.engine.formats_metadata = db_template.template.formats if not self.template_dir.exists(): self.template_dir.mkdir() self.update_template_files(db_template.files) @@ -288,31 +266,28 @@ def prepare_locale(self, *, language: str | None, locale: TemplateLocale | None) locale=locale, ) - def prepare_format(self, format_uuid: str): - for format_meta in self.db_template.template.formats: - if format_uuid == format_meta.get(consts.FormatField.UUID): - self.formats[format_uuid] = Format(self, format_meta) - return True - return False + def prepare_format(self, format_uuid: str) -> bool: + return self.engine.prepare_format(format_uuid) def has_format(self, format_uuid: str) -> bool: - return any( - f[consts.FormatField.UUID] == format_uuid - for f in self.db_template.template.formats - ) + return self.engine.has_format(format_uuid) def __getitem__(self, format_uuid: str) -> Format: - return self.formats[format_uuid] + return self.engine[format_uuid] def render(self, format_uuid: str, project_uuid: str | None, context: dict) -> DocumentFile: - Context.get().app.pm.hook.enrich_document_context(context=context) - self.last_used = datetime.datetime.now(tz=datetime.UTC) - self.project_uuid = project_uuid - result = self[format_uuid].execute(context) - self.project_uuid = None - return result + return self.engine.render( + format_uuid, + context, + render_ctx=self.render_ctx, + project_files=S3ProjectFiles( + tenant_uuid=self.tenant_uuid, + project_uuid=project_uuid, + cache_dir=self.template_dir / 'project-files', + ), + ) class TemplateRegistry: @@ -327,14 +302,7 @@ def get(cls) -> TemplateRegistry: def __init__(self): self._templates: dict[str, dict[str, Template]] = {} - self._load_plugin_steps() - - def _load_plugin_steps(self): - for steps_dict in Context.get().app.pm.hook.provide_steps(): - for name, step_class in steps_dict.items(): - if not issubclass(step_class, Step): - raise RuntimeError(f'Provided class "{step_class}" is not a subclass of Step') - register_step(name, step_class) + register_plugin_steps(Context.get().app.pm) def has_template(self, tenant_uuid: str, template_uuid: str) -> bool: return tenant_uuid in self._templates and \ diff --git a/packages/dsw-document-worker/dsw/document_worker/worker.py b/packages/dsw-document-worker/dsw/document_worker/worker.py index db9ed471..fd7bb710 100644 --- a/packages/dsw-document-worker/dsw/document_worker/worker.py +++ b/packages/dsw-document-worker/dsw/document_worker/worker.py @@ -12,17 +12,17 @@ from dsw.database.database import Database from dsw.models.document_context.graph import check_metamodel_version from dsw.storage import S3Storage +from dsw.templating import ContextDefaults, DocumentFile, TemplateLocale, enrich_context_config +from dsw.templating.utils import byte_size_format from . import consts from .build_info import BUILD_INFO from .context import Context -from .documents import DocumentFile, DocumentNameGiver +from .documents import DocumentNameGiver from .exceptions import DocumentNotFoundError, JobError, create_job_error from .limits import LimitsEnforcer from .pot import PotFileJob from .templates import Format, Template, TemplateRegistry -from .templates.locales import TemplateLocale -from .utils import byte_size_format if typing.TYPE_CHECKING: @@ -180,32 +180,18 @@ def prepare_template(self): ) def _enrich_context_config(self): - old = self.doc_context.get('config', {}) - - client_url = old.get('clientUrl', '').rstrip('/') - app_title = (old.get('appTitle', None) or - self.ctx.app.cfg.context.default_app_title) - app_title_short = (old.get('appTitleShort', None) or - self.ctx.app.cfg.context.default_app_title_short) - primary_color = (old.get('primaryColor', None) or - self.ctx.app.cfg.context.default_primary_color) - illustrations_color = (old.get('illustrationsColor', None) or - self.ctx.app.cfg.context.default_illustrations_color) - logo_url_template = (old.get('logoUrl', None) or - self.ctx.app.cfg.context.default_logo_url) - logo_url = logo_url_template.replace('{{clientUrl}}', client_url) - - self.doc_context['config'].update({ - 'serviceName': self.ctx.app.cfg.context.service_name, - 'serviceNameShort': self.ctx.app.cfg.context.service_name_short, - 'serviceUrl': self.ctx.app.cfg.context.service_url, - 'serviceDomainName': self.ctx.app.cfg.context.service_domain_name, - 'appTitle': app_title, - 'appTitleShort': app_title_short, - 'primaryColor': primary_color, - 'illustrationsColor': illustrations_color, - 'logoUrl': logo_url, - }) + cfg = self.ctx.app.cfg.context + enrich_context_config(self.doc_context, ContextDefaults( + service_name=cfg.service_name, + service_name_short=cfg.service_name_short, + service_url=cfg.service_url, + service_domain_name=cfg.service_domain_name, + default_primary_color=cfg.default_primary_color, + default_illustrations_color=cfg.default_illustrations_color, + default_logo_url=cfg.default_logo_url, + default_app_title=cfg.default_app_title, + default_app_title_short=cfg.default_app_title_short, + )) def _enrich_context(self): extras: dict[str, typing.Any] = {} diff --git a/packages/dsw-document-worker/lambda.Dockerfile b/packages/dsw-document-worker/lambda.Dockerfile index f7c68961..29154ea0 100644 --- a/packages/dsw-document-worker/lambda.Dockerfile +++ b/packages/dsw-document-worker/lambda.Dockerfile @@ -30,7 +30,7 @@ RUN python -m pip wheel --no-deps --wheel-dir=/app/wheels \ /app/packages/dsw-database \ /app/packages/dsw-models \ /app/packages/dsw-storage \ - /app/packages/dsw-document-worker/addons/* \ + /app/packages/dsw-templating \ /app/packages/dsw-document-worker @@ -64,6 +64,7 @@ COPY packages/dsw-mailer/pyproject.toml /app/packages/dsw-mailer/ COPY packages/dsw-models/pyproject.toml /app/packages/dsw-models/ COPY packages/dsw-storage/pyproject.toml /app/packages/dsw-storage/ COPY packages/dsw-tdk/pyproject.toml /app/packages/dsw-tdk/ +COPY packages/dsw-templating/pyproject.toml /app/packages/dsw-templating/ # project.dependencies is dynamic, so `uv export` must build each member's # metadata rather than read it, and hatchling validates project.readme while @@ -77,6 +78,7 @@ COPY packages/dsw-mailer/README.md /app/packages/dsw-mailer/ COPY packages/dsw-models/README.md /app/packages/dsw-models/ COPY packages/dsw-storage/README.md /app/packages/dsw-storage/ COPY packages/dsw-tdk/README.md /app/packages/dsw-tdk/ +COPY packages/dsw-templating/README.md /app/packages/dsw-templating/ # Install Python dependencies (resolved from uv.lock) RUN uv --directory /app export --locked --no-dev --no-emit-workspace --no-hashes --package dsw-document-worker -o /app/requirements.txt \ @@ -95,9 +97,6 @@ ENV APPLICATION_CONFIG_PATH=${LAMBDA_TASK_ROOT}/application.yml \ COPY packages/dsw-document-worker/resources/fonts /usr/share/fonts/truetype/custom RUN fc-cache -## Add Pandoc filters -COPY packages/dsw-document-worker/resources/pandoc/filters /pandoc/filters - WORKDIR ${LAMBDA_TASK_ROOT} # Prepare dirs diff --git a/packages/dsw-document-worker/pyproject.toml b/packages/dsw-document-worker/pyproject.toml index c324b002..6e9cd52c 100644 --- a/packages/dsw-document-worker/pyproject.toml +++ b/packages/dsw-document-worker/pyproject.toml @@ -43,7 +43,7 @@ artifacts = ["dsw/*/build_info.py"] artifacts = ["dsw/*/build_info.py"] [tool.hatch.metadata.hooks.uv-dynamic-versioning] -dependencies = ["Babel", "click", "Jinja2", "panflute", "pathvalidate", "pluggy", "polib", "python-dateutil", "python-slugify", "rdflib", "rdflib-jsonld", "requests", "sentry-sdk", "tenacity", "weasyprint", "XlsxWriter", "dsw-command-queue=={{ version }}", "dsw-config=={{ version }}", "dsw-database=={{ version }}", "dsw-models[rendering]=={{ version }}", "dsw-storage=={{ version }}"] +dependencies = ["click", "pathvalidate", "polib", "python-dateutil", "python-slugify", "sentry-sdk", "tenacity", "dsw-command-queue=={{ version }}", "dsw-config=={{ version }}", "dsw-database=={{ version }}", "dsw-models=={{ version }}", "dsw-storage=={{ version }}", "dsw-templating[all]=={{ version }}"] optional-dependencies = { test = ["pytest"] } [tool.uv-dynamic-versioning] diff --git a/packages/dsw-document-worker/tests/test_locales.py b/packages/dsw-document-worker/tests/test_locales.py index 994b308f..371ea5e2 100644 --- a/packages/dsw-document-worker/tests/test_locales.py +++ b/packages/dsw-document-worker/tests/test_locales.py @@ -4,7 +4,8 @@ import pytest from dsw.document_worker import consts -from dsw.document_worker.templates.locales import LocaleLoader, TemplateLocale +from dsw.document_worker.templates.locales import LocaleLoader +from dsw.templating import TemplateLocale LOCALE_UUID = '44444444-4444-4444-4444-444444444444' diff --git a/packages/dsw-document-worker/tests/test_pot.py b/packages/dsw-document-worker/tests/test_pot.py index cae1c389..ba7ececd 100644 --- a/packages/dsw-document-worker/tests/test_pot.py +++ b/packages/dsw-document-worker/tests/test_pot.py @@ -1,20 +1,9 @@ -import dataclasses import types import pytest from dsw.command_queue import CommandJobError -from dsw.document_worker.pot import ( - PotFileRequest, - extract_catalog, - render_pot_file, -) - - -@dataclasses.dataclass -class FakeTemplateFile: - file_name: str - content: str +from dsw.document_worker.pot import PotFileRequest def make_command(**body): @@ -26,92 +15,6 @@ def make_command(**body): ) -def make_pot(*files, language='en'): - result = extract_catalog( - [FakeTemplateFile(name, content) for name, content in files], - project='org:tid:1.0.0', - version='1.0.0', - language=language, - ) - return result, render_pot_file(result).decode('utf-8') - - -def test_extract_trans_block(): - _, pot = make_pot(('src/a.j2', '{% trans %}Hello{% endtrans %}')) - assert 'msgid "Hello"' in pot - assert '#: src/a.j2:1' in pot - - -def test_extract_plural(): - _, pot = make_pot(( - 'src/a.j2', - '{% trans count %}{{ count }} item{% pluralize %}{{ count }} items{% endtrans %}', - )) - assert 'msgid "%(count)s item"' in pot - assert 'msgid_plural "%(count)s items"' in pot - assert 'msgstr[0] ""' in pot - assert 'msgstr[1] ""' in pot - - -def test_extract_underscore_and_pgettext(): - _, pot = make_pot(('src/a.j2', "{{ _('World') }}\n{{ pgettext('menu', 'Open') }}")) - assert 'msgid "World"' in pot - assert 'msgctxt "menu"' in pot - assert 'msgid "Open"' in pot - - -def test_extract_translators_comment(): - _, pot = make_pot(( - 'src/a.j2', - '{# TRANSLATORS: shown on top #}\n{% trans %}Hello{% endtrans %}', - )) - assert '#. shown on top' in pot - - -def test_extract_survives_do_extension(): - _, pot = make_pot(('src/a.j2', "{% do [] %}{{ _('Alpha') }}")) - assert 'msgid "Alpha"' in pot - - -def test_extract_survives_loopcontrols_extension(): - _, pot = make_pot(( - 'src/a.j2', - "{% for i in [1] %}{% break %}{% endfor %}{{ _('Alpha') }}", - )) - assert 'msgid "Alpha"' in pot - - -def test_broken_file_is_isolated(): - result, pot = make_pot( - ('src/broken.j2', '{% if %}'), - ('src/ok.j2', "{{ _('Alpha') }}"), - ) - assert result.failed_files == ['src/broken.j2'] - assert 'msgid "Alpha"' in pot - assert 'Skipped files that could not be parsed: src/broken.j2' in pot - - -def test_header_fields(): - _, pot = make_pot(('src/a.j2', "{{ _('Alpha') }}"), language='cs') - assert 'Project-Id-Version: org:tid:1.0.0 1.0.0' in pot - assert 'Language: cs' in pot - assert 'Plural-Forms: nplurals=' in pot - assert 'charset=utf-8' in pot - assert '#, fuzzy' not in pot - - -def test_header_without_known_language(): - _, pot = make_pot(('src/a.j2', "{{ _('Alpha') }}"), language='not a language') - assert 'msgid "Alpha"' in pot - assert 'Language:' not in pot - assert 'Plural-Forms:' not in pot - - -def test_messages_are_sorted(): - _, pot = make_pot(('src/a.j2', "{{ _('Beta') }}{{ _('Alpha') }}")) - assert pot.index('msgid "Alpha"') < pot.index('msgid "Beta"') - - def test_request_load_ok(): rq = PotFileRequest.load(make_command( documentTemplateUuid='33333333-3333-3333-3333-333333333333', diff --git a/packages/dsw-document-worker/tests/test_templating.py b/packages/dsw-document-worker/tests/test_templating.py new file mode 100644 index 00000000..3d263c95 --- /dev/null +++ b/packages/dsw-document-worker/tests/test_templating.py @@ -0,0 +1,185 @@ +"""How the worker maps its configuration and errors onto dsw-templating.""" +import pathlib +import types + +import pytest + +from dsw.document_worker.config import ( + CommandConfig, + SecurityConfig, + TemplateConfig, + TemplateRequestsConfig, + TemplatesConfig, +) +from dsw.document_worker.exceptions import JobError, create_job_error +from dsw.document_worker.templates.templates import render_settings +from dsw.templating import FormatStepError, TemplateTriggeredError +from dsw.templating.conversions import Pandoc +from dsw.templating.settings import bundled_pandoc_filters + + +def make_config(*templates: TemplateConfig): + return types.SimpleNamespace( + templates=TemplatesConfig(templates=list(templates)), + security=SecurityConfig( + allow_external_resources=False, + allow_private_network=True, + allowed_hosts=['example.org'], + allowed_paths=['/data'], + max_redirects=5, + ), + pandoc=CommandConfig(executable='pandoc', args='--standalone --wrap=none', timeout=30), + ) + + +def test_render_settings_from_config(): + template_cfg = TemplateConfig( + ids=['org:tid'], + requests=TemplateRequestsConfig(enabled=True, limit=7, timeout=2), + secrets={'token': 'x'}, + send_sentry=False, + ) + settings = render_settings(make_config(template_cfg), 'org:tid:1.0.0') + assert settings.security.allow_external_resources is False + assert settings.security.allow_private_network is True + assert settings.security.allowed_hosts == ['example.org'] + assert settings.security.allowed_paths == ['/data'] + assert settings.security.max_redirects == 5 + assert settings.pandoc.command == ['pandoc', '--standalone', '--wrap=none'] + assert settings.pandoc.timeout == 30 + assert settings.template is not None + assert settings.template.secrets == {'token': 'x'} + assert settings.template.requests.enabled is True + assert settings.template.requests.limit == 7 + assert settings.template.requests.timeout == 2 + + +def test_render_settings_without_template_config(): + assert render_settings(make_config(), 'org:tid:1.0.0').template is None + + +def test_lua_filters_resolve_from_the_package(monkeypatch, tmp_path: pathlib.Path): + monkeypatch.setenv('PANDOC_FILTERS', str(tmp_path / 'missing')) + pandoc_settings = render_settings(make_config(), 'org:tid:1.0.0').pandoc + for name in ('docx-landscape.lua', 'docx-pagebreak.lua', 'docx-toc.lua'): + path = pandoc_settings.filter_path(name) + assert path == bundled_pandoc_filters() / name + assert path.is_file() + args = Pandoc(pandoc_settings, ['docx-pagebreak.lua'], None)._extra_args() + assert args == ['--lua-filter', str(bundled_pandoc_filters() / 'docx-pagebreak.lua')] + + +def test_deployment_filters_take_precedence(monkeypatch, tmp_path: pathlib.Path): + (tmp_path / 'docx-toc.lua').write_text('-- custom', encoding='utf-8') + monkeypatch.setenv('PANDOC_FILTERS', str(tmp_path)) + monkeypatch.setenv('PANDOC_TEMPLATES', str(tmp_path)) + pandoc_settings = render_settings(make_config(), 'org:tid:1.0.0').pandoc + assert pandoc_settings.filter_path('docx-toc.lua') == tmp_path / 'docx-toc.lua' + assert pandoc_settings.templates_dir == tmp_path + + +def test_template_triggered_error_is_a_job_error_not_reported(): + exc = TemplateTriggeredError(title='Oops', message='Bad input') + job_error = create_job_error(job_id='doc', message='Failed to build final document', exc=exc) + assert isinstance(job_error, JobError) + assert job_error.skip_reporting + assert job_error.db_message() == 'Oops\n\nBad input' + + +@pytest.mark.parametrize('exc', [RuntimeError('boom'), None]) +def test_other_errors_are_wrapped(exc): + job_error = create_job_error(job_id='doc', message='Failed', exc=exc) + assert not job_error.skip_reporting + assert job_error.exc is exc + + +class FakeStorage: + + def __init__(self): + self.project_files: list[str] = [] + + def download_template_asset(self, *, tenant_uuid, template_uuid, file_name, target_path): + target_path.write_bytes(b'asset') + return True + + def download_project_file(self, *, tenant_uuid, project_uuid, file_uuid, target_path): + self.project_files.append(f'{project_uuid}/{file_uuid}') + target_path.write_bytes(b'project file') + return True + + +@pytest.fixture +def worker_context(tmp_path: pathlib.Path): + from dsw.document_worker.context import Context + + original = Context._instance + s3 = FakeStorage() + Context.initialize(db=None, s3=s3, config=make_config(), workdir=tmp_path) + yield s3 + Context._instance = original + + +def test_worker_template_renders_from_db_composite(worker_context, tmp_path: pathlib.Path): + from dsw.document_worker.templates.templates import Template, TemplateComposite + + root = types.SimpleNamespace( + uuid='f1', file_name='root.j2', updated_at=1, + content="{{ assets('logo.png').data.decode() }}|" + "{{ assets({'uuid': 'p1', 'fileName': 'a.txt', 'contentType': 'text/plain'})" + '.data.decode() }}|{{ ctx.name }}', + ) + asset = types.SimpleNamespace(uuid='a1', file_name='logo.png', content_type='image/png', + updated_at=1) + db_template = types.SimpleNamespace( + uuid='t1', + coordinates='org:tid:1.0.0', + formats=[{'uuid': 'f', 'name': 'HTML', 'steps': [ + {'name': 'jinja', 'options': {'template': 'root.j2'}}, + ]}], + ) + template = Template( + tenant_uuid='tenant', + template_dir=tmp_path / 'tenant' / 't1', + db_template=TemplateComposite(template=db_template, files={'f1': root}, + assets={'a1': asset}), + ) + template.prepare_fs() + assert template.prepare_format('f') + template.prepare_locale(language='en', locale=None) + document = template.render(format_uuid='f', project_uuid='p', context={'name': 'N'}) + assert document.content == b'asset|project file|N' + assert worker_context.project_files == ['p/p1'] + # a document without a project cannot reach project files + with pytest.raises(FormatStepError, match="'None' has no attribute 'data'"): + template.render(format_uuid='f', project_uuid=None, context={'name': 'N'}) + + +def test_context_config_uses_worker_configuration(): + from dsw.document_worker.config import DocumentContextConfig + from dsw.document_worker.worker import Job + + job = types.SimpleNamespace( + doc_context={'config': {'clientUrl': 'https://fw.example.org', 'appTitle': 'Mine'}}, + ctx=types.SimpleNamespace(app=types.SimpleNamespace(cfg=types.SimpleNamespace( + context=DocumentContextConfig( + service_name='FAIR Wizard', service_name_short='FW', + service_url='https://fair-wizard.com', service_domain_name='fair-wizard.com', + default_primary_color='#111111', default_illustrations_color='#222222', + default_logo_url='{{clientUrl}}/logo.png', default_app_title='FW', + default_app_title_short='FWs', + ), + ))), + ) + Job._enrich_context_config(job) + assert job.doc_context['config'] == { + 'clientUrl': 'https://fw.example.org', + 'serviceName': 'FAIR Wizard', + 'serviceNameShort': 'FW', + 'serviceUrl': 'https://fair-wizard.com', + 'serviceDomainName': 'fair-wizard.com', + 'appTitle': 'Mine', + 'appTitleShort': 'FWs', + 'primaryColor': '#111111', + 'illustrationsColor': '#222222', + 'logoUrl': 'https://fw.example.org/logo.png', + } diff --git a/packages/dsw-mailer/Dockerfile b/packages/dsw-mailer/Dockerfile index 5e10d181..f5f39d27 100644 --- a/packages/dsw-mailer/Dockerfile +++ b/packages/dsw-mailer/Dockerfile @@ -64,6 +64,7 @@ COPY packages/dsw-mailer/pyproject.toml /app/packages/dsw-mailer/ COPY packages/dsw-models/pyproject.toml /app/packages/dsw-models/ COPY packages/dsw-storage/pyproject.toml /app/packages/dsw-storage/ COPY packages/dsw-tdk/pyproject.toml /app/packages/dsw-tdk/ +COPY packages/dsw-templating/pyproject.toml /app/packages/dsw-templating/ # project.dependencies is dynamic, so `uv export` must build each member's # metadata rather than read it, and hatchling validates project.readme while @@ -77,6 +78,7 @@ COPY packages/dsw-mailer/README.md /app/packages/dsw-mailer/ COPY packages/dsw-models/README.md /app/packages/dsw-models/ COPY packages/dsw-storage/README.md /app/packages/dsw-storage/ COPY packages/dsw-tdk/README.md /app/packages/dsw-tdk/ +COPY packages/dsw-templating/README.md /app/packages/dsw-templating/ RUN uv export --locked --no-dev --no-emit-workspace --no-hashes --package dsw-mailer -o /app/requirements.txt \ && python -m pip wheel --no-cache-dir --wheel-dir=/app/wheels -r /app/requirements.txt diff --git a/packages/dsw-tdk/CHANGELOG.md b/packages/dsw-tdk/CHANGELOG.md index 5f29901f..0426cf0d 100644 --- a/packages/dsw-tdk/CHANGELOG.md +++ b/packages/dsw-tdk/CHANGELOG.md @@ -10,12 +10,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added - New `pot` command creating a POT file with translatable strings of the template project +- New `render` command rendering a document from the local template project and a document context (JSON file), optionally with a PO file and project files, without any DSW instance; rendering is done by `dsw-templating`, the same engine as the document worker uses, and only the files selected by `_tdk.files` are available to it, as on the server; the context gets the document worker's defaults (`config` service information and branding fallbacks, missing `extras`), which can be overridden with `-D NAME=VALUE` or the worker's `DOCUMENT_CONTEXT_*` environment variables; `--secret` / `--allow-requests` provide the `secrets` / `requests` globals and `--pandoc-filters` / `--pandoc-templates` the directories of the worker's Pandoc filters and templates +- `all` extra (`pip install 'dsw-tdk[all]'`) with the optional dependencies of rendering steps (PDF, Excel, RDF, HTTP) - `language` field in `template.json` (prompted by `dsw-tdk new`, defaults to `en`) ### Changed - Update to DT metamodel 18.3 - Metamodel version handling is shared with `dsw-models` (now a dependency) +- The `pot` command extracts messages with `dsw-templating` (now a dependency), the same code as the document worker, instead of a copy of it - The `template.json` written into a package is built and validated as the shared `DocumentTemplateBundle` from `dsw-models`, so the package shape has one definition; a template that does not match it is reported as a warning and still packaged, as before ### Fixed diff --git a/packages/dsw-tdk/Dockerfile b/packages/dsw-tdk/Dockerfile index b0d3bad0..682c2cc3 100644 --- a/packages/dsw-tdk/Dockerfile +++ b/packages/dsw-tdk/Dockerfile @@ -26,6 +26,7 @@ COPY packages /app/packages # build environment, but pip itself starts once. RUN python -m pip wheel --no-cache-dir --no-deps --wheel-dir=/app/wheels \ /app/packages/dsw-models \ + /app/packages/dsw-templating \ /app/packages/dsw-tdk @@ -61,6 +62,7 @@ COPY packages/dsw-mailer/pyproject.toml /app/packages/dsw-mailer/ COPY packages/dsw-models/pyproject.toml /app/packages/dsw-models/ COPY packages/dsw-storage/pyproject.toml /app/packages/dsw-storage/ COPY packages/dsw-tdk/pyproject.toml /app/packages/dsw-tdk/ +COPY packages/dsw-templating/pyproject.toml /app/packages/dsw-templating/ # project.dependencies is dynamic, so `uv export` must build each member's # metadata rather than read it, and hatchling validates project.readme while @@ -74,6 +76,7 @@ COPY packages/dsw-mailer/README.md /app/packages/dsw-mailer/ COPY packages/dsw-models/README.md /app/packages/dsw-models/ COPY packages/dsw-storage/README.md /app/packages/dsw-storage/ COPY packages/dsw-tdk/README.md /app/packages/dsw-tdk/ +COPY packages/dsw-templating/README.md /app/packages/dsw-templating/ RUN uv export --locked --no-dev --no-emit-workspace --no-hashes --package dsw-tdk -o /app/requirements.txt \ && python -m pip wheel --no-cache-dir --wheel-dir=/app/wheels -r /app/requirements.txt diff --git a/packages/dsw-tdk/README.md b/packages/dsw-tdk/README.md index 77d401b0..78dfc3c2 100644 --- a/packages/dsw-tdk/README.md +++ b/packages/dsw-tdk/README.md @@ -60,6 +60,45 @@ For further information, visit our [documentation](https://docs.ds-wizard.org). - `verify` = check the metadata of local template project - `package` = create a distribution ZIP package that is importable to DSW via web interface - `pot` = create a POT file with translatable strings of the local template project +- `render` = render a document from the local template project and a document context (no DSW instance needed) + +### Rendering documents locally + +`render` uses the same engine as the DSW document worker ([dsw-templating](../dsw-templating)). You need a document context as a JSON file: + +```shell script +$ dsw-tdk render --context context.json --format "HTML Document" --output document.html +$ dsw-tdk render -c context.json -F "PDF Document" --po cs.po --project-files ./files +``` + +- `--format` accepts the UUID or name of a format (it can be omitted if the template has only one) +- `--po` renders with translations from a PO file (e.g. a translated `dsw-tdk pot` output); the language is taken from `document.language` of the context unless `--language` is given +- `--project-files` is a directory with files uploaded to the project, named by their UUID or file name +- Only files selected by `_tdk.files` are available to the rendering, as on the server +- The context is completed as the document worker does it with its default configuration: `config` gets the service name and URL of the Data Stewardship Wizard and fallbacks for missing branding (app title, colors, logo), and `extras` requested by the format but missing in the context are rendered as for a document without a project (with a warning) +- The worker's defaults can be overridden with `-D NAME=VALUE` (repeatable), named as in the `documentContext` section of the document worker configuration, or with its environment variables (also in `.env`); the option wins over the environment: + +```shell script +$ dsw-tdk render -c context.json -F "HTML Document" -D "serviceName=FAIR Wizard" -D serviceUrl=https://fair-wizard.com +$ echo 'DOCUMENT_CONTEXT_SERVICE_NAME=FAIR Wizard' >> .env +``` + +| Name | Environment variable | Default | +|---|---|---| +| `serviceName` | `DOCUMENT_CONTEXT_SERVICE_NAME` | `Data Stewardship Wizard` | +| `serviceNameShort` | `DOCUMENT_CONTEXT_SERVICE_NAME_SHORT` | `DSW` | +| `serviceUrl` | `DOCUMENT_CONTEXT_SERVICE_URL` | `https://ds-wizard.org` | +| `serviceDomainName` | `DOCUMENT_CONTEXT_SERVICE_DOMAIN_NAME` | `ds-wizard.org` | +| `defaultPrimaryColor` | `DOCUMENT_CONTEXT_DEFAULT_PRIMARY_COLOR` | `#0033aa` | +| `defaultIllustrationsColor` | `DOCUMENT_CONTEXT_DEFAULT_ILLUSTRATIONS_COLOR` | `#0033aa` | +| `defaultLogoUrl` | `DOCUMENT_CONTEXT_DEFAULT_LOGO_URL` | `{{clientUrl}}/assets/logo.svg` | +| `defaultAppTitle` | `DOCUMENT_CONTEXT_DEFAULT_APP_TITLE` | `DS Wizard` | +| `defaultAppTitleShort` | `DOCUMENT_CONTEXT_DEFAULT_APP_TITLE_SHORT` | `DS Wizard` | + +The `service*` values are always set; the `default*` ones apply only where the context has no value. +- Steps `weasyprint`, `excel` and `rdflib-convert`, the `requests` global and the `pandoc-docx-pagebreakpy` Pandoc filter need optional dependencies: `pip install 'dsw-tdk[all]'`; WeasyPrint also needs [Pango](https://doc.courtbouillon.org/weasyprint/stable/first_steps.html#installation) and the `pandoc` step needs [pandoc](https://pandoc.org) installed +- The `docx-*.lua` Pandoc filters of the document worker are always available; `--pandoc-filters DIR` adds more (searched first) and `--pandoc-templates DIR` provides templates for the `template` option of the `pandoc` step (or `PANDOC_FILTERS` / `PANDOC_TEMPLATES`, as for the worker) +- The `secrets` and `requests` globals exist only when configured for the template on the server; locally, `--secret NAME=VALUE` (repeatable) provides `secrets` and `--allow-requests` provides `requests` ### Environment variables diff --git a/packages/dsw-tdk/dsw/tdk/cli.py b/packages/dsw-tdk/dsw/tdk/cli.py index 06f27eda..0b08645f 100644 --- a/packages/dsw-tdk/dsw/tdk/cli.py +++ b/packages/dsw-tdk/dsw/tdk/cli.py @@ -18,11 +18,14 @@ from .api_client import WizardCommunicationError from .config import CONFIG from .core import TDKCore, TDKProcessingError +from .render import context_defaults, render_settings from .utils import FormatSpec, TemplateBuilder, create_dot_env, safe_utf8 from .validation import ValidationError if typing.TYPE_CHECKING: + from dsw.templating import ContextDefaults + from .model import Template @@ -586,6 +589,121 @@ def create_pot_file(ctx, template_dir, output, force: bool): ClickPrinter.success(f'POT file {filename} created') +class _ForwardHandler(logging.Handler): + """Hands records of dsw.templating to the CLI logger, warnings unless debugging. + + Records carrying an exception only accompany the error that is reported + anyway, so they are shown in the debug mode only. + """ + + def __init__(self, logger: logging.Logger): + super().__init__() + self.logger = logger + self.debug = logger.level <= logging.DEBUG + + def emit(self, record: logging.LogRecord): + if self.debug or (record.levelno >= logging.WARNING and record.exc_info is None): + self.logger.log(record.levelno, '%s', record.getMessage()) + + +def _forward_templating_logs(logger: logging.Logger): + templating_logger = logging.getLogger('dsw.templating') + templating_logger.handlers = [_ForwardHandler(logger)] + templating_logger.propagate = False + templating_logger.setLevel(logging.DEBUG) + + +def _parse_pairs(values: tuple[str, ...]) -> dict[str, str]: + pairs: dict[str, str] = {} + for value in values: + name, sep, text = value.partition('=') + if not sep or not name.strip(): + raise click.BadParameter(f'"{value}" is not NAME=VALUE') + pairs[name.strip()] = text + return pairs + + +def _parse_secrets(ctx, param, values: tuple[str, ...]) -> dict[str, str]: + return _parse_pairs(values) + + +def _parse_context_defaults(ctx, param, values: tuple[str, ...]) -> ContextDefaults: + overrides = _parse_pairs(values) + try: + return context_defaults(overrides) + except ValueError as e: + raise click.BadParameter(str(e)) from e + + +@main.command(help='Render a document from the local template and a document context.', + name='render') +@click.argument('TEMPLATE-DIR', type=DIR_TYPE, default=CURRENT_DIR, required=False) +@click.option('-c', '--context', 'context_file', required=True, type=FILE_READ_TYPE, + help='JSON file with the document context.') +@click.option('-F', '--format', 'format_ref', default=None, + help='UUID or name of the format (can be omitted if there is only one).') +@click.option('-o', '--output', default=None, type=click.Path(dir_okay=False, writable=True), + help='Target file [default: .].') +@click.option('-p', '--po', 'po_file', default=None, type=FILE_READ_TYPE, + help='PO file with translations to render with.') +@click.option('-l', '--language', default=None, + help='Language of the document [default: document.language of the context].') +@click.option('--project-files', default=None, + type=click.Path(exists=True, file_okay=False, resolve_path=True), + help='Directory with project files, named by their UUID or file name.') +@click.option('-D', '--context-default', 'context_default', multiple=True, + metavar='NAME=VALUE', callback=_parse_context_defaults, + help='Override a value the document worker adds to ctx.config, named as in its ' + 'documentContext configuration (e.g. serviceName=FAIR Wizard); can be ' + 'repeated. Its DOCUMENT_CONTEXT_* environment variables (and .env) work too.') +@click.option('-s', '--secret', 'secrets', multiple=True, metavar='NAME=VALUE', + callback=_parse_secrets, + help='Value of the secrets global in templates (as configured for the template ' + 'in the document worker); can be repeated.') +@click.option('--allow-requests', is_flag=True, + help='Provide the requests global for HTTP requests from templates ' + '(as enabled for the template in the document worker).') +@click.option('--pandoc-filters', default=None, envvar='PANDOC_FILTERS', + type=click.Path(exists=True, file_okay=False, resolve_path=True), + help='Directory with additional Pandoc filters, searched before the bundled ones.') +@click.option('--pandoc-templates', default=None, envvar='PANDOC_TEMPLATES', + type=click.Path(exists=True, file_okay=False, resolve_path=True), + help='Directory with Pandoc templates (the template option of the pandoc step).') +@click.option('-f', '--force', is_flag=True, help='Overwrite the output file if already exists.') +@click.pass_context +def render_document(ctx, template_dir, context_file, format_ref, output, po_file, language, + project_files, context_default: ContextDefaults, secrets: dict[str, str], + allow_requests: bool, pandoc_filters, pandoc_templates, force: bool): + tdk = TDKCore(logger=ctx.obj.logger) + load_local(tdk, template_dir) + _forward_templating_logs(ctx.obj.logger) + try: + output_path, document = tdk.render( + context_file=pathlib.Path(context_file), + format_ref=format_ref, + output=None if output is None else pathlib.Path(output), + force=force, + po_file=None if po_file is None else pathlib.Path(po_file), + language=language, + project_files_dir=None if project_files is None else pathlib.Path(project_files), + context_defaults=context_default, + settings=render_settings( + pandoc_filters_dir=None if pandoc_filters is None else pathlib.Path(pandoc_filters), + pandoc_templates_dir=(None if pandoc_templates is None + else pathlib.Path(pandoc_templates)), + secrets=secrets, + allow_requests=allow_requests, + ), + ) + except Exception as e: + ClickPrinter.failure('Failed to render the document') + ClickPrinter.error(f'> {e}') + sys.exit(1) + filename = click.style(output_path.as_posix(), bold=True) + size = humanize.naturalsize(document.byte_size) + ClickPrinter.success(f'Document {filename} rendered ({document.content_type}, {size})') + + @main.group(help='Manage shared user configuration (~/.dsw-tdk).', name='config') @click.pass_context def config(ctx): diff --git a/packages/dsw-tdk/dsw/tdk/consts.py b/packages/dsw-tdk/dsw/tdk/consts.py index 6b5019f6..dad1f869 100644 --- a/packages/dsw-tdk/dsw/tdk/consts.py +++ b/packages/dsw-tdk/dsw/tdk/consts.py @@ -32,11 +32,8 @@ DEFAULT_LIST_FORMAT = '{template.id:<50} {template.name:<30} [{template.uuid}]' DEFAULT_ENCODING = 'utf-8' DEFAULT_LANGUAGE = 'en' -DEFAULT_LOCALE_DOMAIN = 'default' DEFAULT_README = pathlib.Path('README.md') -JINJA_EXTENSIONS = ('jinja2.ext.do', 'jinja2.ext.loopcontrols') -JINJA_I18N_TRIMMED = True POT_FILE_DEFAULT = 'template.pot' TEMPLATE_FILE = 'template.json' diff --git a/packages/dsw-tdk/dsw/tdk/core.py b/packages/dsw-tdk/dsw/tdk/core.py index f860e71c..7f3ab653 100644 --- a/packages/dsw-tdk/dsw/tdk/core.py +++ b/packages/dsw-tdk/dsw/tdk/core.py @@ -17,11 +17,20 @@ from dsw.models.errors import MetamodelVersionError from dsw.models.strictness import load as load_model from dsw.models.versions import MetamodelVersion +from dsw.templating import ContextDefaults, RenderContext, RenderSettings from . import consts from .api_client import WizardAPIClient, WizardCommunicationError from .model import Template, TemplateFile, TemplateFileType, TemplateProject from .pot import PotFile, create_pot_file +from .render import ( + DirectoryProjectFiles, + enrich_context, + filtered_native_stderr, + find_format, + load_translations, + render_document, +) from .utils import UUIDGen from .validation import TemplateValidator, ValidationError @@ -29,6 +38,8 @@ if typing.TYPE_CHECKING: from asyncio import Event + from dsw.templating import DocumentFile + ChangeItem = tuple[watchfiles.Change, pathlib.Path] @@ -364,6 +375,49 @@ def create_pot_file(self, output: pathlib.Path, force: bool) -> PotFile: output.write_bytes(pot_file.data) return pot_file + def render(self, *, context_file: pathlib.Path, format_ref: str | None, + output: pathlib.Path | None, force: bool, po_file: pathlib.Path | None = None, + language: str | None = None, + project_files_dir: pathlib.Path | None = None, + context_defaults: ContextDefaults | None = None, + settings: RenderSettings | None = None, + ) -> tuple[pathlib.Path, DocumentFile]: + template = self.safe_project.safe_template + format_spec = find_format(template, format_ref) + self.logger.debug('Loading document context: %s', context_file.as_posix()) + context = json.loads(context_file.read_text(encoding=consts.DEFAULT_ENCODING)) + for name in enrich_context(context, format_spec, context_defaults): + self.logger.warning('Extra "%s" is not in the context, rendering as ' + 'for a document without a project', name) + if language is None: + language = (context.get('document') or {}).get('language') + if po_file is None: + render_ctx = RenderContext.null(language=language) + else: + self.logger.debug('Loading translations: %s', po_file.as_posix()) + render_ctx = RenderContext(translations=load_translations(po_file), language=language) + project_files = None + if project_files_dir is not None: + project_files = DirectoryProjectFiles(project_files_dir) + self.logger.info('Rendering format "%s" of %s', format_spec.name, template.coordinates) + with tempfile.TemporaryDirectory() as workdir, filtered_native_stderr(): + document = render_document( + template, + workdir=pathlib.Path(workdir), + format_uuid=str(format_spec.uuid), + context=context, + render_ctx=render_ctx, + project_files=project_files, + settings=settings, + ) + if output is None: + output = pathlib.Path.cwd() / document.filename(template.template_id or 'document') + if output.exists() and not force: + raise RuntimeError(f'File {output} already exists (not forced)') + self.logger.debug('Writing document: %s', output.as_posix()) + output.write_bytes(document.content) + return output, document + def _package_descriptor(self, descriptor: dict) -> dict: """Normalize the package descriptor through the shared DocumentTemplateBundle. diff --git a/packages/dsw-tdk/dsw/tdk/model.py b/packages/dsw-tdk/dsw/tdk/model.py index 8f1b5e74..9c811358 100644 --- a/packages/dsw-tdk/dsw/tdk/model.py +++ b/packages/dsw-tdk/dsw/tdk/model.py @@ -9,6 +9,8 @@ import pathspec +from dsw.templating.consts import JINJA_FILE_EXTENSIONS + from . import consts @@ -143,7 +145,7 @@ def serialize(self): class TemplateFile: DEFAULT_CONTENT_TYPE = 'application/octet-stream' - TEMPLATE_EXTENSIONS = ('.j2', '.jinja', '.jinja2', '.jnj') + TEMPLATE_EXTENSIONS = JINJA_FILE_EXTENSIONS def __init__(self, *, filename: pathlib.Path, remote_uuid: str | None = None, remote_id: str | None = None, remote_type: TemplateFileType | None = None, diff --git a/packages/dsw-tdk/dsw/tdk/pot.py b/packages/dsw-tdk/dsw/tdk/pot.py index 705c2265..bfb41a7d 100644 --- a/packages/dsw-tdk/dsw/tdk/pot.py +++ b/packages/dsw-tdk/dsw/tdk/pot.py @@ -1,16 +1,9 @@ from __future__ import annotations -import io import typing -import babel -import jinja2.exceptions -import jinja2.ext -from babel.messages.catalog import Catalog -from babel.messages.extract import DEFAULT_KEYWORDS, extract -from babel.messages.pofile import write_po +from dsw.templating.pot import extract_catalog, render_pot_file -from . import consts from .model import TemplateFile @@ -18,72 +11,31 @@ from .model import Template -# Must stay in sync with dsw-document-worker (dsw/document_worker/pot.py), so -# that a POT file created locally matches the one generated by the server. -COMMENT_TAGS = ('TRANSLATORS:',) -EXTRACT_METHOD = typing.cast('typing.Any', jinja2.ext.babel_extract) -NO_WRAP = 0 -EXTRACT_OPTIONS = { - 'encoding': consts.DEFAULT_ENCODING, - 'extensions': ','.join(consts.JINJA_EXTENSIONS), - 'silent': 'false', - 'newstyle_gettext': 'true', - 'trimmed': str(consts.JINJA_I18N_TRIMMED).lower(), -} - - class PotFile(typing.NamedTuple): data: bytes failed_files: list[str] -def _template_files(template: Template) -> list[TemplateFile]: - return sorted( - (f for f in template.files.values() - if f.filename.name.endswith(TemplateFile.TEMPLATE_EXTENSIONS)), - key=lambda f: f.filename.as_posix(), - ) +class _SourceFile(typing.NamedTuple): + file_name: str + content: str -def _make_catalog(template: Template) -> Catalog: - locale: babel.Locale | None = None - try: - locale = babel.Locale.parse(template.language.replace('-', '_')) - except (ValueError, babel.UnknownLocaleError): - locale = None - return Catalog( - locale=locale, - domain=consts.DEFAULT_LOCALE_DOMAIN, - project=template.coordinates, - version=template.version, - charset=consts.DEFAULT_ENCODING, - fuzzy=False, - ) +def _template_files(template: Template) -> list[_SourceFile]: + return [ + _SourceFile(file_name=f.filename.as_posix(), content=f.content.decode('utf-8')) + for f in template.files.values() + if f.filename.name.endswith(TemplateFile.TEMPLATE_EXTENSIONS) + ] def create_pot_file(template: Template) -> PotFile: - catalog = _make_catalog(template) - failed_files = [] - for file in _template_files(template): - filename = file.filename.as_posix() - try: - messages = list(extract( - method=EXTRACT_METHOD, - fileobj=io.BytesIO(file.content), - keywords=DEFAULT_KEYWORDS, - comment_tags=COMMENT_TAGS, - options=EXTRACT_OPTIONS, - )) - except jinja2.exceptions.TemplateSyntaxError: - failed_files.append(filename) - continue - for lineno, message, comments, context in messages: - catalog.add(message, None, [(filename, lineno)], - auto_comments=comments, context=context) - lines = [f'# Translations template for {template.coordinates}.'] - if failed_files: - lines.append(f'# Skipped files that could not be parsed: {", ".join(failed_files)}') - catalog.header_comment = '\n'.join(lines) + '\n' - buffer = io.BytesIO() - write_po(buffer, catalog, width=NO_WRAP, omit_header=False, sort_output=True) - return PotFile(data=buffer.getvalue(), failed_files=failed_files) + # The same extraction as the document worker uses, so a POT file created + # locally matches the one generated by the server. + result = extract_catalog( + _template_files(template), + project=template.coordinates, + version=template.version or '', + language=template.language, + ) + return PotFile(data=render_pot_file(result), failed_files=result.failed_files) diff --git a/packages/dsw-tdk/dsw/tdk/render.py b/packages/dsw-tdk/dsw/tdk/render.py new file mode 100644 index 00000000..3c6011b8 --- /dev/null +++ b/packages/dsw-tdk/dsw/tdk/render.py @@ -0,0 +1,260 @@ +"""Rendering a local template project with dsw-templating, without any DSW instance.""" +from __future__ import annotations + +import contextlib +import dataclasses +import gettext +import io +import os +import re +import sys +import tempfile +import typing +import uuid + +from babel.messages.mofile import write_mo +from babel.messages.pofile import read_po + +from dsw.templating import ( + ContextDefaults, + PandocSettings, + RenderContext, + RenderSettings, + RequestsSettings, + TemplateSettings, + create_manager, + enrich_context_config, + register_plugin_steps, +) +from dsw.templating import Template as TemplateEngine +from dsw.templating.settings import bundled_pandoc_filters + +from .model import TemplateFileType + + +if typing.TYPE_CHECKING: + import pathlib + + from dsw.templating import DocumentFile + + from .model import Format, Template + + +class LocalAsset(typing.NamedTuple): + uuid: str + file_name: str + content_type: str + + +class DirectoryProjectFiles: + """Project files stored in a directory, named by their UUID or file name.""" + + def __init__(self, root: pathlib.Path): + self.root = root + + def resolve(self, file_uuid: str, name: str, + content_type: str) -> pathlib.Path | None: + for candidate in (self.root / file_uuid, self.root / name): + if candidate.is_file(): + return candidate + return None + + +#: Harmless messages fontconfig >= 2.17 prints for the `normal` +#: WeasyPrint uses for `font-stretch: normal` of a @font-face (the edit is ignored) +FONTCONFIG_NOISE = re.compile( + r'^Fontconfig (error: the ambiguous constant name: normal: ' + r'|warning: "memory", line \d+: invalid constant used : normal\s*$)', +) + + +@contextlib.contextmanager +def filtered_native_stderr(noise: re.Pattern[str] = FONTCONFIG_NOISE): + """Drop known noise that C libraries write directly to stderr (file descriptor 2). + + Everything else they write is passed on once the block ends. + """ + sys.stderr.flush() + try: + saved = os.dup(2) + except OSError: # no stderr to filter + yield + return + with tempfile.TemporaryFile() as capture: + os.dup2(capture.fileno(), 2) + try: + yield + finally: + sys.stderr.flush() + os.dup2(saved, 2) + os.close(saved) + capture.seek(0) + text = capture.read().decode('utf-8', errors='replace') + kept = [line for line in text.splitlines(keepends=True) if not noise.match(line)] + if kept: + sys.stderr.write(''.join(kept)) + sys.stderr.flush() + + +def render_settings(*, pandoc_filters_dir: pathlib.Path | None = None, + pandoc_templates_dir: pathlib.Path | None = None, + secrets: dict[str, str] | None = None, + allow_requests: bool = False) -> RenderSettings: + """Settings the document worker takes from its configuration. + + Filters are searched in `pandoc_filters_dir` first, then in the bundled + ones. `secrets` and `requests` exist in templates only when configured + (as in the worker, for a template with a configuration). + """ + filter_dirs = [bundled_pandoc_filters()] + if pandoc_filters_dir is not None: + filter_dirs.insert(0, pandoc_filters_dir) + template = None + if secrets or allow_requests: + template = TemplateSettings( + secrets=secrets or {}, + requests=RequestsSettings(enabled=allow_requests), + ) + return RenderSettings( + pandoc=PandocSettings(filter_dirs=filter_dirs, templates_dir=pandoc_templates_dir), + template=template, + ) + + +def find_format(template: Template, format_ref: str | None) -> Format: + """Find a format by its UUID or name; the only format when not specified.""" + if format_ref is None: + if len(template.formats) == 1: + return template.formats[0] + names = ', '.join(f'"{f.name}"' for f in template.formats) + raise RuntimeError(f'The template has {len(template.formats)} formats, ' + f'choose one of: {names}') + for format_spec in template.formats: + if format_ref in (format_spec.uuid, format_spec.name) or \ + format_ref.casefold() == (format_spec.name or '').casefold(): + return format_spec + raise RuntimeError(f'Format "{format_ref}" not found in the template') + + +#: Keys of `documentContext` in the document worker configuration, with the +#: fields of ContextDefaults and the environment variables the worker reads +CONTEXT_DEFAULTS_KEYS: dict[str, tuple[str, str]] = { + 'serviceName': ('service_name', 'DOCUMENT_CONTEXT_SERVICE_NAME'), + 'serviceNameShort': ('service_name_short', 'DOCUMENT_CONTEXT_SERVICE_NAME_SHORT'), + 'serviceUrl': ('service_url', 'DOCUMENT_CONTEXT_SERVICE_URL'), + 'serviceDomainName': ('service_domain_name', 'DOCUMENT_CONTEXT_SERVICE_DOMAIN_NAME'), + 'defaultPrimaryColor': ('default_primary_color', 'DOCUMENT_CONTEXT_DEFAULT_PRIMARY_COLOR'), + 'defaultIllustrationsColor': ('default_illustrations_color', + 'DOCUMENT_CONTEXT_DEFAULT_ILLUSTRATIONS_COLOR'), + 'defaultLogoUrl': ('default_logo_url', 'DOCUMENT_CONTEXT_DEFAULT_LOGO_URL'), + 'defaultAppTitle': ('default_app_title', 'DOCUMENT_CONTEXT_DEFAULT_APP_TITLE'), + 'defaultAppTitleShort': ('default_app_title_short', + 'DOCUMENT_CONTEXT_DEFAULT_APP_TITLE_SHORT'), +} + + +def context_defaults(overrides: dict[str, str] | None = None, + environ: typing.Mapping[str, str] = os.environ) -> ContextDefaults: + """The worker's defaults, overridden by its environment variables and then `overrides`. + + `overrides` are keyed as `documentContext` in the worker configuration. + """ + overrides = overrides or {} + unknown = sorted(set(overrides) - set(CONTEXT_DEFAULTS_KEYS)) + if unknown: + raise ValueError(f'Unknown context default {", ".join(unknown)} ' + f'(known: {", ".join(CONTEXT_DEFAULTS_KEYS)})') + values: dict[str, str] = {} + for key, (field, var_name) in CONTEXT_DEFAULTS_KEYS.items(): + if key in overrides: + values[field] = overrides[key] + elif environ.get(var_name): + values[field] = environ[var_name] + return dataclasses.replace(ContextDefaults(), **values) + + +# What the document worker provides for a document without a project; the +# real values come from the database, which is not available locally. +EXTRAS_WITHOUT_PROJECT: dict[str, typing.Any] = { + 'submissions': [], + 'project': None, + 'questionnaire': None, +} + + +def requested_extras(format_spec: Format) -> list[str]: + """Extras the steps of the format ask for (the ``extras`` step option).""" + names: set[str] = set() + for step in format_spec.steps: + names.update(step.options.get('extras', '').split(',')) + return sorted(name for name in names if name in EXTRAS_WITHOUT_PROJECT) + + +def enrich_context(context: dict, format_spec: Format, + defaults: ContextDefaults | None = None) -> list[str]: + """Add what the document worker adds to the context before rendering. + + ``config`` gets the service information and branding fallbacks (the + worker's defaults), and extras requested by the format that the context + does not provide are filled as for a document without a project. Returns + the names of those extras. + """ + context.setdefault('config', {}) + enrich_context_config(context, defaults or ContextDefaults()) + extras = context.setdefault('extras', {}) + missing = [name for name in requested_extras(format_spec) if name not in extras] + for name in missing: + extras[name] = EXTRAS_WITHOUT_PROJECT[name] + return missing + + +def load_translations(po_file: pathlib.Path) -> gettext.GNUTranslations: + """Compile a PO file (e.g. a translated POT file) for rendering.""" + with po_file.open('rb') as file: + catalog = read_po(file) + buffer = io.BytesIO() + write_mo(buffer, catalog) + buffer.seek(0) + return gettext.GNUTranslations(buffer) + + +def _assets(template: Template) -> list[LocalAsset]: + # only what is uploaded as an asset can be fetched as one on the server + return [ + LocalAsset( + uuid=file.remote_uuid or str(uuid.uuid5(uuid.NAMESPACE_URL, name)), + file_name=name, + content_type=file.content_type, + ) + for name, file in template.files.items() + if file.remote_type == TemplateFileType.ASSET + ] + + +def prepare_directory(template: Template, target: pathlib.Path): + """Write the files of the template as the server would have them.""" + for name, file in template.files.items(): + path = target / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(file.content) + + +def render_document(template: Template, *, workdir: pathlib.Path, format_uuid: str, + context: dict, render_ctx: RenderContext, + project_files: DirectoryProjectFiles | None = None, + settings: RenderSettings | None = None) -> DocumentFile: + """Render the document from the template files, written into `workdir`.""" + prepare_directory(template, workdir) + plugins = create_manager() + register_plugin_steps(plugins) + engine = TemplateEngine( + template_dir=workdir, + formats=[f.serialize() for f in template.formats], + assets=_assets(template), + template_uuid=template.uuid or template.coordinates, + coordinates=template.coordinates, + settings=settings or RenderSettings(), + plugins=plugins, + ) + return engine.render(format_uuid, context, + render_ctx=render_ctx, project_files=project_files) diff --git a/packages/dsw-tdk/pyproject.toml b/packages/dsw-tdk/pyproject.toml index 4737108f..f961f4ba 100644 --- a/packages/dsw-tdk/pyproject.toml +++ b/packages/dsw-tdk/pyproject.toml @@ -19,14 +19,7 @@ classifiers = [ "Topic :: Utilities", ] requires-python = ">=3.12, <4" -dynamic = ["version", "dependencies"] - -[project.optional-dependencies] -test = [ - "pytest", - "pytest-recording", - "vcrpy", -] +dynamic = ["version", "dependencies", "optional-dependencies"] [project.urls] Homepage = "https://ds-wizard.org" @@ -46,7 +39,12 @@ build-backend = "hatchling.build" source = "uv-dynamic-versioning" [tool.hatch.metadata.hooks.uv-dynamic-versioning] -dependencies = ["aiohttp", "Babel", "click", "colorama", "humanize", "Jinja2", "multidict", "pathspec", "pydantic", "python-dotenv", "python-slugify", "watchfiles", "dsw-models=={{ version }}"] +dependencies = ["aiohttp", "Babel", "click", "colorama", "humanize", "Jinja2", "multidict", "pathspec", "pydantic", "python-dotenv", "python-slugify", "watchfiles", "dsw-models=={{ version }}", "dsw-templating=={{ version }}"] + +# `all` adds what the optional rendering steps need (PDF, Excel, RDF, HTTP), see dsw-templating +[tool.hatch.metadata.hooks.uv-dynamic-versioning.optional-dependencies] +all = ["dsw-templating[all]=={{ version }}"] +test = ["pytest", "pytest-recording", "vcrpy"] [tool.hatch.build.targets.wheel] packages = ["dsw"] diff --git a/packages/dsw-tdk/tests/fixtures/test_render01/README.md b/packages/dsw-tdk/tests/fixtures/test_render01/README.md new file mode 100644 index 00000000..bad78ddf --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render01/README.md @@ -0,0 +1 @@ +# Test template for rendering diff --git a/packages/dsw-tdk/tests/fixtures/test_render01/assets/logo.png b/packages/dsw-tdk/tests/fixtures/test_render01/assets/logo.png new file mode 100644 index 00000000..f37764b1 Binary files /dev/null and b/packages/dsw-tdk/tests/fixtures/test_render01/assets/logo.png differ diff --git a/packages/dsw-tdk/tests/fixtures/test_render01/context.json b/packages/dsw-tdk/tests/fixtures/test_render01/context.json new file mode 100644 index 00000000..4308f53b --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render01/context.json @@ -0,0 +1,5 @@ +{ + "config": {"clientUrl": "https://wizard.example.org/", "appTitle": "My Wizard"}, + "document": {"name": "My Plan", "language": "cs"}, + "files": [{"uuid": "22222222-0000-0000-0000-000000000001", "fileName": "notes.txt", "contentType": "text/plain"}] +} diff --git a/packages/dsw-tdk/tests/fixtures/test_render01/cs.po b/packages/dsw-tdk/tests/fixtures/test_render01/cs.po new file mode 100644 index 00000000..7c42f637 --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render01/cs.po @@ -0,0 +1,7 @@ +msgid "" +msgstr "" +"Content-Type: text/plain; charset=utf-8\n" +"Language: cs\n" + +msgid "Data Management Plan" +msgstr "Plán správy dat" diff --git a/packages/dsw-tdk/tests/fixtures/test_render01/project-files/notes.txt b/packages/dsw-tdk/tests/fixtures/test_render01/project-files/notes.txt new file mode 100644 index 00000000..04030e40 --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render01/project-files/notes.txt @@ -0,0 +1 @@ +Attached notes \ No newline at end of file diff --git a/packages/dsw-tdk/tests/fixtures/test_render01/src/document.html.j2 b/packages/dsw-tdk/tests/fixtures/test_render01/src/document.html.j2 new file mode 100644 index 00000000..cdfc54e9 --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render01/src/document.html.j2 @@ -0,0 +1,6 @@ +{%- set logo = assets('assets/logo.png') -%} +{%- set attachment = assets(ctx.files[0]) -%} +

{% trans %}Data Management Plan{% endtrans %}

+

{{ ctx.document.name }}

+ +

{{ attachment.data.decode() if attachment else 'no attachment' }}

diff --git a/packages/dsw-tdk/tests/fixtures/test_render01/src/excluded.html.j2 b/packages/dsw-tdk/tests/fixtures/test_render01/src/excluded.html.j2 new file mode 100644 index 00000000..61daed30 --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render01/src/excluded.html.j2 @@ -0,0 +1 @@ +

not uploaded

diff --git a/packages/dsw-tdk/tests/fixtures/test_render01/template.json b/packages/dsw-tdk/tests/fixtures/test_render01/template.json new file mode 100644 index 00000000..86ac5fe6 --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render01/template.json @@ -0,0 +1,50 @@ +{ + "organizationId": "test", + "templateId": "render01", + "version": "1.0.0", + "name": "Test template for rendering", + "description": "Dummy document template for rendering locally", + "metamodelVersion": "18.3", + "language": "en", + "license": "Apache-2.0", + "allowedPackages": [], + "formats": [ + { + "uuid": "d3e98eb6-344d-481f-8e37-6a67b6cd1ad2", + "name": "JSON Data", + "icon": "far fa-file", + "steps": [{"name": "json", "options": {}}] + }, + { + "uuid": "a9293d08-59a4-4e6b-ae62-7a6a570b031c", + "name": "HTML Document", + "icon": "far fa-file-code", + "steps": [ + {"name": "jinja", "options": {"template": "src/document.html.j2", "content-type": "text/html", "extension": "html"}} + ] + }, + { + "uuid": "0f5c8a55-9d2e-4b8e-8e57-7d6c0b7f3e10", + "name": "Project JSON", + "icon": "far fa-file", + "steps": [{"name": "json", "options": {"extras": "project,submissions"}}] + }, + { + "uuid": "6e1b4ee5-4c47-4a2c-9a5b-3cf8b1b6c2a1", + "name": "Excluded", + "icon": "far fa-file-code", + "steps": [ + {"name": "jinja", "options": {"template": "src/excluded.html.j2", "content-type": "text/html", "extension": "html"}} + ] + } + ], + "_tdk": { + "version": "4.34.0", + "readmeFile": "README.md", + "files": [ + "src/**/*", + "assets/**/*", + "!src/excluded.html.j2" + ] + } +} diff --git a/packages/dsw-tdk/tests/fixtures/test_render02/README.md b/packages/dsw-tdk/tests/fixtures/test_render02/README.md new file mode 100644 index 00000000..6523a58a --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render02/README.md @@ -0,0 +1 @@ +# Test template for rendering options diff --git a/packages/dsw-tdk/tests/fixtures/test_render02/context.json b/packages/dsw-tdk/tests/fixtures/test_render02/context.json new file mode 100644 index 00000000..44782d0c --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render02/context.json @@ -0,0 +1 @@ +{"config": {"clientUrl": ""}, "document": {"name": "Doc"}} diff --git a/packages/dsw-tdk/tests/fixtures/test_render02/pandoc-filters/mine.lua b/packages/dsw-tdk/tests/fixtures/test_render02/pandoc-filters/mine.lua new file mode 100644 index 00000000..d6ed9661 --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render02/pandoc-filters/mine.lua @@ -0,0 +1 @@ +-- custom filter diff --git a/packages/dsw-tdk/tests/fixtures/test_render02/pandoc-templates/custom.docx b/packages/dsw-tdk/tests/fixtures/test_render02/pandoc-templates/custom.docx new file mode 100644 index 00000000..8f91b47d --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render02/pandoc-templates/custom.docx @@ -0,0 +1 @@ +template \ No newline at end of file diff --git a/packages/dsw-tdk/tests/fixtures/test_render02/src/document.html.j2 b/packages/dsw-tdk/tests/fixtures/test_render02/src/document.html.j2 new file mode 100644 index 00000000..8a748730 --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render02/src/document.html.j2 @@ -0,0 +1 @@ +

{{ ctx.document.name }}

diff --git a/packages/dsw-tdk/tests/fixtures/test_render02/src/globals.txt.j2 b/packages/dsw-tdk/tests/fixtures/test_render02/src/globals.txt.j2 new file mode 100644 index 00000000..661a75c4 --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render02/src/globals.txt.j2 @@ -0,0 +1 @@ +{{ secrets.token if secrets is defined else "no secrets" }}|{{ "requests" if requests is defined else "no requests" }} diff --git a/packages/dsw-tdk/tests/fixtures/test_render02/template.json b/packages/dsw-tdk/tests/fixtures/test_render02/template.json new file mode 100644 index 00000000..0390d542 --- /dev/null +++ b/packages/dsw-tdk/tests/fixtures/test_render02/template.json @@ -0,0 +1,35 @@ +{ + "organizationId": "test", + "templateId": "render02", + "version": "1.0.0", + "name": "Test template for rendering options", + "description": "Dummy document template for secrets, requests and pandoc", + "metamodelVersion": "18.3", + "language": "en", + "license": "Apache-2.0", + "allowedPackages": [], + "formats": [ + { + "uuid": "b5a1d3f2-3f9e-4f5b-9a37-2a1a4c9b8d01", + "name": "Globals", + "icon": "far fa-file-code", + "steps": [ + {"name": "jinja", "options": {"template": "src/globals.txt.j2", "content-type": "text/plain", "extension": "txt"}} + ] + }, + { + "uuid": "b5a1d3f2-3f9e-4f5b-9a37-2a1a4c9b8d02", + "name": "Word", + "icon": "far fa-file-word", + "steps": [ + {"name": "jinja", "options": {"template": "src/document.html.j2"}}, + {"name": "pandoc", "options": {"from": "html", "to": "docx", "template": "custom.docx", "filters": "docx-pagebreak.lua,mine.lua"}} + ] + } + ], + "_tdk": { + "version": "4.34.0", + "readmeFile": "README.md", + "files": ["src/**/*"] + } +} diff --git a/packages/dsw-tdk/tests/test_cmd_render.py b/packages/dsw-tdk/tests/test_cmd_render.py new file mode 100644 index 00000000..8b95278a --- /dev/null +++ b/packages/dsw-tdk/tests/test_cmd_render.py @@ -0,0 +1,166 @@ +# cspell:ignore Plán správy +import json +import pathlib + +import click.testing + +from dsw.tdk import main + + +def render(fixtures_path: pathlib.Path, tmp_path: pathlib.Path, *args: str): + template_path = fixtures_path / 'test_render01' + runner = click.testing.CliRunner() + return runner.invoke(main, args=[ + 'render', template_path.as_posix(), + '--context', (template_path / 'context.json').as_posix(), + *args, + ]) + + +def test_render_json(fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + output = tmp_path / 'out.json' + result = render(fixtures_path, tmp_path, '--format', 'JSON Data', '-o', output.as_posix()) + assert result.exit_code == 0, result.output + assert 'rendered (application/json' in result.output + assert json.loads(output.read_text(encoding='utf-8'))['document']['name'] == 'My Plan' + + +def test_render_adds_worker_defaults(fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + output = tmp_path / 'out.json' + result = render(fixtures_path, tmp_path, '--format', 'JSON Data', '-o', output.as_posix()) + assert result.exit_code == 0, result.output + context = json.loads(output.read_text(encoding='utf-8')) + assert context['config']['serviceName'] == 'Data Stewardship Wizard' + assert context['config']['appTitle'] == 'My Wizard' # kept from the context + assert context['config']['primaryColor'] == '#0033aa' # default for a missing one + assert context['config']['logoUrl'] == 'https://wizard.example.org/assets/logo.svg' + assert context['extras'] == {} + + +def test_render_fills_requested_extras(fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + output = tmp_path / 'out.json' + result = render(fixtures_path, tmp_path, '--format', 'Project JSON', '-o', output.as_posix()) + assert result.exit_code == 0, result.output + assert 'Extra "project" is not in the context' in result.output + assert 'Extra "submissions" is not in the context' in result.output + context = json.loads(output.read_text(encoding='utf-8')) + assert context['extras'] == {'project': None, 'submissions': []} + + +def test_render_html_by_uuid_with_translations_and_project_files( + fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + template_path = fixtures_path / 'test_render01' + output = tmp_path / 'out.html' + result = render( + fixtures_path, tmp_path, + '--format', 'a9293d08-59a4-4e6b-ae62-7a6a570b031c', + '--po', (template_path / 'cs.po').as_posix(), + '--project-files', (template_path / 'project-files').as_posix(), + '-o', output.as_posix(), + ) + assert result.exit_code == 0, result.output + content = output.read_text(encoding='utf-8') + assert '

Plán správy dat

' in content + assert '

My Plan

' in content + assert 'src="data:image/png;base64,' in content + assert '

Attached notes

' in content + + +def test_render_html_without_translations_and_project_files( + fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + output = tmp_path / 'out.html' + result = render(fixtures_path, tmp_path, '--format', 'html document', '-o', output.as_posix()) + assert result.exit_code == 0, result.output + content = output.read_text(encoding='utf-8') + assert '

Data Management Plan

' in content + assert '

no attachment

' in content + assert 'WARNING' in result.output # the project file cannot be fetched + + +def test_render_default_output_name(fixtures_path: pathlib.Path, tmp_path: pathlib.Path, + monkeypatch): + monkeypatch.chdir(tmp_path) + result = render(fixtures_path, tmp_path, '--format', 'JSON Data') + assert result.exit_code == 0, result.output + assert (tmp_path / 'render01.json').is_file() + + +def test_render_existing_output_without_force(fixtures_path: pathlib.Path, + tmp_path: pathlib.Path): + output = tmp_path / 'out.json' + output.write_text('original', encoding='utf-8') + result = render(fixtures_path, tmp_path, '--format', 'JSON Data', '-o', output.as_posix()) + assert result.exit_code == 1 + assert 'already exists' in result.output + assert output.read_text(encoding='utf-8') == 'original' + result = render(fixtures_path, tmp_path, '--format', 'JSON Data', '-o', output.as_posix(), + '--force') + assert result.exit_code == 0, result.output + + +def test_render_requires_format_choice(fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + result = render(fixtures_path, tmp_path) + assert result.exit_code == 1 + assert 'choose one of: "JSON Data", "HTML Document", "Project JSON", "Excluded"' \ + in result.output + + +def test_render_unknown_format(fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + result = render(fixtures_path, tmp_path, '--format', 'PDF') + assert result.exit_code == 1 + assert 'Format "PDF" not found' in result.output + + +def test_render_uses_only_template_files(fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + # src/excluded.html.j2 is not in _tdk.files, so the server would not have it either + result = render(fixtures_path, tmp_path, '--format', 'Excluded', + '-o', (tmp_path / 'out.html').as_posix()) + assert result.exit_code == 1 + assert 'Failed to render the document' in result.output + assert 'src/excluded.html.j2' in result.output + + +def test_render_context_defaults_override(fixtures_path: pathlib.Path, tmp_path: pathlib.Path, + monkeypatch): + monkeypatch.setenv('DOCUMENT_CONTEXT_SERVICE_URL', 'https://fair-wizard.com') + monkeypatch.setenv('DOCUMENT_CONTEXT_SERVICE_NAME', 'From environment') + output = tmp_path / 'out.json' + result = render(fixtures_path, tmp_path, '--format', 'JSON Data', '-o', output.as_posix(), + '-D', 'serviceName=FAIR Wizard', + '--context-default', 'defaultPrimaryColor=#123456', + '-D', 'defaultAppTitle=FW') + assert result.exit_code == 0, result.output + config = json.loads(output.read_text(encoding='utf-8'))['config'] + assert config['serviceName'] == 'FAIR Wizard' # option wins over the environment + assert config['serviceUrl'] == 'https://fair-wizard.com' # from the environment + assert config['primaryColor'] == '#123456' # missing in the context + assert config['appTitle'] == 'My Wizard' # the context wins over a default + assert config['serviceNameShort'] == 'DSW' # untouched default + + +def test_render_context_defaults_from_dot_env(fixtures_path: pathlib.Path, + tmp_path: pathlib.Path, monkeypatch): + monkeypatch.delenv('DOCUMENT_CONTEXT_SERVICE_NAME', raising=False) + dot_env = tmp_path / '.env' + dot_env.write_text('DOCUMENT_CONTEXT_SERVICE_NAME=Dot Env Wizard\n', encoding='utf-8') + template_path = fixtures_path / 'test_render01' + output = tmp_path / 'out.json' + result = click.testing.CliRunner().invoke(main, args=[ + '--dot-env', dot_env.as_posix(), '--no-config', + 'render', template_path.as_posix(), '-c', (template_path / 'context.json').as_posix(), + '-F', 'JSON Data', '-o', output.as_posix(), + ]) + monkeypatch.delenv('DOCUMENT_CONTEXT_SERVICE_NAME', raising=False) # set by load_dotenv + assert result.exit_code == 0, result.output + config = json.loads(output.read_text(encoding='utf-8'))['config'] + assert config['serviceName'] == 'Dot Env Wizard' + + +def test_render_context_defaults_invalid(fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + result = render(fixtures_path, tmp_path, '-F', 'JSON Data', '-D', 'serviceName') + assert result.exit_code == 2 + assert '"serviceName" is not NAME=VALUE' in result.output + result = render(fixtures_path, tmp_path, '-F', 'JSON Data', '-D', 'appTitle=X') + assert result.exit_code == 2 + assert 'Unknown context default appTitle' in result.output + assert 'defaultAppTitle' in result.output diff --git a/packages/dsw-tdk/tests/test_cmd_render_options.py b/packages/dsw-tdk/tests/test_cmd_render_options.py new file mode 100644 index 00000000..81104631 --- /dev/null +++ b/packages/dsw-tdk/tests/test_cmd_render_options.py @@ -0,0 +1,80 @@ +import importlib.util +import pathlib + +import click.testing + +from dsw.tdk import main + + +def render(fixtures_path: pathlib.Path, output: pathlib.Path, *args: str): + template_path = fixtures_path / 'test_render02' + return click.testing.CliRunner().invoke(main, args=[ + '--no-config', '--no-dot-env', + 'render', template_path.as_posix(), + '--context', (template_path / 'context.json').as_posix(), + '-o', output.as_posix(), '--force', + *args, + ]) + + +def test_no_globals_by_default(fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + output = tmp_path / 'out.txt' + result = render(fixtures_path, output, '-F', 'Globals') + assert result.exit_code == 0, result.output + assert output.read_text(encoding='utf-8').strip() == 'no secrets|no requests' + + +def test_secrets_and_requests(fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + output = tmp_path / 'out.txt' + result = render(fixtures_path, output, '-F', 'Globals', + '--secret', 'token=abc=123', '--allow-requests') + if importlib.util.find_spec('requests') is None: # without dsw-tdk[all] + assert result.exit_code == 1 + assert 'install dsw-templating[http]' in result.output + return + assert result.exit_code == 0, result.output + assert output.read_text(encoding='utf-8').strip() == 'abc=123|requests' + + +def test_secret_without_requests(fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + output = tmp_path / 'out.txt' + result = render(fixtures_path, output, '-F', 'Globals', '-s', 'token=x') + assert result.exit_code == 0, result.output + assert output.read_text(encoding='utf-8').strip() == 'x|no requests' + + +def test_invalid_secret(fixtures_path: pathlib.Path, tmp_path: pathlib.Path): + result = render(fixtures_path, tmp_path / 'out.txt', '-F', 'Globals', '-s', 'token') + assert result.exit_code == 2 + assert '"token" is not NAME=VALUE' in result.output + + +def test_pandoc_needs_templates_and_filters(fixtures_path: pathlib.Path, tmp_path: pathlib.Path, + monkeypatch): + monkeypatch.delenv('PANDOC_TEMPLATES', raising=False) + monkeypatch.delenv('PANDOC_FILTERS', raising=False) + result = render(fixtures_path, tmp_path / 'out.docx', '-F', 'Word') + assert result.exit_code == 1 + assert 'Pandoc filter "mine.lua" not found' in result.output + + +def test_pandoc_with_templates_and_filters(fixtures_path: pathlib.Path, tmp_path: pathlib.Path, + monkeypatch): + monkeypatch.setenv('PATH', str(tmp_path)) # no pandoc, whatever is installed + template_path = fixtures_path / 'test_render02' + monkeypatch.setenv('PANDOC_TEMPLATES', (template_path / 'pandoc-templates').as_posix()) + result = render(fixtures_path, tmp_path / 'out.docx', '-F', 'Word', + '--pandoc-filters', (template_path / 'pandoc-filters').as_posix()) + assert result.exit_code == 1 + # templates and filters were found, it got as far as running pandoc + assert 'Executable "pandoc" not found, is it installed?' in result.output + + +def test_pandoc_template_missing(fixtures_path: pathlib.Path, tmp_path: pathlib.Path, + monkeypatch): + template_path = fixtures_path / 'test_render02' + monkeypatch.delenv('PANDOC_TEMPLATES', raising=False) + result = render(fixtures_path, tmp_path / 'out.docx', '-F', 'Word', + '--pandoc-filters', (template_path / 'pandoc-filters').as_posix()) + assert result.exit_code == 1 + assert 'Pandoc template "custom.docx" not found' in result.output diff --git a/packages/dsw-tdk/tests/test_render_stderr.py b/packages/dsw-tdk/tests/test_render_stderr.py new file mode 100644 index 00000000..cd6c6f86 --- /dev/null +++ b/packages/dsw-tdk/tests/test_render_stderr.py @@ -0,0 +1,33 @@ +import os + +from dsw.tdk.render import filtered_native_stderr + + +NOISE = ( + b'Fontconfig error: the ambiguous constant name: normal: ' + b'Use := instead of :\n' + b'Fontconfig warning: "memory", line 3: invalid constant used : normal\n' +) + + +def test_fontconfig_noise_is_dropped(capfd): + with filtered_native_stderr(): + os.write(2, NOISE * 3) + assert capfd.readouterr().err == '' + + +def test_other_native_output_is_kept(capfd): + with filtered_native_stderr(): + os.write(2, NOISE + b'Fontconfig error: something else\nreal problem\n' + NOISE) + assert capfd.readouterr().err == 'Fontconfig error: something else\nreal problem\n' + + +def test_stderr_is_restored_after_an_error(capfd): + try: + with filtered_native_stderr(): + os.write(2, b'before failure\n') + raise RuntimeError('boom') + except RuntimeError: + pass + os.write(2, b'after\n') + assert capfd.readouterr().err == 'before failure\nafter\n' diff --git a/packages/dsw-templating/.gitignore b/packages/dsw-templating/.gitignore new file mode 100644 index 00000000..d18f6ea5 --- /dev/null +++ b/packages/dsw-templating/.gitignore @@ -0,0 +1,134 @@ +# Byte-compiled / optimized / DLL files +__pycache__/ +*.py[cod] +*$py.class + +# C extensions +*.so + +# Distribution / packaging +.Python +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +wheels/ +pip-wheel-metadata/ +share/python-wheels/ +*.egg-info/ +.installed.cfg +*.egg +MANIFEST + +# PyInstaller +# Usually these files are written by a python script from a template +# before PyInstaller builds the exe, so as to inject date/other infos into it. +*.manifest +*.spec + +# Installer logs +pip-log.txt +pip-delete-this-directory.txt + +# Unit test / coverage reports +htmlcov/ +.tox/ +.nox/ +.coverage +.coverage.* +.cache +nosetests.xml +coverage.xml +*.cover +*.py,cover +.hypothesis/ +.pytest_cache/ + +# Translations +*.mo +*.pot + +# Django stuff: +*.log +local_settings.py +db.sqlite3 +db.sqlite3-journal + +# Flask stuff: +instance/ +.webassets-cache + +# Scrapy stuff: +.scrapy + +# Sphinx documentation +docs/_build/ + +# PyBuilder +target/ + +# Jupyter Notebook +.ipynb_checkpoints + +# IPython +profile_default/ +ipython_config.py + +# pyenv +.python-version + +# pipenv +# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control. +# However, in case of collaboration, if having platform-specific dependencies or dependencies +# having no cross-platform support, pipenv may install dependencies that don't work, or not +# install all needed dependencies. +#Pipfile.lock + +# PEP 582; used by e.g. github.com/David-OConnor/pyflow +__pypackages__/ + +# Celery stuff +celerybeat-schedule +celerybeat.pid + +# SageMath parsed files +*.sage.py + +# Environments +.env +.venv +env/ +venv/ +ENV/ +env.bak/ +venv.bak/ + +# Spyder project settings +.spyderproject +.spyproject + +# Rope project settings +.ropeproject + +# mkdocs documentation +/site + +# mypy +.mypy_cache/ +.dmypy.json +dmypy.json + +# Pyre type checker +.pyre/ + +# IDE +.idea/ +.vscode/ + diff --git a/packages/dsw-templating/CHANGELOG.md b/packages/dsw-templating/CHANGELOG.md new file mode 100644 index 00000000..a80944f5 --- /dev/null +++ b/packages/dsw-templating/CHANGELOG.md @@ -0,0 +1,23 @@ +# Changelog + +All notable changes to this project will be documented in this file. + +The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), and this project adheres +to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). + +## [Unreleased] + +### Added + +- Rendering engine of document templates extracted from `dsw-document-worker`: formats, steps, + Jinja filters and tests, document context helpers, conversions (Pandoc, WeasyPrint, RDFLib, + Excel), URL policy, HTTP requests from templates and POT file extraction +- `Template` built from a template directory and its format metadata (`Template.from_directory` + reads `template.json`), rendered with explicit `RenderSettings`, `RenderContext` and + `ProjectFileResolver` instead of the worker's configuration and services +- Optional extras `pdf`, `rdf`, `excel`, `docx`, `http` and `all`; a step whose extra is missing + fails with `MissingExtraError` when it is used +- Pandoc filters shipped as package data (`settings.bundled_pandoc_filters`) +- `enrich_context_config` and `ContextDefaults`: the service information and branding fallbacks the document worker adds to `ctx.config` +- Plugins use the `dsw-templating` pluggy project and the `dsw_templating_plugins` entry point +- The `pandoc-docx-pagebreakpy` Pandoc filter is a command of this package (`docx` extra) instead of a separate addon installed only in the document worker image diff --git a/packages/dsw-templating/LICENSE b/packages/dsw-templating/LICENSE new file mode 100644 index 00000000..de3f76eb --- /dev/null +++ b/packages/dsw-templating/LICENSE @@ -0,0 +1,202 @@ + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright 2020 Marek Suchánek + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/packages/dsw-templating/Makefile b/packages/dsw-templating/Makefile new file mode 100644 index 00000000..b6c7eb26 --- /dev/null +++ b/packages/dsw-templating/Makefile @@ -0,0 +1,10 @@ +PYTHON = python + +.PHONY: verify +verify: + @echo "Trying to import: dsw.templating" + $(PYTHON) -c 'import dsw.templating' + +.PHONY: test +test: + pytest -s tests diff --git a/packages/dsw-templating/README.md b/packages/dsw-templating/README.md new file mode 100644 index 00000000..3f0d3a6f --- /dev/null +++ b/packages/dsw-templating/README.md @@ -0,0 +1,78 @@ +# Data Stewardship Wizard: Templating + +[![GitHub release (latest SemVer)](https://img.shields.io/github/v/release/ds-wizard/engine-tools)](https://github.com/ds-wizard/engine-tools/releases) +[![PyPI](https://img.shields.io/pypi/v/dsw-templating)](https://pypi.org/project/dsw-templating/) +[![LICENSE](https://img.shields.io/github/license/ds-wizard/engine-tools)](LICENSE) +[![CII Best Practices](https://bestpractices.coreinfrastructure.org/projects/4975/badge)](https://bestpractices.coreinfrastructure.org/projects/4975) +[![Python Version](https://img.shields.io/badge/Python-%E2%89%A5%203.12-blue)](https://python.org) + +*Rendering of Data Stewardship Wizard document templates* + +The engine behind the [document worker](../dsw-document-worker): it renders a document from a +document template on disk and a document context, with no database, storage or queue behind it. + +## Installation + +```bash +pip install dsw-templating +pip install 'dsw-templating[all]' # every optional step dependency +``` + +| Extra | Needed for | +|---|---| +| `pdf` | `weasyprint` step (WeasyPrint, needs Pango installed) | +| `rdf` | `rdflib-convert` step and the `rdflib` global in Jinja templates | +| `excel` | `excel` step (XlsxWriter) | +| `docx` | the `pandoc-docx-pagebreakpy` Pandoc filter command (panflute) | +| `http` | `requests` global in Jinja templates (Requests) | + +A step whose extra is missing raises `MissingExtraError` ("install dsw-templating[pdf]") when it +is used. The `pandoc` step needs the [pandoc](https://pandoc.org) executable. + +## Usage + +```python +import json +import pathlib + +from dsw.templating import RenderContext, RenderSettings, Template + +template = Template.from_directory(pathlib.Path('my-template'), settings=RenderSettings()) +context = json.loads(pathlib.Path('context.json').read_text()) +document = template.render('d3e98eb6-344d-481f-8e37-6a67b6cd1ad2', context, + render_ctx=RenderContext.null(language='en')) +document.store('document') # document. +``` + +Everything the rendering depends on is passed in explicitly: + +| Input | Purpose | +|---|---| +| `Template(template_dir=..., formats=..., assets=...)` | Template files on disk and the formats of `template.json` (`from_directory` reads both) | +| `RenderSettings` | URL security policy (`SecuritySettings`), Pandoc command, filters and timeout (`PandocSettings`), per-template `secrets` and HTTP `requests` (`TemplateSettings`) | +| `RenderContext` | Translations (`gettext`), language and locale of the document | +| `ProjectFileResolver` | Files uploaded to the project (`assets(file)` in templates); none by default | +| `plugins` | A pluggy manager; `create_manager()` loads plugins of the `dsw_templating_plugins` entry point | + +Plugins implement the hooks of `dsw.templating.plugins.specs` (marked with +`dsw.templating.hookimpl`); steps they provide are registered with `register_plugin_steps(pm)`. + +`dsw.templating.pot` extracts translatable messages of Jinja files into a POT file, and +`enrich_context_config(context, ContextDefaults())` adds to `ctx.config` what the document worker +adds (service information and fallbacks for missing branding). + +## Documentation for template developers + +* [Document Context](./support/DocumentContext.md) +* [Jinja Filters](./support/JinjaFilters.md) +* [Jinja Tests](./support/JinjaTests.md) +* [Translations](./support/Translations.md) +* Steps: [archive](./support/steps/archive.md), [enrich-docx](./support/steps/enrich-docx.md), + [excel](./support/steps/excel.md), [jinja](./support/steps/jinja.md), [json](./support/steps/json.md), + [pandoc](./support/steps/pandoc.md), [rdflib-convert](./support/steps/rdflib-convert.md), + [weasyprint](./support/steps/weasyprint.md) + +## License + +This project is licensed under the Apache License v2.0 - see the +[LICENSE](LICENSE) file for more details. diff --git a/packages/dsw-templating/dsw/templating/__init__.py b/packages/dsw-templating/dsw/templating/__init__.py new file mode 100644 index 00000000..687c0fb3 --- /dev/null +++ b/packages/dsw-templating/dsw/templating/__init__.py @@ -0,0 +1,32 @@ +"""Rendering of DSW document templates, without any services behind it.""" +from .consts import VERSION +from .context import ContextDefaults, enrich_context_config +from .documents import DocumentFile, FileFormat, FileFormats +from .exceptions import MissingExtraError, TemplateError, TemplateTriggeredError +from .formats import Format +from .locales import RenderContext, TemplateLocale +from .plugins import create_manager, hookimpl, hookspec, register_plugin_steps +from .settings import ( + PandocSettings, + RenderSettings, + RequestsSettings, + SecuritySettings, + TemplateSettings, +) +from .steps import FormatStepError, Step +from .steps.base import register_step +from .template import Asset, AssetMetadata, ProjectFileResolver, Template + + +__all__ = [ + 'VERSION', + 'ContextDefaults', 'enrich_context_config', + 'Asset', 'AssetMetadata', 'ProjectFileResolver', 'Template', + 'DocumentFile', 'FileFormat', 'FileFormats', + 'Format', 'FormatStepError', 'Step', 'register_step', + 'MissingExtraError', 'TemplateError', 'TemplateTriggeredError', + 'RenderContext', 'TemplateLocale', + 'PandocSettings', 'RenderSettings', 'RequestsSettings', 'SecuritySettings', + 'TemplateSettings', + 'create_manager', 'hookimpl', 'hookspec', 'register_plugin_steps', +] diff --git a/packages/dsw-templating/dsw/templating/consts.py b/packages/dsw-templating/dsw/templating/consts.py new file mode 100644 index 00000000..72ec8841 --- /dev/null +++ b/packages/dsw-templating/dsw/templating/consts.py @@ -0,0 +1,40 @@ +from __future__ import annotations + +from importlib.metadata import PackageNotFoundError, version + + +DEFAULT_ENCODING = 'utf-8' +EXIT_SUCCESS = 0 +PACKAGE_NAME = 'dsw-templating' +PLUGINS_ENTRYPOINT = 'dsw_templating_plugins' + +JINJA_EXTENSIONS = ('jinja2.ext.do', 'jinja2.ext.loopcontrols') +JINJA_FILE_EXTENSIONS = ('.j2', '.jinja', '.jinja2', '.jnj') + +# Rendering and POT extraction must agree on this: the msgid of a {% trans %} +# block depends on it, and the POT file is per document template while a step +# option would be per format. +JINJA_I18N_TRIMMED = True + +DEFAULT_LANGUAGE = 'en' +DEFAULT_LOCALE_DOMAIN = 'default' + +TEMPLATE_JSON_FILE_NAME = 'template.json' +PROJECT_FILES_DIR = 'project-files' + +try: + __version__ = version(PACKAGE_NAME) +except PackageNotFoundError: + __version__ = '0.0.0' +VERSION = __version__ + + +class FormatField: + UUID = 'uuid' + NAME = 'name' + STEPS = 'steps' + + +class StepField: + NAME = 'name' + OPTIONS = 'options' diff --git a/packages/dsw-templating/dsw/templating/context.py b/packages/dsw-templating/dsw/templating/context.py new file mode 100644 index 00000000..2358d554 --- /dev/null +++ b/packages/dsw-templating/dsw/templating/context.py @@ -0,0 +1,55 @@ +"""Values the document worker adds to the document context before rendering.""" +from __future__ import annotations + +import dataclasses + + +@dataclasses.dataclass +class ContextDefaults: + """Service information and fallbacks of the instance branding (``ctx.config``). + + The defaults are those of the document worker's configuration. + """ + service_name: str = 'Data Stewardship Wizard' + service_name_short: str = 'DSW' + service_url: str = 'https://ds-wizard.org' + service_domain_name: str = 'ds-wizard.org' + default_primary_color: str = '#0033aa' + default_illustrations_color: str = '#0033aa' + default_logo_url: str = '{{clientUrl}}/assets/logo.svg' + default_app_title: str = 'DS Wizard' + default_app_title_short: str = 'DS Wizard' + + +def enrich_context_config(context: dict, defaults: ContextDefaults): + """Fill ``context['config']`` as the document worker does. + + The service fields are always set from `defaults`; the branding of the + instance is kept when present and falls back to `defaults` otherwise. + """ + old = context.get('config', {}) + + client_url = old.get('clientUrl', '').rstrip('/') + app_title = (old.get('appTitle', None) or + defaults.default_app_title) + app_title_short = (old.get('appTitleShort', None) or + defaults.default_app_title_short) + primary_color = (old.get('primaryColor', None) or + defaults.default_primary_color) + illustrations_color = (old.get('illustrationsColor', None) or + defaults.default_illustrations_color) + logo_url_template = (old.get('logoUrl', None) or + defaults.default_logo_url) + logo_url = logo_url_template.replace('{{clientUrl}}', client_url) + + context['config'].update({ + 'serviceName': defaults.service_name, + 'serviceNameShort': defaults.service_name_short, + 'serviceUrl': defaults.service_url, + 'serviceDomainName': defaults.service_domain_name, + 'appTitle': app_title, + 'appTitleShort': app_title_short, + 'primaryColor': primary_color, + 'illustrationsColor': illustrations_color, + 'logoUrl': logo_url, + }) diff --git a/packages/dsw-document-worker/dsw/document_worker/conversions.py b/packages/dsw-templating/dsw/templating/conversions.py similarity index 70% rename from packages/dsw-document-worker/dsw/document_worker/conversions.py rename to packages/dsw-templating/dsw/templating/conversions.py index 9c74758f..9ce5f4c6 100644 --- a/packages/dsw-document-worker/dsw/document_worker/conversions.py +++ b/packages/dsw-templating/dsw/templating/conversions.py @@ -1,20 +1,17 @@ from __future__ import annotations import logging -import os -import pathlib import shlex import subprocess import typing -import rdflib - from . import consts from .documents import FileFormat, FileFormats +from .exceptions import require if typing.TYPE_CHECKING: - from .config import DocumentWorkerConfig + from .settings import PandocSettings LOG = logging.getLogger(__name__) @@ -25,8 +22,15 @@ def run_conversion(*, args: list, workdir: str, input_data: bytes, name: str, command = ' '.join(args) LOG.info('Calling "%s" to convert from %s to %s', command, source_format, target_format) - with subprocess.Popen(args, cwd=workdir, stdin=subprocess.PIPE, - stdout=subprocess.PIPE, stderr=subprocess.PIPE) as proc: + try: + proc = subprocess.Popen(args, cwd=workdir, stdin=subprocess.PIPE, + stdout=subprocess.PIPE, stderr=subprocess.PIPE) + except FileNotFoundError as e: + raise FormatConversionError( + name, source_format, target_format, + f'Executable "{args[0]}" not found, is it installed?', + ) from e + with proc: stdout, stderr = proc.communicate(input=input_data, timeout=timeout) exit_code = proc.returncode if exit_code != consts.EXIT_SUCCESS: @@ -52,12 +56,10 @@ def __str__(self): class Pandoc: - FILTERS_PATH = pathlib.Path(os.getenv('PANDOC_FILTERS', '/pandoc/filters')) - TEMPLATES_PATH = pathlib.Path(os.getenv('PANDOC_TEMPLATES', '/pandoc/templates')) - def __init__(self, config: DocumentWorkerConfig, filter_names: list[str], + def __init__(self, settings: PandocSettings, filter_names: list[str], template_name: str | None): - self.config = config + self.settings = settings self.filter_names = filter_names self.template_name = template_name self._check_filters() @@ -65,30 +67,29 @@ def __init__(self, config: DocumentWorkerConfig, filter_names: list[str], def _check_filters(self): for name in self.filter_names: - if not (self.FILTERS_PATH / name).is_file(): + if self.settings.filter_path(name) is None: raise RuntimeError(f'Pandoc filter "{name}" not found') def _check_template(self): - if self.template_name and not (self.TEMPLATES_PATH / self.template_name).is_file(): + if self.template_name and self.settings.template_path(self.template_name) is None: raise RuntimeError(f'Pandoc template "{self.template_name}" not found') - def _extra_args(self): - args = [] + def _extra_args(self) -> list[str]: + # paths are passed as they are (not re-split), they may contain spaces + args: list[str] = [] if self.template_name: - args.extend(['--template', str(self.TEMPLATES_PATH / self.template_name)]) + args.extend(['--template', str(self.settings.template_path(self.template_name))]) for filter_name in self.filter_names: - if filter_name.endswith('.lua'): - args.extend(['--lua-filter', str(self.FILTERS_PATH / filter_name)]) - else: - args.extend(['--filter', str(self.FILTERS_PATH / filter_name)]) - return shlex.split(' '.join(args)) + option = '--lua-filter' if filter_name.endswith('.lua') else '--filter' + args.extend([option, str(self.settings.filter_path(filter_name))]) + return args def __call__(self, *, source_format: FileFormat, target_format: FileFormat, data: bytes, metadata: dict, workdir: str) -> bytes: args = ['-f', source_format.name, '-t', target_format.name, '-o', '-'] template_args = self.extract_template_args(metadata) extra_args = self._extra_args() - command = self.config.pandoc.command + template_args + extra_args + args + command = self.settings.command + template_args + extra_args + args return run_conversion( args=command, workdir=workdir, @@ -96,7 +97,7 @@ def __call__(self, *, source_format: FileFormat, target_format: FileFormat, name=type(self).__name__, source_format=source_format, target_format=target_format, - timeout=self.config.pandoc.timeout, + timeout=self.settings.timeout, ) @staticmethod @@ -115,12 +116,12 @@ class RdfLibConvert: FileFormats.JSONLD: 'json-ld', } - def __init__(self, config: DocumentWorkerConfig): - self.config = config + def __init__(self): + self.rdflib = require('rdflib', 'rdf') def __call__(self, *, source_format: FileFormat, target_format: FileFormat, data: bytes, metadata: dict) -> bytes: - g = rdflib.Dataset() + g = self.rdflib.Dataset() g.parse( data=data.decode(consts.DEFAULT_ENCODING), format=self.FORMATS.get(source_format) or 'turtle', diff --git a/packages/dsw-templating/dsw/templating/documents.py b/packages/dsw-templating/dsw/templating/documents.py new file mode 100644 index 00000000..6a298b7a --- /dev/null +++ b/packages/dsw-templating/dsw/templating/documents.py @@ -0,0 +1,147 @@ +from __future__ import annotations + +import pathlib + +from . import consts + + +class FileFormat: + + def __init__(self, name: str, content_type: str, file_extension: str): + self.name = name + self.content_type = content_type + self.file_extension = file_extension + + def __eq__(self, other): + return isinstance(other, FileFormat) and other.name == self.name + + def __hash__(self): + return hash(self.name) + + def __str__(self): + return self.name + + def __repr__(self): + return f'Format[{self.name}]' + + +class FileFormats: + JSON = FileFormat('json', 'application/json', 'json') + HTML = FileFormat('html', 'text/html', 'html') + PDF = FileFormat('pdf', 'application/pdf', 'pdf') + DOCX = FileFormat( + 'docx', + 'application/vnd.openxmlformats-officedocument.wordprocessingml.document', + 'docx', + ) + Markdown = FileFormat('markdown', 'text/markdown', 'md') + ODT = FileFormat('odt', 'application/vnd.oasis.opendocument.text', 'odt') + RST = FileFormat('rst', 'text/x-rst', 'rst') + LaTeX = FileFormat('latex', 'application/x-tex', 'tex') + EPUB = FileFormat('epub', 'application/epub+zip', 'epub') + DocBook4 = FileFormat('docbook4', 'application/docbook+xml', 'dbk') + DocBook5 = FileFormat('docbook5', 'application/docbook+xml', 'dbk') + PPTX = FileFormat( + 'pptx', + 'application/vnd.openxmlformats-officedocument.presentationml.presentation', + 'pptx', + ) + RTF = FileFormat('rtf', 'application/rtf', 'rtf') + ADoc = FileFormat('asciidoc', 'text/asciidoc', 'adoc') + RDF_XML = FileFormat('rdf', 'application/rdf+xml', 'rdf') + N3 = FileFormat('n3', 'text/n3', 'n3') + NTRIPLES = FileFormat('nt', 'application/n-triples', 'nt') + TURTLE = FileFormat('ttl', 'text/turtle', 'ttl') + TRIG = FileFormat('trig', 'application/trig', 'trig') + JSONLD = FileFormat('jsonld', 'application/ld+json', 'jsonld') + ZIP = FileFormat('zip', 'application/zip', 'zip') + TAR = FileFormat('tar', 'application/x-tar', 'tar') + TAR_GZIP = FileFormat('gzip', 'application/gzip', 'tar.gz') + TAR_BZIP2 = FileFormat('bzip2', 'application/x-bzip2', 'tar.bz2') + TAR_LZMA = FileFormat('lzma', 'application/x-lzma', 'tar.xz') + XLSX = FileFormat( + 'xlsx', + 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet', + 'xlsx', + ) + XLSM = FileFormat( + 'xlsm', + 'application/vnd.ms-excel.sheet.macroEnabled.12', + 'xlsm', + ) + + @staticmethod + def get(name: str): + known_formats = { + 'html': FileFormats.HTML, + 'pdf': FileFormats.PDF, + 'docx': FileFormats.DOCX, + 'markdown': FileFormats.Markdown, + 'odt': FileFormats.ODT, + 'rst': FileFormats.RST, + 'latex': FileFormats.LaTeX, + 'json': FileFormats.JSON, + 'epub': FileFormats.EPUB, + 'docbook4': FileFormats.DocBook4, + 'docbook5': FileFormats.DocBook5, + 'pptx': FileFormats.PPTX, + 'rtf': FileFormats.RTF, + 'asciidoc': FileFormats.ADoc, + 'rdf': FileFormats.RDF_XML, + 'rdf/xml': FileFormats.RDF_XML, + 'turtle': FileFormats.TURTLE, + 'ttl': FileFormats.TURTLE, + 'n3': FileFormats.N3, + 'ntriples': FileFormats.NTRIPLES, + 'n-triples': FileFormats.NTRIPLES, + 'trig': FileFormats.TRIG, + 'json-ld': FileFormats.JSONLD, + 'jsonld': FileFormats.JSONLD, + 'zip': FileFormats.ZIP, + 'tar': FileFormats.TAR, + 'gzip': FileFormats.TAR_GZIP, + 'bzip2': FileFormats.TAR_BZIP2, + 'lzma': FileFormats.TAR_LZMA, + 'xlsx': FileFormats.XLSX, + 'xlsm': FileFormats.XLSM, + } + return known_formats.get(name) + + +class DocumentFile: + + def __init__(self, file_format: FileFormat, content: bytes, + encoding: str | None = None): + self.file_format = file_format + self._content = content + self.byte_size = len(content) + self.encoding = encoding + + @property + def content_type(self) -> str: + return self.file_format.content_type + + @property + def safe_encoding(self) -> str: + return self.encoding or consts.DEFAULT_ENCODING + + @property + def content(self) -> bytes: + return self._content + + @content.setter + def content(self, content: bytes): + self._content = content + self.byte_size = len(content) + + def filename(self, name: str) -> str: + return f'{name}.{self.file_format.file_extension}' + + def store(self, name: str): + pathlib.Path(self.filename(name)).write_bytes(self.content) + + @property + def object_content_type(self) -> str: + if self.encoding is not None: + return f'{self.content_type}; charset={self.encoding}' + return self.content_type diff --git a/packages/dsw-templating/dsw/templating/docx_pagebreak.py b/packages/dsw-templating/dsw/templating/docx_pagebreak.py new file mode 100644 index 00000000..e116e515 --- /dev/null +++ b/packages/dsw-templating/dsw/templating/docx_pagebreak.py @@ -0,0 +1,91 @@ +"""Pandoc filter `pandoc-docx-pagebreakpy`: page breaks and TOC as OpenXML raw blocks (DOCX only). + +`\\newpage` becomes a page break and `\\toc` (`\\toc1` to `\\toc6` for the depth) a table of +contents. Installed as the `pandoc-docx-pagebreakpy` command with the `docx` extra, for +`--filter=pandoc-docx-pagebreakpy` in the `args` of the `pandoc` step. Deprecated in favour of the +`docx-pagebreak.lua` and `docx-toc.lua` filters. + +Based on https://github.com/pandocker/pandoc-docx-pagebreak-py (formerly the +`pandoc-docx-pagebreak` addon of the document worker): +""" +# MIT License +# +# Copyright (c) 2018 pandocker +# +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. +from __future__ import annotations + +import sys + +from .exceptions import MissingExtraError, require + + +TOC_DEPTHS = {r'\toc1': 1, r'\toc2': 2, r'\toc4': 4, r'\toc5': 5, r'\toc6': 6} +DEFAULT_TOC_DEPTH = 3 + + +class DocxPagebreak: + + def __init__(self): + self.pf = require('panflute', 'docx') + + def _make_pagebreak(self): + return self.pf.RawBlock('', format='openxml') + + def _make_toc(self, instr: str): + depth = TOC_DEPTHS.get(instr, DEFAULT_TOC_DEPTH) + toc_lines = [ + r'', + r'', + r'', + r'', + rf'TOC \o "1-{depth}" \h \z \u', + r'', + r'', + r'', + r'', + r'', + ] + return self.pf.RawBlock('\n'.join(toc_lines), format='openxml') + + def action(self, elem, doc): + pf = self.pf + if doc.format != 'docx': + return elem + if isinstance(elem, (pf.Para, pf.Plain)): + for child in elem.content: + if isinstance(child, pf.Str) and child.text == r'\newpage': + elem = self._make_pagebreak() + elif isinstance(child, pf.Str) and child.text.startswith(r'\toc'): + elem = self._make_toc(child.text) + if isinstance(elem, pf.RawBlock): + if elem.text == r'\newpage': + elem = self._make_pagebreak() + elif elem.text.startswith(r'\toc'): + elem = self._make_toc(elem.text) + return elem + + +def main(doc=None): + try: + dp = DocxPagebreak() + except MissingExtraError as e: + # run by pandoc: a message on stderr rather than a traceback + sys.exit(f'pandoc-docx-pagebreakpy: {e}') + return dp.pf.run_filter(dp.action, doc=doc) diff --git a/packages/dsw-templating/dsw/templating/exceptions.py b/packages/dsw-templating/dsw/templating/exceptions.py new file mode 100644 index 00000000..d10ca0a3 --- /dev/null +++ b/packages/dsw-templating/dsw/templating/exceptions.py @@ -0,0 +1,71 @@ +from __future__ import annotations + +import importlib +import typing + + +if typing.TYPE_CHECKING: + import types + + +class TemplateError(Exception): + + def __init__(self, template_uuid: str, message: str): + self.template_uuid = template_uuid + self.message = message + + def __str__(self): + return f'Error in template "{self.template_uuid}"\n' \ + f'- {self.message}' + + +class TemplateTriggeredError(Exception): + """Error invoked from a template to report a problem to a user (not system).""" + + def __init__(self, title, message): + self.title = title + self.message = message + self.msg = f'{title}\n\n{message}' + super().__init__(self.msg) + + def __str__(self): + return self.msg + + +class MissingExtraError(ImportError): + """An optional dependency of a step is not installed.""" + + def __init__(self, module: str, extra: str): + self.module = module + self.extra = extra + super().__init__( + f'Module "{module}" is not installed, ' + f'install dsw-templating[{extra}] to use it', + name=module, + ) + + +def require(module: str, extra: str) -> types.ModuleType: + """Import an optional dependency, failing with the extra to install.""" + try: + return importlib.import_module(module) + except ImportError as e: + raise MissingExtraError(module, extra) from e + + +class MissingModule: + """Stands in for an optional module; fails as soon as it is used.""" + + def __init__(self, module: str, extra: str): + self._module = module + self._extra = extra + + def __getattr__(self, name: str): + raise MissingExtraError(self._module, self._extra) + + +def optional(module: str, extra: str) -> types.ModuleType | MissingModule: + try: + return importlib.import_module(module) + except ImportError: + return MissingModule(module, extra) diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/extraction.py b/packages/dsw-templating/dsw/templating/extraction.py similarity index 100% rename from packages/dsw-document-worker/dsw/document_worker/templates/extraction.py rename to packages/dsw-templating/dsw/templating/extraction.py diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/filters.py b/packages/dsw-templating/dsw/templating/filters.py similarity index 93% rename from packages/dsw-document-worker/dsw/document_worker/templates/filters.py rename to packages/dsw-templating/dsw/templating/filters.py index c56433b2..7d03d1d6 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/filters.py +++ b/packages/dsw-templating/dsw/templating/filters.py @@ -11,10 +11,10 @@ from dsw.models.document_context.graph import DocumentContext from dsw.models.document_context.rendering import render_markdown -from ..exceptions import JobError -from ..utils import JinjaEnvironment, byte_size_format +from .exceptions import TemplateTriggeredError from .extraction import extract_replies from .tests import tests +from .utils import JinjaEnvironment, byte_size_format LOG = logging.getLogger(__name__) @@ -182,18 +182,6 @@ def to_context_obj(ctx, **options) -> DocumentContext: return result -class TemplateTriggeredError(JobError): - """Error invoked from a template to report a problem to a user (not system).""" - - def __init__(self, title, message): - super().__init__( - job_id='', - msg=f'{title}\n\n{message}', - exc=None, - skip_reporting=True, - ) - - def raise_error(message, title='Document rendering error'): raise TemplateTriggeredError( title=title, diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/formats.py b/packages/dsw-templating/dsw/templating/formats.py similarity index 95% rename from packages/dsw-document-worker/dsw/document_worker/templates/formats.py rename to packages/dsw-templating/dsw/templating/formats.py index ea8cb6b8..0234624c 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/formats.py +++ b/packages/dsw-templating/dsw/templating/formats.py @@ -3,12 +3,12 @@ import logging import typing -from .. import consts -from ..templates.steps import FormatStepError, Step, create_step +from . import consts +from .steps import FormatStepError, Step, create_step if typing.TYPE_CHECKING: - from ..documents import DocumentFile + from .documents import DocumentFile LOG = logging.getLogger(__name__) diff --git a/packages/dsw-document-worker/dsw/document_worker/http.py b/packages/dsw-templating/dsw/templating/http.py similarity index 89% rename from packages/dsw-document-worker/dsw/document_worker/http.py rename to packages/dsw-templating/dsw/templating/http.py index 66744ef5..f229d300 100644 --- a/packages/dsw-document-worker/dsw/document_worker/http.py +++ b/packages/dsw-templating/dsw/templating/http.py @@ -4,11 +4,13 @@ import typing import urllib.parse -import requests +from .exceptions import require if typing.TYPE_CHECKING: - from .config import TemplateConfig + import requests + + from .settings import RequestsSettings from .urls import UrlPolicy @@ -30,9 +32,11 @@ class RequestsWrapper: or otherwise private addresses (e.g. cloud metadata endpoints). """ - def __init__(self, template_cfg: TemplateConfig, policy: UrlPolicy): - self.limit = template_cfg.requests.limit - self.timeout = template_cfg.requests.timeout + def __init__(self, settings: RequestsSettings, policy: UrlPolicy): + # private: the Jinja sandbox must not hand the raw module to templates + self._requests = require('requests', 'http') + self.limit = settings.limit + self.timeout = settings.timeout self.policy = policy self.request_counter = 0 @@ -71,7 +75,7 @@ def _send(self, method: str, url: str, **kwargs) -> requests.Response: while True: self._prepare_for_request() self.policy.check_http_url(url) - response = requests.request( + response = self._requests.request( method=method, url=url, timeout=self.timeout, diff --git a/packages/dsw-templating/dsw/templating/locales.py b/packages/dsw-templating/dsw/templating/locales.py new file mode 100644 index 00000000..8a1185e1 --- /dev/null +++ b/packages/dsw-templating/dsw/templating/locales.py @@ -0,0 +1,47 @@ +from __future__ import annotations + +import dataclasses +import gettext +import logging +import uuid + + +LOG = logging.getLogger(__name__) + + +@dataclasses.dataclass(frozen=True) +class TemplateLocale: + uuid: str + name: str + code: str + updated_at: str + + @staticmethod + def load(data: dict | None) -> TemplateLocale | None: + if not isinstance(data, dict): + return None + try: + locale_uuid = str(uuid.UUID(str(data['uuid']))) + except (KeyError, ValueError): + LOG.warning('Ignoring locale without a valid UUID') + return None + return TemplateLocale( + uuid=locale_uuid, + name=str(data.get('name', '')), + code=str(data.get('code', '')), + updated_at=str(data.get('updatedAt', '')), + ) + + +@dataclasses.dataclass +class RenderContext: + translations: gettext.NullTranslations + language: str | None = None + locale: TemplateLocale | None = None + + @staticmethod + def null(language: str | None = None) -> RenderContext: + return RenderContext( + translations=gettext.NullTranslations(), + language=language, + ) diff --git a/packages/dsw-templating/dsw/templating/plugins/__init__.py b/packages/dsw-templating/dsw/templating/plugins/__init__.py new file mode 100644 index 00000000..55550886 --- /dev/null +++ b/packages/dsw-templating/dsw/templating/plugins/__init__.py @@ -0,0 +1,5 @@ +from .manager import create_manager, register_plugin_steps +from .specs import hookimpl, hookspec + + +__all__ = ['create_manager', 'hookimpl', 'hookspec', 'register_plugin_steps'] diff --git a/packages/dsw-templating/dsw/templating/plugins/manager.py b/packages/dsw-templating/dsw/templating/plugins/manager.py new file mode 100644 index 00000000..5e5be93c --- /dev/null +++ b/packages/dsw-templating/dsw/templating/plugins/manager.py @@ -0,0 +1,28 @@ +from __future__ import annotations + +import typing + +import pluggy + +from .. import consts +from ..steps.base import Step, register_step + + +def create_manager() -> pluggy.PluginManager: + """Plugin manager with every plugin installed under the entry point.""" + from . import specs as hookspecs + + pm = pluggy.PluginManager(consts.PACKAGE_NAME) + pm.load_setuptools_entrypoints(consts.PLUGINS_ENTRYPOINT) + pm.add_hookspecs(hookspecs) + return pm + + +def register_plugin_steps(pm: pluggy.PluginManager): + """Register the steps provided by plugins, alongside the built-in ones.""" + steps_dicts: typing.Iterable[dict[str, type[Step]]] = pm.hook.provide_steps() + for steps_dict in steps_dicts: + for name, step_class in steps_dict.items(): + if not issubclass(step_class, Step): + raise RuntimeError(f'Provided class "{step_class}" is not a subclass of Step') + register_step(name, step_class) diff --git a/packages/dsw-document-worker/dsw/document_worker/plugins/specs.py b/packages/dsw-templating/dsw/templating/plugins/specs.py similarity index 93% rename from packages/dsw-document-worker/dsw/document_worker/plugins/specs.py rename to packages/dsw-templating/dsw/templating/plugins/specs.py index e4843c23..194a8fc7 100644 --- a/packages/dsw-document-worker/dsw/document_worker/plugins/specs.py +++ b/packages/dsw-templating/dsw/templating/plugins/specs.py @@ -10,7 +10,7 @@ if typing.TYPE_CHECKING: from jinja2 import Environment - from ..templates.steps import Step + from ..steps import Step hookspec = pluggy.HookspecMarker(consts.PACKAGE_NAME) @@ -25,8 +25,9 @@ def provide_steps() -> dict[str, type[Step]]: Steps are used to process the document generation in a specific way. Each step must comply with the interface of the provided `Step` class. The plugin must make sure to return a list of steps that it can execute and in compliance with the - `Step` class in the current implementation (use correct dsw-document-worker as - a dependency). + `Step` class in the current implementation (use correct dsw-templating as + a dependency). Plugins are loaded from the `dsw_templating_plugins` entry point + and their steps are registered by `register_plugin_steps`. Before the format pipeline runs, every step gets `before_render(render_ctx)` called with the `RenderContext` of the document. The base implementation stores it, so a diff --git a/packages/dsw-templating/dsw/templating/pot.py b/packages/dsw-templating/dsw/templating/pot.py new file mode 100644 index 00000000..2a136283 --- /dev/null +++ b/packages/dsw-templating/dsw/templating/pot.py @@ -0,0 +1,109 @@ +"""Extraction of translatable messages from Jinja templates into a POT file.""" +from __future__ import annotations + +import dataclasses +import io +import logging +import typing + +import babel +import jinja2.exceptions +import jinja2.ext +from babel.messages.catalog import Catalog +from babel.messages.extract import DEFAULT_KEYWORDS, extract +from babel.messages.pofile import write_po + +from . import consts + + +if typing.TYPE_CHECKING: + from collections.abc import Iterable + + +LOG = logging.getLogger(__name__) + +COMMENT_TAGS = ('TRANSLATORS:',) +EXTRACT_METHOD = typing.cast('typing.Any', jinja2.ext.babel_extract) +NO_WRAP = 0 +EXTRACT_OPTIONS = { + 'encoding': consts.DEFAULT_ENCODING, + 'extensions': ','.join(consts.JINJA_EXTENSIONS), + 'silent': 'false', + 'newstyle_gettext': 'true', + 'trimmed': str(consts.JINJA_I18N_TRIMMED).lower(), +} + + +class SourceFile(typing.Protocol): + """A template file: a path relative to the template root and its text.""" + + @property + def file_name(self) -> str: ... + + @property + def content(self) -> str: ... + + +@dataclasses.dataclass +class ExtractionResult: + catalog: Catalog + failed_files: list[str] + + +def extract_messages(content: str) -> list[tuple]: + return list(extract( + method=EXTRACT_METHOD, + fileobj=io.BytesIO(content.encode(consts.DEFAULT_ENCODING)), + keywords=DEFAULT_KEYWORDS, + comment_tags=COMMENT_TAGS, + options=EXTRACT_OPTIONS, + )) + + +def make_catalog(*, project: str, version: str, language: str) -> Catalog: + locale: babel.Locale | None = None + try: + locale = babel.Locale.parse(language.replace('-', '_')) + except (ValueError, babel.UnknownLocaleError): + LOG.warning('Cannot parse language "%s" - POT file without locale info', language) + return Catalog( + locale=locale, + domain=consts.DEFAULT_LOCALE_DOMAIN, + project=project, + version=version, + charset=consts.DEFAULT_ENCODING, + fuzzy=False, + ) + + +def extract_catalog(files: Iterable[SourceFile], *, project: str, + version: str, language: str) -> ExtractionResult: + catalog = make_catalog(project=project, version=version, language=language) + failed_files = [] + for file in sorted(files, key=lambda f: f.file_name): + try: + messages = extract_messages(file.content) + except jinja2.exceptions.TemplateSyntaxError as e: + LOG.warning('Skipping file "%s" that cannot be parsed: %s', file.file_name, str(e)) + failed_files.append(file.file_name) + continue + for lineno, message, comments, context in messages: + catalog.add( + message, + None, + [(file.file_name, lineno)], + auto_comments=comments, + context=context, + ) + return ExtractionResult(catalog=catalog, failed_files=failed_files) + + +def render_pot_file(result: ExtractionResult) -> bytes: + lines = [f'# Translations template for {result.catalog.project}.'] + if result.failed_files: + lines.append(f'# Skipped files that could not be parsed: ' + f'{", ".join(result.failed_files)}') + result.catalog.header_comment = '\n'.join(lines) + '\n' + buffer = io.BytesIO() + write_po(buffer, result.catalog, width=NO_WRAP, omit_header=False, sort_output=True) + return buffer.getvalue() diff --git a/packages/dsw-templating/dsw/templating/py.typed b/packages/dsw-templating/dsw/templating/py.typed new file mode 100644 index 00000000..e69de29b diff --git a/packages/dsw-document-worker/resources/pandoc/filters/docx-landscape.lua b/packages/dsw-templating/dsw/templating/resources/pandoc/filters/docx-landscape.lua similarity index 100% rename from packages/dsw-document-worker/resources/pandoc/filters/docx-landscape.lua rename to packages/dsw-templating/dsw/templating/resources/pandoc/filters/docx-landscape.lua diff --git a/packages/dsw-document-worker/resources/pandoc/filters/docx-pagebreak.lua b/packages/dsw-templating/dsw/templating/resources/pandoc/filters/docx-pagebreak.lua similarity index 100% rename from packages/dsw-document-worker/resources/pandoc/filters/docx-pagebreak.lua rename to packages/dsw-templating/dsw/templating/resources/pandoc/filters/docx-pagebreak.lua diff --git a/packages/dsw-document-worker/resources/pandoc/filters/docx-toc.lua b/packages/dsw-templating/dsw/templating/resources/pandoc/filters/docx-toc.lua similarity index 100% rename from packages/dsw-document-worker/resources/pandoc/filters/docx-toc.lua rename to packages/dsw-templating/dsw/templating/resources/pandoc/filters/docx-toc.lua diff --git a/packages/dsw-templating/dsw/templating/settings.py b/packages/dsw-templating/dsw/templating/settings.py new file mode 100644 index 00000000..aa3f45d2 --- /dev/null +++ b/packages/dsw-templating/dsw/templating/settings.py @@ -0,0 +1,72 @@ +"""Everything a rendering needs to know about its environment. + +The caller (e.g. the document worker) maps its own configuration onto these; +the library never reads configuration files or environment variables itself. +""" +from __future__ import annotations + +import dataclasses +import importlib.resources +import pathlib + + +def bundled_pandoc_filters() -> pathlib.Path: + """Directory with the Pandoc filters shipped with this package.""" + resource = importlib.resources.files('dsw.templating') / 'resources' / 'pandoc' / 'filters' + return pathlib.Path(str(resource)) + + +@dataclasses.dataclass +class SecuritySettings: + """Which URLs may be fetched while rendering (see `UrlPolicy`).""" + allow_external_resources: bool = True + allow_private_network: bool = False + allowed_hosts: list[str] = dataclasses.field(default_factory=list) + allowed_paths: list[str] = dataclasses.field(default_factory=list) + max_redirects: int = 3 + + +@dataclasses.dataclass +class PandocSettings: + command: list[str] = dataclasses.field(default_factory=lambda: ['pandoc', '--standalone']) + timeout: float | None = None + #: searched in order for a filter name, the first directory having it wins + filter_dirs: list[pathlib.Path] = dataclasses.field( + default_factory=lambda: [bundled_pandoc_filters()], + ) + templates_dir: pathlib.Path | None = None + + def filter_path(self, name: str) -> pathlib.Path | None: + for filter_dir in self.filter_dirs: + path = filter_dir / name + if path.is_file(): + return path + return None + + def template_path(self, name: str) -> pathlib.Path | None: + if self.templates_dir is None: + return None + path = self.templates_dir / name + return path if path.is_file() else None + + +@dataclasses.dataclass +class RequestsSettings: + """HTTP requests from Jinja templates (the `requests` global).""" + enabled: bool = False + limit: int = 100 + timeout: int = 1 + + +@dataclasses.dataclass +class TemplateSettings: + """Per-template settings; templates without them get no `secrets` global.""" + secrets: dict[str, str] = dataclasses.field(default_factory=dict) + requests: RequestsSettings = dataclasses.field(default_factory=RequestsSettings) + + +@dataclasses.dataclass +class RenderSettings: + security: SecuritySettings = dataclasses.field(default_factory=SecuritySettings) + pandoc: PandocSettings = dataclasses.field(default_factory=PandocSettings) + template: TemplateSettings | None = None diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/steps/__init__.py b/packages/dsw-templating/dsw/templating/steps/__init__.py similarity index 100% rename from packages/dsw-document-worker/dsw/document_worker/templates/steps/__init__.py rename to packages/dsw-templating/dsw/templating/steps/__init__.py diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/steps/archive.py b/packages/dsw-templating/dsw/templating/steps/archive.py similarity index 99% rename from packages/dsw-document-worker/dsw/document_worker/templates/steps/archive.py rename to packages/dsw-templating/dsw/templating/steps/archive.py index 3508359d..2f88d54c 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/steps/archive.py +++ b/packages/dsw-templating/dsw/templating/steps/archive.py @@ -5,7 +5,7 @@ import tempfile import zipfile -from ...documents import DocumentFile, FileFormats +from ..documents import DocumentFile, FileFormats from .base import Step, register_step diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/steps/base.py b/packages/dsw-templating/dsw/templating/steps/base.py similarity index 98% rename from packages/dsw-document-worker/dsw/document_worker/templates/steps/base.py rename to packages/dsw-templating/dsw/templating/steps/base.py index b5f34389..57c2269b 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/steps/base.py +++ b/packages/dsw-templating/dsw/templating/steps/base.py @@ -8,7 +8,7 @@ if typing.TYPE_CHECKING: from gettext import NullTranslations - from ...documents import DocumentFile + from ..documents import DocumentFile class FormatStepError(Exception): diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/steps/conversion.py b/packages/dsw-templating/dsw/templating/steps/conversion.py similarity index 80% rename from packages/dsw-document-worker/dsw/document_worker/templates/steps/conversion.py rename to packages/dsw-templating/dsw/templating/steps/conversion.py index 4d49fb75..27bbc11d 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/steps/conversion.py +++ b/packages/dsw-templating/dsw/templating/steps/conversion.py @@ -1,16 +1,14 @@ from __future__ import annotations +import functools import logging import typing -import weasyprint -import weasyprint.urls - -from ... import consts -from ...context import Context -from ...conversions import Pandoc, RdfLibConvert -from ...documents import DocumentFile, FileFormats -from ...urls import UrlNotAllowedError, UrlPolicy +from .. import consts +from ..conversions import Pandoc, RdfLibConvert +from ..documents import DocumentFile, FileFormats +from ..exceptions import require +from ..urls import UrlNotAllowedError, UrlPolicy from .base import Step, register_step @@ -25,27 +23,34 @@ def _is_true(value: str) -> bool: return value.lower() == 'true' -class RestrictedURLFetcher(weasyprint.urls.URLFetcher): - """URL fetcher that only retrieves resources allowed by the policy. +@functools.cache +def restricted_url_fetcher_class() -> type: + """The fetcher class, built on first use as WeasyPrint is an optional dependency.""" + weasyprint_urls = require('weasyprint.urls', 'pdf') + + class RestrictedURLFetcher(weasyprint_urls.URLFetcher): + """URL fetcher that only retrieves resources allowed by the policy. + + Without it, WeasyPrint resolves any reference in the rendered HTML, + including ``file://`` URLs (local file disclosure) and internal http(s) + addresses such as cloud metadata endpoints (SSRF). Note that redirects + are checked as well, since they are dispatched back through ``fetch``. + """ - Without it, WeasyPrint resolves any reference in the rendered HTML, - including ``file://`` URLs (local file disclosure) and internal http(s) - addresses such as cloud metadata endpoints (SSRF). Note that redirects - are checked as well, since they are dispatched back through ``fetch``. - """ + def __init__(self, *, policy: UrlPolicy, base_dir: pathlib.Path, **kwargs): + super().__init__(allow_redirects=policy.max_redirects > 0, **kwargs) + self.policy = policy + self.base_dir = base_dir.expanduser().resolve() - def __init__(self, *, policy: UrlPolicy, base_dir: pathlib.Path, **kwargs): - super().__init__(allow_redirects=policy.max_redirects > 0, **kwargs) - self.policy = policy - self.base_dir = base_dir.expanduser().resolve() + def fetch(self, url, headers=None): + try: + self.policy.check_resource_url(url, base_dir=self.base_dir) + except UrlNotAllowedError as e: + LOG.warning('Blocked resource while rendering document: %s', e) + raise + return super().fetch(url, headers) - def fetch(self, url, headers=None): - try: - self.policy.check_resource_url(url, base_dir=self.base_dir) - except UrlNotAllowedError as e: - LOG.warning('Blocked resource while rendering document: %s', e) - raise - return super().fetch(url, headers) + return RestrictedURLFetcher class WeasyPrintStep(Step): @@ -55,8 +60,10 @@ class WeasyPrintStep(Step): def __init__(self, template, options: dict): super().__init__(template, options) + self.weasyprint = require('weasyprint', 'pdf') # PDF options - self.wp_options: typing.MutableMapping[str, typing.Any] = weasyprint.DEFAULT_OPTIONS.copy() + self.wp_options: typing.MutableMapping[str, typing.Any] = \ + self.weasyprint.DEFAULT_OPTIONS.copy() self.wp_update_options(options) self.wp_zoom = float(options.get('pdf.zoom', '1')) @@ -92,12 +99,12 @@ def execute_follow(self, document: DocumentFile, context: dict) -> DocumentFile: self.raise_exc(f'WeasyPrint does not support {document.file_format.name}' f' format as input') file_uri = self.template.template_dir / '_file.html' - wp_html = weasyprint.HTML( + wp_html = self.weasyprint.HTML( string=document.content.decode(consts.DEFAULT_ENCODING), media_type='print', base_url=file_uri.as_uri(), - url_fetcher=RestrictedURLFetcher( - policy=UrlPolicy(Context.get().app.cfg.security), + url_fetcher=restricted_url_fetcher_class()( + policy=UrlPolicy(self.template.settings.security), base_dir=self.template.template_dir, ), ) @@ -166,7 +173,7 @@ def __init__(self, template, options: dict): if self.output_format not in self.OUTPUT_FORMATS: self.raise_exc(f'Unknown output format "{self.output_format.name}"') self.pandoc = Pandoc( - config=Context.get().app.cfg, + settings=self.template.settings.pandoc, filter_names=self._extract_filter_names( filters=options.get(self.OPTION_FILTERS, ''), ), @@ -221,7 +228,7 @@ class RdfLibConvertStep(Step): def __init__(self, template, options: dict): super().__init__(template, options) - self.rdflib_convert = RdfLibConvert(config=Context.get().app.cfg) + self.rdflib_convert = RdfLibConvert() self.input_format = FileFormats.get(options[self.OPTION_FROM]) self.output_format = FileFormats.get(options[self.OPTION_TO]) if self.input_format not in self.INPUT_FORMATS: diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/steps/excel.py b/packages/dsw-templating/dsw/templating/steps/excel.py similarity index 99% rename from packages/dsw-document-worker/dsw/document_worker/templates/steps/excel.py rename to packages/dsw-templating/dsw/templating/steps/excel.py index a1db4fdd..adb28067 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/steps/excel.py +++ b/packages/dsw-templating/dsw/templating/steps/excel.py @@ -9,13 +9,14 @@ import typing import dateutil.parser -import xlsxwriter -from ...documents import DocumentFile, FileFormats +from ..documents import DocumentFile, FileFormats +from ..exceptions import require from .base import FormatStepError, Step, register_step if typing.TYPE_CHECKING: + import xlsxwriter from xlsxwriter.chart import Chart from xlsxwriter.worksheet import Worksheet @@ -890,6 +891,7 @@ def cleanup(self): def build_to_bytes(tmp_file: pathlib.Path, input_data: dict) -> bytes: options = input_data.get('options') tmp_file.unlink(missing_ok=True) + xlsxwriter = require('xlsxwriter', 'excel') with xlsxwriter.Workbook(str(tmp_file), options) as workbook: builder = WorkbookBuilder(workbook=workbook) builder.build(data=input_data) @@ -909,6 +911,7 @@ class ExcelStep(Step): def __init__(self, template, options: dict): super().__init__(template, options) + require('xlsxwriter', 'excel') def execute_first(self, context: dict) -> DocumentFile: return self.raise_exc(f'Step "{self.NAME}" cannot be first') diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/steps/template.py b/packages/dsw-templating/dsw/templating/steps/template.py similarity index 92% rename from packages/dsw-document-worker/dsw/document_worker/templates/steps/template.py rename to packages/dsw-templating/dsw/templating/steps/template.py index 59849a7b..07db4250 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/steps/template.py +++ b/packages/dsw-templating/dsw/templating/steps/template.py @@ -7,18 +7,17 @@ import jinja2 import jinja2.exceptions -import rdflib from dsw.models.document_context.graph import ProjectFile -from ...consts import DEFAULT_ENCODING, JINJA_EXTENSIONS, JINJA_I18N_TRIMMED -from ...context import Context -from ...documents import DocumentFile, FileFormat, FileFormats -from ...http import RequestsWrapper -from ...urls import UrlPolicy -from ...utils import JinjaEnvironment +from ..consts import DEFAULT_ENCODING, JINJA_EXTENSIONS, JINJA_I18N_TRIMMED +from ..documents import DocumentFile, FileFormat, FileFormats +from ..exceptions import optional from ..filters import filters +from ..http import RequestsWrapper from ..tests import tests +from ..urls import UrlPolicy +from ..utils import JinjaEnvironment from .base import Step, register_step @@ -67,7 +66,7 @@ def __init__(self, template, options): self._add_j2_enhancements() self._install_translations(None) - Context.get().app.pm.hook.enrich_jinja_env( + self.template.plugins.hook.enrich_jinja_env( jinja_env=self.j2_env, options=options, ) @@ -152,12 +151,10 @@ def _j2_globals(self) -> typing.MutableMapping[str, typing.Any]: def _add_j2_enhancements(self): self._j2_filters.update(filters) self._j2_tests.update(tests) - app_cfg = Context.get().app.cfg - template_cfg = app_cfg.templates.get_config( - self.template.coordinates, - ) + settings = self.template.settings + template_cfg = settings.template self._j2_globals.update({ - 'rdflib': rdflib, + 'rdflib': optional('rdflib', 'rdf'), 'json': json, 'datetime': datetime, 'zoneinfo': zoneinfo, @@ -166,8 +163,8 @@ def _add_j2_enhancements(self): global_vars: dict[str, typing.Any] = {'secrets': template_cfg.secrets} if template_cfg.requests.enabled: global_vars['requests'] = RequestsWrapper( - template_cfg=template_cfg, - policy=UrlPolicy(app_cfg.security), + settings=template_cfg.requests, + policy=UrlPolicy(settings.security), ) self.j2_env.globals.update(global_vars) diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/steps/word.py b/packages/dsw-templating/dsw/templating/steps/word.py similarity index 97% rename from packages/dsw-document-worker/dsw/document_worker/templates/steps/word.py rename to packages/dsw-templating/dsw/templating/steps/word.py index 4be0d5a4..0e4d1568 100644 --- a/packages/dsw-document-worker/dsw/document_worker/templates/steps/word.py +++ b/packages/dsw-templating/dsw/templating/steps/word.py @@ -7,8 +7,8 @@ import jinja2 -from ...consts import DEFAULT_ENCODING -from ...documents import DocumentFile, FileFormats +from ..consts import DEFAULT_ENCODING +from ..documents import DocumentFile, FileFormats from .base import register_step from .template import JinjaPoweredStep diff --git a/packages/dsw-templating/dsw/templating/template.py b/packages/dsw-templating/dsw/templating/template.py new file mode 100644 index 00000000..e102a8d8 --- /dev/null +++ b/packages/dsw-templating/dsw/templating/template.py @@ -0,0 +1,250 @@ +from __future__ import annotations + +import base64 +import json +import logging +import mimetypes +import typing +import uuid + +from . import consts +from .exceptions import TemplateError +from .formats import Format +from .locales import RenderContext +from .settings import RenderSettings + + +if typing.TYPE_CHECKING: + import pathlib + from collections.abc import Iterable + + from pluggy import PluginManager + + from dsw.models.document_context.graph import ProjectFile + + from .documents import DocumentFile + + +LOG = logging.getLogger(__name__) + + +class AssetMetadata(typing.Protocol): + """A template asset as registered with the template (e.g. in a database).""" + + @property + def uuid(self) -> str: ... + + @property + def file_name(self) -> str: ... + + @property + def content_type(self) -> str: ... + + +class LocalAsset(typing.NamedTuple): + uuid: str + file_name: str + content_type: str + + +class ProjectFileResolver(typing.Protocol): + """Provides the content of a file uploaded to the project being rendered. + + Returns the content, a path to a local file holding it, or None when the + file cannot be retrieved. + """ + + def resolve(self, file_uuid: str, name: str, + content_type: str) -> bytes | pathlib.Path | None: ... + + +class Asset: + + def __init__(self, *, uuid: str, name: str, content_type: str, + data: bytes, path: pathlib.Path): + self.uuid = uuid + self.name = name + self.content_type = content_type + self.data = data + self.path = path + + @property + def is_image(self) -> bool: + return self.content_type.startswith('image/') + + @property + def data_base64(self) -> str: + return base64.b64encode(self.data).decode('ascii') + + @property + def data_url(self) -> str: + return f'data:{self.content_type};base64,{self.data_base64}' + + @property + def src_value(self): + return self.data_url + + +class Template: + """A document template on disk, rendered by one of its formats. + + Everything the rendering depends on is passed in: the template directory + and its metadata, `RenderSettings`, the plugin manager and, per rendering, + the `RenderContext` with translations and a `ProjectFileResolver`. + """ + + def __init__(self, *, template_dir: pathlib.Path, formats: list[dict], + assets: Iterable[AssetMetadata] = (), template_uuid: str = '', + coordinates: str = '', settings: RenderSettings | None = None, + plugins: PluginManager | None = None): + self.template_dir = template_dir + self.formats_metadata = formats + self.assets: list[AssetMetadata] = list(assets) + self.template_uuid = template_uuid + self.coordinates = coordinates + self.settings = settings or RenderSettings() + if plugins is None: + from .plugins import create_manager + plugins = create_manager() + self.plugins = plugins + + self.formats: dict[str, Format] = {} + self.render_ctx = RenderContext.null() + self._project_files: ProjectFileResolver | None = None + + @classmethod + def from_directory(cls, template_dir: pathlib.Path, **kwargs) -> Template: + """Load a template from its directory with ``template.json``. + + Every other file in the directory is an asset, with its content type + guessed from the file name. + """ + from dsw.models.document_template.metadata import DocumentTemplateMetadata + from dsw.models.strictness import UnknownKeys, load + + metadata_file = template_dir / consts.TEMPLATE_JSON_FILE_NAME + metadata = load( + DocumentTemplateMetadata, + json.loads(metadata_file.read_text(encoding=consts.DEFAULT_ENCODING)), + unknown_keys=UnknownKeys.IGNORE, + ) + assets = [] + for path in sorted(template_dir.rglob('*')): + file_name = path.relative_to(template_dir).as_posix() + if not path.is_file() or file_name == consts.TEMPLATE_JSON_FILE_NAME: + continue + assets.append(LocalAsset( + uuid=str(uuid.uuid5(uuid.NAMESPACE_URL, file_name)), + file_name=file_name, + content_type=mimetypes.guess_type(file_name)[0] or 'application/octet-stream', + )) + kwargs.setdefault('assets', assets) + kwargs.setdefault('coordinates', metadata.coordinate) + return cls( + template_dir=template_dir, + formats=[f.model_dump(mode='json', by_alias=True) for f in metadata.formats], + **kwargs, + ) + + def raise_exc(self, message: str): + raise TemplateError(self.template_uuid, message) + + def fetch_asset(self, file_name: str) -> Asset | None: + LOG.info('Fetching asset "%s"', file_name) + file_path = self.template_dir / file_name + asset = None + for a in self.assets: + if a.file_name == file_name: + asset = a + break + if asset is None or not file_path.exists(): + LOG.error('Asset "%s" not found', file_name) + return None + return Asset( + uuid=asset.uuid, + name=file_name, + content_type=asset.content_type, + data=file_path.read_bytes(), + path=file_path, + ) + + def fetch_project_file(self, file: ProjectFile) -> Asset | None: + return self._fetch_project_file( + file_uuid=file.uuid, + name=file.name, + content_type=file.content_type, + ) + + def fetch_project_file_dict(self, file: dict) -> Asset | None: + file_uuid = file.get('uuid') + name = file.get('fileName') + content_type = file.get('contentType') + if isinstance(file_uuid, str) and isinstance(name, str) and isinstance(content_type, str): + return self._fetch_project_file( + file_uuid=file_uuid, + name=name, + content_type=content_type, + ) + return None + + def _fetch_project_file(self, file_uuid: str, name: str, + content_type: str) -> Asset | None: + LOG.info('Fetching project file "%s"', file_uuid) + if self._project_files is None: + LOG.warning('No project files available, cannot fetch project file') + return None + result = self._project_files.resolve(file_uuid, name, content_type) + if result is None: + LOG.error('Project file "%s" cannot be retrieved', file_uuid) + return None + if isinstance(result, bytes): + file_path = self.template_dir / consts.PROJECT_FILES_DIR / file_uuid + file_path.parent.mkdir(parents=True, exist_ok=True) + file_path.write_bytes(result) + else: + file_path = result + return Asset( + uuid=file_uuid, + name=name, + content_type=content_type, + data=file_path.read_bytes(), + path=file_path, + ) + + def asset_path(self, filename: str) -> str: + return str(self.template_dir / filename) + + def prepare_format(self, format_uuid: str) -> bool: + for format_meta in self.formats_metadata: + if format_uuid == format_meta.get(consts.FormatField.UUID): + self.formats[format_uuid] = Format(self, format_meta) + return True + return False + + def has_format(self, format_uuid: str) -> bool: + return any( + f[consts.FormatField.UUID] == format_uuid + for f in self.formats_metadata + ) + + def __getitem__(self, format_uuid: str) -> Format: + return self.formats[format_uuid] + + def render(self, format_uuid: str, context: dict, *, + render_ctx: RenderContext | None = None, + project_files: ProjectFileResolver | None = None) -> DocumentFile: + """Render the document with the (prepared) format. + + The format is prepared on the first use if `prepare_format` has not + been called. The document context may be enriched by plugins. + """ + if format_uuid not in self.formats and not self.prepare_format(format_uuid): + self.raise_exc(f'Format {format_uuid} not found') + self.plugins.hook.enrich_document_context(context=context) + + self.render_ctx = RenderContext.null() if render_ctx is None else render_ctx + self._project_files = project_files + try: + return self[format_uuid].execute(context) + finally: + self._project_files = None diff --git a/packages/dsw-document-worker/dsw/document_worker/templates/tests.py b/packages/dsw-templating/dsw/templating/tests.py similarity index 100% rename from packages/dsw-document-worker/dsw/document_worker/templates/tests.py rename to packages/dsw-templating/dsw/templating/tests.py diff --git a/packages/dsw-document-worker/dsw/document_worker/urls.py b/packages/dsw-templating/dsw/templating/urls.py similarity index 98% rename from packages/dsw-document-worker/dsw/document_worker/urls.py rename to packages/dsw-templating/dsw/templating/urls.py index bd07bc8b..1e092bb7 100644 --- a/packages/dsw-document-worker/dsw/document_worker/urls.py +++ b/packages/dsw-templating/dsw/templating/urls.py @@ -10,7 +10,7 @@ if typing.TYPE_CHECKING: - from .config import SecurityConfig + from .settings import SecuritySettings LOG = logging.getLogger(__name__) @@ -55,7 +55,7 @@ class UrlPolicy: is a concern. """ - def __init__(self, cfg: SecurityConfig): + def __init__(self, cfg: SecuritySettings): self.allow_external = cfg.allow_external_resources self.allow_private_network = cfg.allow_private_network self.allowed_hosts = frozenset(host.lower() for host in cfg.allowed_hosts) diff --git a/packages/dsw-document-worker/dsw/document_worker/utils.py b/packages/dsw-templating/dsw/templating/utils.py similarity index 100% rename from packages/dsw-document-worker/dsw/document_worker/utils.py rename to packages/dsw-templating/dsw/templating/utils.py diff --git a/packages/dsw-templating/pyproject.toml b/packages/dsw-templating/pyproject.toml new file mode 100644 index 00000000..8506b3da --- /dev/null +++ b/packages/dsw-templating/pyproject.toml @@ -0,0 +1,63 @@ +[project] +name = "dsw-templating" +description = "Data Stewardship Wizard document template rendering" +readme = "README.md" +keywords = ["documents", "dsw", "generation", "jinja2", "pandoc", "template"] +license = { text = "Apache License 2.0" } +authors = [ + { name = "Marek Suchánek", email = "marek.suchanek@ds-wizard.org" } +] +classifiers = [ + "Development Status :: 4 - Beta", + "License :: OSI Approved :: Apache Software License", + "Programming Language :: Python", + "Programming Language :: Python :: 3.12", + "Programming Language :: Python :: 3.13", + "Programming Language :: Python :: 3.14", + "Topic :: Text Processing", +] +requires-python = ">=3.12, <4" +dynamic = ["version", "dependencies", "optional-dependencies"] + +[project.urls] +Homepage = "https://ds-wizard.org" +Repository = "https://github.com/ds-wizard/engine-tools" +Documentation = "https://guide.ds-wizard.org" +Issues = "https://github.com/ds-wizard/ds-wizard/issues" + +[project.scripts] +# Pandoc filter for `--filter=pandoc-docx-pagebreakpy` (needs the docx extra) +pandoc-docx-pagebreakpy = "dsw.templating.docx_pagebreak:main" + +[build-system] +requires = ["hatchling==1.32.0", "uv-dynamic-versioning==0.14.1"] +build-backend = "hatchling.build" + + +[tool.hatch.version] +source = "uv-dynamic-versioning" + +[tool.hatch.build.targets.wheel] +packages = ["dsw"] +artifacts = ["dsw/*/build_info.py"] + +[tool.hatch.build.targets.sdist] +artifacts = ["dsw/*/build_info.py"] + +# Heavy dependencies are extras: a step whose extra is missing fails with an +# "install dsw-templating[]" error when it is used, not on import. +[tool.hatch.metadata.hooks.uv-dynamic-versioning] +dependencies = ["Babel", "Jinja2", "pluggy", "python-dateutil", "dsw-models[rendering]=={{ version }}"] + +[tool.hatch.metadata.hooks.uv-dynamic-versioning.optional-dependencies] +docx = ["panflute"] +excel = ["XlsxWriter"] +http = ["requests"] +pdf = ["weasyprint"] +rdf = ["rdflib", "rdflib-jsonld"] +all = ["panflute", "rdflib", "rdflib-jsonld", "requests", "weasyprint", "XlsxWriter"] +test = ["polib", "pytest"] + +[tool.uv-dynamic-versioning] +vcs = "git" +style = "pep440" diff --git a/packages/dsw-document-worker/support/DocumentContext.md b/packages/dsw-templating/support/DocumentContext.md similarity index 100% rename from packages/dsw-document-worker/support/DocumentContext.md rename to packages/dsw-templating/support/DocumentContext.md diff --git a/packages/dsw-document-worker/support/JinjaFilters.md b/packages/dsw-templating/support/JinjaFilters.md similarity index 97% rename from packages/dsw-document-worker/support/JinjaFilters.md rename to packages/dsw-templating/support/JinjaFilters.md index f2003be7..ffa2eec3 100644 --- a/packages/dsw-document-worker/support/JinjaFilters.md +++ b/packages/dsw-templating/support/JinjaFilters.md @@ -2,7 +2,7 @@ Within Jinja templates, you can use so-called [*filters*](https://jinja.palletsprojects.com/en/3.0.x/templates/#filters). Basically, those are functions applied to a first argument using pipe `|` symbol. -All of our filters are implemented in [templates.filters](../document_worker/templates/filters.py) module. +All of our filters are implemented in [filters](../dsw/templating/filters.py) module. ## Bultin Filters diff --git a/packages/dsw-document-worker/support/JinjaTests.md b/packages/dsw-templating/support/JinjaTests.md similarity index 88% rename from packages/dsw-document-worker/support/JinjaTests.md rename to packages/dsw-templating/support/JinjaTests.md index 41476437..a3bea52c 100644 --- a/packages/dsw-document-worker/support/JinjaTests.md +++ b/packages/dsw-templating/support/JinjaTests.md @@ -8,7 +8,7 @@ Within Jinja templates, you can use so-called [*tests*](https://jinja.palletspro {% endif %} ``` -All of our filters are implemented in [templates.tests](../document_worker/templates/tests.py) module. +All of our filters are implemented in [tests](../dsw/templating/tests.py) module. ## Bultin Tests diff --git a/packages/dsw-document-worker/support/Translations.md b/packages/dsw-templating/support/Translations.md similarity index 100% rename from packages/dsw-document-worker/support/Translations.md rename to packages/dsw-templating/support/Translations.md diff --git a/packages/dsw-document-worker/support/diagrams/dsw-document-context.svg b/packages/dsw-templating/support/diagrams/dsw-document-context.svg similarity index 100% rename from packages/dsw-document-worker/support/diagrams/dsw-document-context.svg rename to packages/dsw-templating/support/diagrams/dsw-document-context.svg diff --git a/packages/dsw-document-worker/support/diagrams/dsw-document-context.uxf b/packages/dsw-templating/support/diagrams/dsw-document-context.uxf similarity index 100% rename from packages/dsw-document-worker/support/diagrams/dsw-document-context.uxf rename to packages/dsw-templating/support/diagrams/dsw-document-context.uxf diff --git a/packages/dsw-document-worker/support/steps/archive.md b/packages/dsw-templating/support/steps/archive.md similarity index 100% rename from packages/dsw-document-worker/support/steps/archive.md rename to packages/dsw-templating/support/steps/archive.md diff --git a/packages/dsw-document-worker/support/steps/enrich-docx.md b/packages/dsw-templating/support/steps/enrich-docx.md similarity index 100% rename from packages/dsw-document-worker/support/steps/enrich-docx.md rename to packages/dsw-templating/support/steps/enrich-docx.md diff --git a/packages/dsw-document-worker/support/steps/excel.md b/packages/dsw-templating/support/steps/excel.md similarity index 100% rename from packages/dsw-document-worker/support/steps/excel.md rename to packages/dsw-templating/support/steps/excel.md diff --git a/packages/dsw-document-worker/support/steps/jinja.md b/packages/dsw-templating/support/steps/jinja.md similarity index 100% rename from packages/dsw-document-worker/support/steps/jinja.md rename to packages/dsw-templating/support/steps/jinja.md diff --git a/packages/dsw-document-worker/support/steps/json.md b/packages/dsw-templating/support/steps/json.md similarity index 100% rename from packages/dsw-document-worker/support/steps/json.md rename to packages/dsw-templating/support/steps/json.md diff --git a/packages/dsw-document-worker/support/steps/pandoc.md b/packages/dsw-templating/support/steps/pandoc.md similarity index 73% rename from packages/dsw-document-worker/support/steps/pandoc.md rename to packages/dsw-templating/support/steps/pandoc.md index 9f943f48..a367e999 100644 --- a/packages/dsw-document-worker/support/steps/pandoc.md +++ b/packages/dsw-templating/support/steps/pandoc.md @@ -18,12 +18,12 @@ Results in a document in desired format specified using `to` option. * `from` = specification of the input format (passed to Pandoc via `--from`, see [docs](https://pandoc.org/MANUAL.html#general-options)) * `to` = specification of the output format (passed to Pandoc via `--to`, see [docs](https://pandoc.org/MANUAL.html#general-options)) * (optional) `args` = additional command line arguments passed to [pandoc](https://pandoc.org/MANUAL.html) -* (optional, experimental) `filters` = additional [Pandoc filters](https://pandoc.org/MANUAL.html#general-options) to be used, need to be located under `/pandoc/filters` directory (or other set by `PANDOC_FILTERS` environment variable), comma separated +* (optional, experimental) `filters` = additional [Pandoc filters](https://pandoc.org/MANUAL.html#general-options) to be used, comma separated; the filters shipped with `dsw-templating` (`docx-landscape.lua`, `docx-pagebreak.lua`, `docx-toc.lua`) are always available, others need to be located in a directory configured by the runner (the Document Worker uses `/pandoc/filters`, or other set by `PANDOC_FILTERS` environment variable) * (optional, experimental) `template` = [Pandoc template](https://pandoc.org/MANUAL.html#general-options) to be used, need to be located under `/pandoc/templates` directory (or other set by `PANDOC_TEMPLATES` environment variable) ## Notes -* Pandoc filter `pandoc-docx-pagebreakpy` can be found in [addons](../../addons) directory. +* Pandoc filter `pandoc-docx-pagebreakpy` is installed as a command with `dsw-templating[docx]` (see [docx_pagebreak](../../dsw/templating/docx_pagebreak.py)). * Pandoc filter `pandoc-docx-pagebreakpy` will be removed with the next template metamodel version, use `` for the `filters` option instead. ## Example diff --git a/packages/dsw-document-worker/support/steps/rdflib-convert.md b/packages/dsw-templating/support/steps/rdflib-convert.md similarity index 100% rename from packages/dsw-document-worker/support/steps/rdflib-convert.md rename to packages/dsw-templating/support/steps/rdflib-convert.md diff --git a/packages/dsw-document-worker/support/steps/weasyprint.md b/packages/dsw-templating/support/steps/weasyprint.md similarity index 100% rename from packages/dsw-document-worker/support/steps/weasyprint.md rename to packages/dsw-templating/support/steps/weasyprint.md diff --git a/packages/dsw-templating/tests/conftest.py b/packages/dsw-templating/tests/conftest.py new file mode 100644 index 00000000..08236fb2 --- /dev/null +++ b/packages/dsw-templating/tests/conftest.py @@ -0,0 +1,17 @@ +import pathlib + +import pytest + +from dsw.templating import RenderSettings, Template, create_manager + + +@pytest.fixture +def make_template(): + """A template over a directory, with default settings and no project files.""" + def make(template_dir: pathlib.Path, **kwargs) -> Template: + kwargs.setdefault('formats', []) + kwargs.setdefault('coordinates', 'org:tid:1.0.0') + kwargs.setdefault('settings', RenderSettings()) + kwargs.setdefault('plugins', create_manager()) + return Template(template_dir=template_dir, **kwargs) + return make diff --git a/packages/dsw-templating/tests/fixtures/.gitattributes b/packages/dsw-templating/tests/fixtures/.gitattributes new file mode 100644 index 00000000..f7472e79 --- /dev/null +++ b/packages/dsw-templating/tests/fixtures/.gitattributes @@ -0,0 +1,2 @@ +# Golden outputs are compared byte for byte: no line ending conversion on checkout +* -text diff --git a/packages/dsw-templating/tests/fixtures/context.json b/packages/dsw-templating/tests/fixtures/context.json new file mode 100644 index 00000000..e85cf81d --- /dev/null +++ b/packages/dsw-templating/tests/fixtures/context.json @@ -0,0 +1,16 @@ +{ + "document": { + "name": "Golden Plan", + "language": "cs", + "createdAt": "2026-03-01T23:30:00Z" + }, + "entries": [ + {"title": "Datasets", "size": 1536000}, + {"title": "Software.", "size": null}, + {"title": "Publications", "size": 42} + ], + "summary": "Plan with **bold** text and a [link](https://example.org).\n\n", + "files": [ + {"uuid": "22222222-0000-0000-0000-000000000001", "fileName": "notes.txt", "contentType": "text/plain"} + ] +} diff --git a/packages/dsw-templating/tests/fixtures/golden/document.html b/packages/dsw-templating/tests/fixtures/golden/document.html new file mode 100644 index 00000000..dbbc660f --- /dev/null +++ b/packages/dsw-templating/tests/fixtures/golden/document.html @@ -0,0 +1,18 @@ + + +Golden Plan + +

Plán správy dat

+assets/logo.png +

02. 03. 2026

+
    +
  1. I. Datasets. (1.54 MB)
  2. +
  3. II. Software.
  4. +
  5. III. Publications. (42.0 B)
  6. +
+

Plan with bold text and a link.

+ +

notes.txt: Project file content (text/plain)

+

Enriched by a plugin

+ + \ No newline at end of file diff --git a/packages/dsw-templating/tests/fixtures/golden/document.json b/packages/dsw-templating/tests/fixtures/golden/document.json new file mode 100644 index 00000000..75ceeacc --- /dev/null +++ b/packages/dsw-templating/tests/fixtures/golden/document.json @@ -0,0 +1,32 @@ +{ + "document": { + "createdAt": "2026-03-01T23:30:00Z", + "language": "cs", + "name": "Golden Plan" + }, + "entries": [ + { + "size": 1536000, + "title": "Datasets" + }, + { + "size": null, + "title": "Software." + }, + { + "size": 42, + "title": "Publications" + } + ], + "extras": { + "plugin": "Enriched by a plugin" + }, + "files": [ + { + "contentType": "text/plain", + "fileName": "notes.txt", + "uuid": "22222222-0000-0000-0000-000000000001" + } + ], + "summary": "Plan with **bold** text and a [link](https://example.org).\n\n" +} \ No newline at end of file diff --git a/packages/dsw-templating/tests/fixtures/golden/sharedStrings.xml b/packages/dsw-templating/tests/fixtures/golden/sharedStrings.xml new file mode 100644 index 00000000..f8e5fa2e --- /dev/null +++ b/packages/dsw-templating/tests/fixtures/golden/sharedStrings.xml @@ -0,0 +1,2 @@ + +TitleSizeDatasetsSoftware.Publications \ No newline at end of file diff --git a/packages/dsw-templating/tests/fixtures/golden/sheet1.xml b/packages/dsw-templating/tests/fixtures/golden/sheet1.xml new file mode 100644 index 00000000..982d0f75 --- /dev/null +++ b/packages/dsw-templating/tests/fixtures/golden/sheet1.xml @@ -0,0 +1,2 @@ + +01215360003442 \ No newline at end of file diff --git a/packages/dsw-templating/tests/fixtures/template/assets/logo.png b/packages/dsw-templating/tests/fixtures/template/assets/logo.png new file mode 100644 index 00000000..f37764b1 Binary files /dev/null and b/packages/dsw-templating/tests/fixtures/template/assets/logo.png differ diff --git a/packages/dsw-templating/tests/fixtures/template/src/custom.xml.j2 b/packages/dsw-templating/tests/fixtures/template/src/custom.xml.j2 new file mode 100644 index 00000000..e72ad234 --- /dev/null +++ b/packages/dsw-templating/tests/fixtures/template/src/custom.xml.j2 @@ -0,0 +1,2 @@ + +{{ ctx.document.name }} diff --git a/packages/dsw-templating/tests/fixtures/template/src/document.html.j2 b/packages/dsw-templating/tests/fixtures/template/src/document.html.j2 new file mode 100644 index 00000000..5a7b8f18 --- /dev/null +++ b/packages/dsw-templating/tests/fixtures/template/src/document.html.j2 @@ -0,0 +1,19 @@ +{%- set logo = assets('assets/logo.png') -%} +{%- set attachment = assets(ctx.files[0]) -%} + + +{{ ctx.document.name }} + +

{% trans %}Data Management Plan{% endtrans %}

+{{ logo.name }} +

{{ ctx.document.createdAt|datetime_format('%d. %m. %Y', 'Europe/Prague') }}

+
    +{%- for item in ctx.entries %} +
  1. {{ loop.index|roman }}. {{ item.title|dot }}{% if item.size is not none %} ({{ item.size|bytesize_format }}){% endif %}
  2. +{%- endfor %} +
+{{ ctx.summary|markdown }} +{% if attachment %}

{{ attachment.name }}: {{ attachment.data.decode('utf-8') }} ({{ attachment.content_type }})

{% endif %} +

{{ ctx.extras.plugin }}

+ + diff --git a/packages/dsw-templating/tests/fixtures/template/src/sheet.json.j2 b/packages/dsw-templating/tests/fixtures/template/src/sheet.json.j2 new file mode 100644 index 00000000..fae756b7 --- /dev/null +++ b/packages/dsw-templating/tests/fixtures/template/src/sheet.json.j2 @@ -0,0 +1,15 @@ +{ + "properties": {"document": {"title": {{ ctx.document.name|tojson }}, "created": "2026-01-01"}}, + "formats": {"bold": {"bold": true}}, + "sheets": [ + { + "name": "Items", + "data": [ + {"type": "row", "cell": "A1", "data": ["Title", "Size"], "format": "bold"}, + {%- for item in ctx.entries %} + {"type": "row", "row": {{ loop.index }}, "col": 0, "data": [{{ item.title|tojson }}, {{ item.size|tojson }}]}{{ "," if not loop.last }} + {%- endfor %} + ] + } + ] +} diff --git a/packages/dsw-templating/tests/fixtures/template/template.json b/packages/dsw-templating/tests/fixtures/template/template.json new file mode 100644 index 00000000..e5b3c931 --- /dev/null +++ b/packages/dsw-templating/tests/fixtures/template/template.json @@ -0,0 +1,65 @@ +{ + "organizationId": "dsw", + "templateId": "golden", + "version": "1.0.0", + "name": "Golden Fixture", + "description": "Small template exercising the built-in steps", + "license": "Apache-2.0", + "metamodelVersion": "18.3", + "allowedPackages": [], + "formats": [ + { + "uuid": "11111111-0000-0000-0000-000000000001", + "name": "JSON Data", + "icon": "far fa-file", + "steps": [ + {"name": "json", "options": {}} + ] + }, + { + "uuid": "11111111-0000-0000-0000-000000000002", + "name": "HTML Document", + "icon": "far fa-file-code", + "steps": [ + {"name": "jinja", "options": {"template": "src/document.html.j2", "content-type": "text/html", "extension": "html"}} + ] + }, + { + "uuid": "11111111-0000-0000-0000-000000000003", + "name": "Zipped HTML", + "icon": "far fa-file-archive", + "steps": [ + {"name": "jinja", "options": {"template": "src/document.html.j2", "content-type": "text/html", "extension": "html"}}, + {"name": "archive", "options": {"type": "zip", "compression": "gzip", "inputFileDst": "report.html"}} + ] + }, + { + "uuid": "11111111-0000-0000-0000-000000000004", + "name": "Excel", + "icon": "far fa-file-excel", + "steps": [ + {"name": "jinja", "options": {"template": "src/sheet.json.j2", "content-type": "application/json", "extension": "json"}}, + {"name": "excel", "options": {}} + ] + }, + { + "uuid": "11111111-0000-0000-0000-000000000005", + "name": "PDF Document", + "icon": "far fa-file-pdf", + "steps": [ + {"name": "jinja", "options": {"template": "src/document.html.j2", "content-type": "text/html", "extension": "html"}}, + {"name": "weasyprint", "options": {}} + ] + }, + { + "uuid": "11111111-0000-0000-0000-000000000006", + "name": "MS Word Document", + "icon": "far fa-file-word", + "steps": [ + {"name": "jinja", "options": {"template": "src/document.html.j2", "content-type": "text/html", "extension": "html"}}, + {"name": "pandoc", "options": {"from": "html", "to": "docx", "filters": "docx-pagebreak.lua"}}, + {"name": "enrich-docx", "options": {"rewrite:docProps/custom.xml": "render:src/custom.xml.j2"}} + ] + } + ] +} diff --git a/packages/dsw-templating/tests/test_context.py b/packages/dsw-templating/tests/test_context.py new file mode 100644 index 00000000..57528496 --- /dev/null +++ b/packages/dsw-templating/tests/test_context.py @@ -0,0 +1,42 @@ +import pytest + +from dsw.templating import ContextDefaults, enrich_context_config + + +def test_defaults_fill_missing_branding(): + context = {'config': {'clientUrl': 'https://fw.example.org/', 'appTitle': None}} + enrich_context_config(context, ContextDefaults()) + assert context['config'] == { + 'clientUrl': 'https://fw.example.org/', + 'serviceName': 'Data Stewardship Wizard', + 'serviceNameShort': 'DSW', + 'serviceUrl': 'https://ds-wizard.org', + 'serviceDomainName': 'ds-wizard.org', + 'appTitle': 'DS Wizard', + 'appTitleShort': 'DS Wizard', + 'primaryColor': '#0033aa', + 'illustrationsColor': '#0033aa', + 'logoUrl': 'https://fw.example.org/assets/logo.svg', + } + + +def test_branding_is_kept_and_service_is_set(): + context = {'config': { + 'clientUrl': '', 'appTitle': 'My Wizard', 'appTitleShort': 'MW', + 'primaryColor': '#ff0000', 'illustrationsColor': '#00ff00', + 'logoUrl': 'https://cdn.example.org/logo.png', 'serviceName': 'Stale', + }} + enrich_context_config(context, ContextDefaults(service_name='FAIR Wizard')) + config = context['config'] + assert config['appTitle'] == 'My Wizard' + assert config['appTitleShort'] == 'MW' + assert config['primaryColor'] == '#ff0000' + assert config['illustrationsColor'] == '#00ff00' + assert config['logoUrl'] == 'https://cdn.example.org/logo.png' + assert config['serviceName'] == 'FAIR Wizard' + + +def test_context_without_config_is_rejected(): + # as in the document worker: the server always sends it + with pytest.raises(KeyError): + enrich_context_config({}, ContextDefaults()) diff --git a/packages/dsw-document-worker/tests/test_context_reference.py b/packages/dsw-templating/tests/test_context_reference.py similarity index 99% rename from packages/dsw-document-worker/tests/test_context_reference.py rename to packages/dsw-templating/tests/test_context_reference.py index 59d038f6..0e691972 100644 --- a/packages/dsw-document-worker/tests/test_context_reference.py +++ b/packages/dsw-templating/tests/test_context_reference.py @@ -11,7 +11,7 @@ import pathlib import uuid -from dsw.document_worker.templates.filters import to_context_obj +from dsw.templating.filters import to_context_obj from dsw.models.knowledge_model import graph as km_graph from dsw.models.knowledge_model.bundle import compile_bundle from dsw.models.knowledge_model.package import KnowledgeModelBundle diff --git a/packages/dsw-templating/tests/test_extras.py b/packages/dsw-templating/tests/test_extras.py new file mode 100644 index 00000000..5c2ed642 --- /dev/null +++ b/packages/dsw-templating/tests/test_extras.py @@ -0,0 +1,86 @@ +"""Optional dependencies: importing works without them, using a step does not.""" +import pathlib +import subprocess +import sys + +import pytest + +from dsw.templating import MissingExtraError, TemplateError + + +EXTRA_MODULES = ('panflute', 'rdflib', 'requests', 'weasyprint', 'xlsxwriter') + +IMPORT_ALL = f""" +import importlib, pkgutil, sys +for name in {EXTRA_MODULES!r}: + sys.modules[name] = None # makes any import of it fail +import dsw.templating +for mod in pkgutil.walk_packages(dsw.templating.__path__, 'dsw.templating.'): + importlib.import_module(mod.name) +leaked = sorted(name for name in {EXTRA_MODULES!r} if sys.modules.get(name) is not None) +assert not leaked, leaked +print('ok') +""" + + +def block(monkeypatch, *modules: str): + for module in modules: + monkeypatch.setitem(sys.modules, module, None) + + +def jinja_format(template: str, *steps: dict) -> dict: + return { + 'uuid': 'f', + 'name': 'Format', + 'steps': [ + {'name': 'jinja', 'options': {'template': template}}, + *steps, + ], + } + + +@pytest.fixture +def template_dir(tmp_path: pathlib.Path) -> pathlib.Path: + (tmp_path / 'root.j2').write_text('{{ rdflib.Graph() }}', encoding='utf-8') + return tmp_path + + +def test_import_without_extras(): + result = subprocess.run( + [sys.executable, '-c', IMPORT_ALL], + capture_output=True, text=True, check=False, + ) + assert result.returncode == 0, result.stderr + assert result.stdout.strip() == 'ok' + + +@pytest.mark.parametrize(('step', 'options', 'module', 'extra'), [ + ('weasyprint', {}, 'weasyprint', 'pdf'), + ('excel', {}, 'xlsxwriter', 'excel'), + ('rdflib-convert', {'from': 'turtle', 'to': 'jsonld'}, 'rdflib', 'rdf'), +]) +def test_step_without_extra_fails_clearly(monkeypatch, make_template, template_dir, + step, options, module, extra): + block(monkeypatch, module) + template = make_template(template_dir, formats=[ + jinja_format('root.j2', {'name': step, 'options': options}), + ]) + with pytest.raises(TemplateError, match=rf'install dsw-templating\[{extra}\]'): + template.prepare_format('f') + + +def test_requests_without_extra_fails_clearly(monkeypatch, make_template, template_dir): + from dsw.templating import RenderSettings, RequestsSettings, TemplateSettings + + block(monkeypatch, 'requests') + settings = RenderSettings(template=TemplateSettings(requests=RequestsSettings(enabled=True))) + template = make_template(template_dir, formats=[jinja_format('root.j2')], settings=settings) + with pytest.raises(TemplateError, match=r'install dsw-templating\[http\]'): + template.prepare_format('f') + + +def test_rdflib_global_without_extra_fails_when_used(monkeypatch, make_template, template_dir): + block(monkeypatch, 'rdflib') + template = make_template(template_dir, formats=[jinja_format('root.j2')]) + with pytest.raises(MissingExtraError, match=r'install dsw-templating\[rdf\]'): + template.render('f', {}) diff --git a/packages/dsw-templating/tests/test_golden.py b/packages/dsw-templating/tests/test_golden.py new file mode 100644 index 00000000..5b0a86f2 --- /dev/null +++ b/packages/dsw-templating/tests/test_golden.py @@ -0,0 +1,161 @@ +# cspell:ignore Plán správy +"""Golden outputs of a small fixture template, one per built-in step chain. + +Deterministic outputs are compared byte for byte with ``fixtures/golden``; for +archives and spreadsheets the relevant members are compared instead, as the +containers carry timestamps (and the line endings of XML members are normalized). Run with ``UPDATE_GOLDEN=1`` to rewrite them. +""" +import gettext +import io +import json +import os +import pathlib +import shutil +import zipfile + +import polib +import pytest + +from dsw.templating import ( + FileFormats, + RenderContext, + RenderSettings, + Template, + create_manager, + hookimpl, +) + + +FIXTURES = pathlib.Path(__file__).parent / 'fixtures' +GOLDEN = FIXTURES / 'golden' +UPDATE = os.getenv('UPDATE_GOLDEN') == '1' + +FORMAT_JSON = '11111111-0000-0000-0000-000000000001' +FORMAT_HTML = '11111111-0000-0000-0000-000000000002' +FORMAT_ZIP = '11111111-0000-0000-0000-000000000003' +FORMAT_XLSX = '11111111-0000-0000-0000-000000000004' +FORMAT_PDF = '11111111-0000-0000-0000-000000000005' +FORMAT_DOCX = '11111111-0000-0000-0000-000000000006' + + +class EnrichingPlugin: + + @hookimpl + def enrich_document_context(self, context: dict) -> None: + context.setdefault('extras', {})['plugin'] = 'Enriched by a plugin' + + +class FakeProjectFiles: + + def __init__(self): + self.requested: list[str] = [] + + def resolve(self, file_uuid: str, name: str, content_type: str) -> bytes | None: + self.requested.append(file_uuid) + return b'Project file content' + + +def assert_golden(name: str, data: bytes): + path = GOLDEN / name + if name.endswith('.xml'): + # XlsxWriter ends the XML declaration with the platform line separator + data = data.replace(b'\r\n', b'\n') + if UPDATE: + path.write_bytes(data) + assert data == path.read_bytes(), f'{name} differs from its golden file' + + +@pytest.fixture +def template(tmp_path: pathlib.Path) -> Template: + template_dir = tmp_path / 'template' + shutil.copytree(FIXTURES / 'template', template_dir) + plugins = create_manager() + plugins.register(EnrichingPlugin()) + return Template.from_directory(template_dir, settings=RenderSettings(), plugins=plugins) + + +@pytest.fixture +def context() -> dict: + return json.loads((FIXTURES / 'context.json').read_text(encoding='utf-8')) + + +@pytest.fixture +def render_ctx(tmp_path: pathlib.Path) -> RenderContext: + po = polib.POFile() + po.metadata = {'Content-Type': 'text/plain; charset=utf-8', 'Language': 'cs'} + po.append(polib.POEntry(msgid='Data Management Plan', msgstr='Plán správy dat')) + mo_path = tmp_path / 'messages.mo' + po.save_as_mofile(str(mo_path)) + with mo_path.open('rb') as fp: + return RenderContext(translations=gettext.GNUTranslations(fp), language='cs') + + +def test_template_from_directory(template): + assert template.coordinates == 'dsw:golden:1.0.0' + assert template.has_format(FORMAT_HTML) + asset = template.fetch_asset('assets/logo.png') + assert asset is not None + assert asset.content_type == 'image/png' + assert template.fetch_asset('assets/missing.png') is None + + +def test_json(template, context): + document = template.render(FORMAT_JSON, context) + assert document.file_format == FileFormats.JSON + assert document.encoding == 'utf-8' + assert_golden('document.json', document.content) + + +def test_html(template, context, render_ctx): + project_files = FakeProjectFiles() + document = template.render(FORMAT_HTML, context, + render_ctx=render_ctx, project_files=project_files) + assert document.file_format == FileFormats.HTML + assert project_files.requested == ['22222222-0000-0000-0000-000000000001'] + assert_golden('document.html', document.content) + + +def test_html_without_translations_and_project_files(template, context): + content = template.render(FORMAT_HTML, context).content.decode('utf-8') + assert '

Data Management Plan

' in content + assert 'notes.txt' not in content # no project files, the template renders on + + +def test_zip(template, context, render_ctx): + document = template.render(FORMAT_ZIP, context, + render_ctx=render_ctx, project_files=FakeProjectFiles()) + assert document.file_format == FileFormats.ZIP + with zipfile.ZipFile(io.BytesIO(document.content)) as archive: + assert archive.namelist() == ['report.html'] + assert archive.getinfo('report.html').compress_type == zipfile.ZIP_DEFLATED + assert_golden('document.html', archive.read('report.html')) + + +def test_excel(template, context): + pytest.importorskip('xlsxwriter') + document = template.render(FORMAT_XLSX, context) + assert document.file_format == FileFormats.XLSX + with zipfile.ZipFile(io.BytesIO(document.content)) as workbook: + assert_golden('sheet1.xml', workbook.read('xl/worksheets/sheet1.xml')) + assert_golden('sharedStrings.xml', workbook.read('xl/sharedStrings.xml')) + + +def test_pdf(template, context): + try: + import weasyprint # noqa: F401 + except (ImportError, OSError) as e: # OSError: Pango is not installed + pytest.skip(f'WeasyPrint is not usable: {e}') + document = template.render(FORMAT_PDF, context, project_files=FakeProjectFiles()) + assert document.file_format == FileFormats.PDF + assert document.content.startswith(b'%PDF-') + + +def test_docx(template, context): + if shutil.which('pandoc') is None: + pytest.skip('pandoc is not installed') + document = template.render(FORMAT_DOCX, context, project_files=FakeProjectFiles()) + assert document.file_format == FileFormats.DOCX + with zipfile.ZipFile(io.BytesIO(document.content)) as docx: + assert 'Golden Plan' in docx.read('docProps/custom.xml').decode('utf-8') + # the bundled Lua filter turned the class into an OOXML page break + assert 'w:type="page"' in docx.read('word/document.xml').decode('utf-8') diff --git a/packages/dsw-document-worker/tests/test_http.py b/packages/dsw-templating/tests/test_http.py similarity index 88% rename from packages/dsw-document-worker/tests/test_http.py rename to packages/dsw-templating/tests/test_http.py index 198558db..3bb01fcc 100644 --- a/packages/dsw-document-worker/tests/test_http.py +++ b/packages/dsw-templating/tests/test_http.py @@ -1,13 +1,9 @@ import pytest import requests -from dsw.document_worker.config import ( - SecurityConfig, - TemplateConfig, - TemplateRequestsConfig, -) -from dsw.document_worker.http import RequestsWrapper -from dsw.document_worker.urls import UrlNotAllowedError, UrlPolicy +from dsw.templating.http import RequestsWrapper +from dsw.templating.settings import RequestsSettings, SecuritySettings +from dsw.templating.urls import UrlNotAllowedError, UrlPolicy def make_wrapper(*, limit=100, **security) -> RequestsWrapper: @@ -19,15 +15,9 @@ def make_wrapper(*, limit=100, **security) -> RequestsWrapper: 'max_redirects': 3, } options.update(security) - template_cfg = TemplateConfig( - ids=['dsw:'], - requests=TemplateRequestsConfig(enabled=True, limit=limit, timeout=1), - secrets={}, - send_sentry=False, - ) return RequestsWrapper( - template_cfg=template_cfg, - policy=UrlPolicy(SecurityConfig(**options)), + settings=RequestsSettings(enabled=True, limit=limit, timeout=1), + policy=UrlPolicy(SecuritySettings(**options)), ) diff --git a/packages/dsw-document-worker/tests/test_i18n_roundtrip.py b/packages/dsw-templating/tests/test_i18n_roundtrip.py similarity index 86% rename from packages/dsw-document-worker/tests/test_i18n_roundtrip.py rename to packages/dsw-templating/tests/test_i18n_roundtrip.py index 4227cf90..d94a56fd 100644 --- a/packages/dsw-document-worker/tests/test_i18n_roundtrip.py +++ b/packages/dsw-templating/tests/test_i18n_roundtrip.py @@ -8,14 +8,13 @@ """ import gettext import pathlib -import types import polib import pytest -from dsw.document_worker.pot import extract_messages -from dsw.document_worker.templates.locales import RenderContext -from dsw.document_worker.templates.steps.template import Jinja2Step +from dsw.templating.locales import RenderContext +from dsw.templating.pot import extract_messages +from dsw.templating.steps.template import Jinja2Step ROOT_FILE = 'src/root.j2' @@ -60,15 +59,11 @@ def build_catalog(source: str, translate) -> gettext.GNUTranslations: @pytest.fixture -def step(fake_context, tmp_path: pathlib.Path) -> Jinja2Step: +def step(make_template, tmp_path: pathlib.Path) -> Jinja2Step: root = tmp_path / ROOT_FILE root.parent.mkdir(parents=True, exist_ok=True) root.write_text(SOURCE, encoding='utf-8') - template = types.SimpleNamespace( - template_dir=tmp_path, - coordinates='org:tid:1.0.0', - ) - return Jinja2Step(template, {'template': ROOT_FILE}) + return Jinja2Step(make_template(tmp_path), {'template': ROOT_FILE}) def install(step: Jinja2Step, tmp_path: pathlib.Path, translate) -> None: @@ -107,7 +102,7 @@ def test_plural_lookup_agrees(step, tmp_path): assert '' in render(step, n=5) -def test_reindenting_the_template_keeps_the_msgid(fake_context, tmp_path): +def test_reindenting_the_template_keeps_the_msgid(): reindented = SOURCE.replace(' Hello there', ' Hello there') assert reindented != SOURCE original = [m for _, m, _, _ in extract_messages(SOURCE)] diff --git a/packages/dsw-templating/tests/test_pot.py b/packages/dsw-templating/tests/test_pot.py new file mode 100644 index 00000000..8ceaff91 --- /dev/null +++ b/packages/dsw-templating/tests/test_pot.py @@ -0,0 +1,95 @@ +import dataclasses + +from dsw.templating.pot import extract_catalog, render_pot_file + + +@dataclasses.dataclass +class FakeTemplateFile: + file_name: str + content: str + + +def make_pot(*files, language='en'): + result = extract_catalog( + [FakeTemplateFile(name, content) for name, content in files], + project='org:tid:1.0.0', + version='1.0.0', + language=language, + ) + return result, render_pot_file(result).decode('utf-8') + + +def test_extract_trans_block(): + _, pot = make_pot(('src/a.j2', '{% trans %}Hello{% endtrans %}')) + assert 'msgid "Hello"' in pot + assert '#: src/a.j2:1' in pot + + +def test_extract_plural(): + _, pot = make_pot(( + 'src/a.j2', + '{% trans count %}{{ count }} item{% pluralize %}{{ count }} items{% endtrans %}', + )) + assert 'msgid "%(count)s item"' in pot + assert 'msgid_plural "%(count)s items"' in pot + assert 'msgstr[0] ""' in pot + assert 'msgstr[1] ""' in pot + + +def test_extract_underscore_and_pgettext(): + _, pot = make_pot(('src/a.j2', "{{ _('World') }}\n{{ pgettext('menu', 'Open') }}")) + assert 'msgid "World"' in pot + assert 'msgctxt "menu"' in pot + assert 'msgid "Open"' in pot + + +def test_extract_translators_comment(): + _, pot = make_pot(( + 'src/a.j2', + '{# TRANSLATORS: shown on top #}\n{% trans %}Hello{% endtrans %}', + )) + assert '#. shown on top' in pot + + +def test_extract_survives_do_extension(): + _, pot = make_pot(('src/a.j2', "{% do [] %}{{ _('Alpha') }}")) + assert 'msgid "Alpha"' in pot + + +def test_extract_survives_loopcontrols_extension(): + _, pot = make_pot(( + 'src/a.j2', + "{% for i in [1] %}{% break %}{% endfor %}{{ _('Alpha') }}", + )) + assert 'msgid "Alpha"' in pot + + +def test_broken_file_is_isolated(): + result, pot = make_pot( + ('src/broken.j2', '{% if %}'), + ('src/ok.j2', "{{ _('Alpha') }}"), + ) + assert result.failed_files == ['src/broken.j2'] + assert 'msgid "Alpha"' in pot + assert 'Skipped files that could not be parsed: src/broken.j2' in pot + + +def test_header_fields(): + _, pot = make_pot(('src/a.j2', "{{ _('Alpha') }}"), language='cs') + assert 'Project-Id-Version: org:tid:1.0.0 1.0.0' in pot + assert 'Language: cs' in pot + assert 'Plural-Forms: nplurals=' in pot + assert 'charset=utf-8' in pot + assert '#, fuzzy' not in pot + + +def test_header_without_known_language(): + _, pot = make_pot(('src/a.j2', "{{ _('Alpha') }}"), language='not a language') + assert 'msgid "Alpha"' in pot + assert 'Language:' not in pot + assert 'Plural-Forms:' not in pot + + +def test_messages_are_sorted(): + _, pot = make_pot(('src/a.j2', "{{ _('Beta') }}{{ _('Alpha') }}")) + assert pot.index('msgid "Alpha"') < pot.index('msgid "Beta"') diff --git a/packages/dsw-document-worker/tests/test_sanitizer.py b/packages/dsw-templating/tests/test_sanitizer.py similarity index 100% rename from packages/dsw-document-worker/tests/test_sanitizer.py rename to packages/dsw-templating/tests/test_sanitizer.py diff --git a/packages/dsw-document-worker/tests/test_steps_i18n.py b/packages/dsw-templating/tests/test_steps_i18n.py similarity index 78% rename from packages/dsw-document-worker/tests/test_steps_i18n.py rename to packages/dsw-templating/tests/test_steps_i18n.py index cfa43cd0..f3773ff5 100644 --- a/packages/dsw-document-worker/tests/test_steps_i18n.py +++ b/packages/dsw-templating/tests/test_steps_i18n.py @@ -1,12 +1,11 @@ import gettext import pathlib -import types import polib import pytest -from dsw.document_worker.templates.locales import RenderContext -from dsw.document_worker.templates.steps.template import Jinja2Step +from dsw.templating.locales import RenderContext +from dsw.templating.steps.template import Jinja2Step ROOT_FILE = 'src/root.j2' @@ -22,12 +21,8 @@ def template_dir(tmp_path: pathlib.Path) -> pathlib.Path: @pytest.fixture -def step(fake_context, template_dir: pathlib.Path) -> Jinja2Step: - template = types.SimpleNamespace( - template_dir=template_dir, - coordinates='org:tid:1.0.0', - ) - return Jinja2Step(template, {'template': ROOT_FILE}) +def step(make_template, template_dir: pathlib.Path) -> Jinja2Step: + return Jinja2Step(make_template(template_dir), {'template': ROOT_FILE}) def make_translations(tmp_path: pathlib.Path, @@ -73,12 +68,8 @@ def test_translation_helpers_on_step(step, tmp_path): assert step.translations is translations -def test_legacy_i18n_options_are_ignored(fake_context, template_dir): - template = types.SimpleNamespace( - template_dir=template_dir, - coordinates='org:tid:1.0.0', - ) - step = Jinja2Step(template, { +def test_legacy_i18n_options_are_ignored(make_template, template_dir): + step = Jinja2Step(make_template(template_dir), { 'template': ROOT_FILE, 'jinja-ext': 'i18n', 'i18n-dir': 'locale', diff --git a/packages/dsw-document-worker/tests/test_steps_policies.py b/packages/dsw-templating/tests/test_steps_policies.py similarity index 68% rename from packages/dsw-document-worker/tests/test_steps_policies.py rename to packages/dsw-templating/tests/test_steps_policies.py index 60ff0001..2c0178c7 100644 --- a/packages/dsw-document-worker/tests/test_steps_policies.py +++ b/packages/dsw-templating/tests/test_steps_policies.py @@ -1,9 +1,9 @@ import pathlib -import types import pytest -from dsw.document_worker.templates.steps.template import Jinja2Step +from dsw.templating import Template +from dsw.templating.steps.template import Jinja2Step ROOT_FILE = 'src/root.j2' @@ -18,19 +18,16 @@ def template_dir(tmp_path: pathlib.Path) -> pathlib.Path: def make_step(template_dir: pathlib.Path, **options) -> Jinja2Step: - template = types.SimpleNamespace( - template_dir=template_dir, - coordinates='org:tid:1.0.0', - ) + template = Template(template_dir=template_dir, formats=[], coordinates='org:tid:1.0.0') return Jinja2Step(template, {'template': ROOT_FILE, **options}) -def test_extra_schemes_applied(fake_context, template_dir): +def test_extra_schemes_applied(template_dir): step = make_step(template_dir, **{'policy.urlize.extra_schemes': 'ftp:,tel:'}) assert step.j2_env.policies['urlize.extra_schemes'] == ['ftp:', 'tel:'] -def test_extra_schemes_does_not_clobber_truncate_leeway(fake_context, template_dir): +def test_extra_schemes_does_not_clobber_truncate_leeway(template_dir): step = make_step(template_dir, **{ 'policy.urlize.extra_schemes': 'ftp:', 'policy.truncate.leeway': '7', @@ -38,6 +35,6 @@ def test_extra_schemes_does_not_clobber_truncate_leeway(fake_context, template_d assert step.j2_env.policies['truncate.leeway'] == '7' -def test_truncate_leeway_default_kept_without_extra_schemes(fake_context, template_dir): +def test_truncate_leeway_default_kept_without_extra_schemes(template_dir): step = make_step(template_dir, **{'policy.urlize.extra_schemes': 'ftp:'}) assert step.j2_env.policies['truncate.leeway'] == 5 diff --git a/packages/dsw-templating/tests/test_template.py b/packages/dsw-templating/tests/test_template.py new file mode 100644 index 00000000..aa9a90bc --- /dev/null +++ b/packages/dsw-templating/tests/test_template.py @@ -0,0 +1,148 @@ +import io +import pathlib +import zipfile + +import pytest + +from dsw.templating import ( + DocumentFile, + FileFormats, + PandocSettings, + TemplateError, + TemplateTriggeredError, +) +from dsw.templating.conversions import FormatConversionError, Pandoc +from dsw.templating.settings import bundled_pandoc_filters +from dsw.templating.steps.word import EnrichDocxStep + + +def html_format(template: str) -> dict: + return { + 'uuid': 'f', + 'name': 'HTML', + 'steps': [{'name': 'jinja', 'options': {'template': template}}], + } + + +class PathProjectFiles: + + def __init__(self, path: pathlib.Path): + self.path = path + + def resolve(self, file_uuid, name, content_type): + return self.path if self.path.exists() else None + + +def test_unknown_format(make_template, tmp_path): + with pytest.raises(TemplateError, match='Format x not found'): + make_template(tmp_path).render('x', {}) + + +def test_project_file_from_path(make_template, tmp_path): + (tmp_path / 'root.j2').write_text( + "{% set f = assets({'uuid': 'u', 'fileName': 'a.txt', 'contentType': 'text/plain'}) %}" + '{{ f.data.decode() if f else "missing" }}', + encoding='utf-8', + ) + source = tmp_path / 'outside.txt' + template = make_template(tmp_path, formats=[html_format('root.j2')]) + + def render(**kwargs): + return template.render('f', {}, **kwargs).content.decode('utf-8') + + assert render() == 'missing' + assert render(project_files=PathProjectFiles(source)) == 'missing' + source.write_text('content', encoding='utf-8') + assert render(project_files=PathProjectFiles(source)) == 'content' + assert render() == 'missing' # the resolver does not outlive its rendering + + +def test_error_filter_raises_template_triggered_error(make_template, tmp_path): + (tmp_path / 'root.j2').write_text("{{ 'Bad input'|error('Oops') }}", encoding='utf-8') + template = make_template(tmp_path, formats=[html_format('root.j2')]) + with pytest.raises(TemplateTriggeredError) as e: + template.render('f', {}) + assert e.value.msg == 'Oops\n\nBad input' + + +def test_enrich_docx_rewrites(make_template, tmp_path): + (tmp_path / 'custom.xml.j2').write_text('

{{ ctx.name }}|{{ content }}

', + encoding='utf-8') + (tmp_path / 'static.xml').write_text('', encoding='utf-8') + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, mode='w') as docx: + docx.writestr('word/document.xml', '') + docx.writestr('docProps/custom.xml', '') + step = EnrichDocxStep(make_template(tmp_path), { + 'rewrite:docProps/custom.xml': 'render:custom.xml.j2', + 'rewrite:word/extra.xml': 'static:static.xml', + }) + result = step.execute_follow(DocumentFile(FileFormats.DOCX, buffer.getvalue()), {'name': 'N'}) + with zipfile.ZipFile(io.BytesIO(result.content)) as docx: + assert docx.read('docProps/custom.xml') == b'

N|<old/>

' + assert docx.read('word/extra.xml') == b'' + assert docx.read('word/document.xml') == b'' + + +def test_bundled_pandoc_filters_are_packaged(): + names = {path.name for path in bundled_pandoc_filters().iterdir()} + assert {'docx-landscape.lua', 'docx-pagebreak.lua', 'docx-toc.lua'} <= names + + +def test_pandoc_filter_lookup_order(tmp_path): + custom = tmp_path / 'custom' + custom.mkdir() + (custom / 'docx-toc.lua').write_text('-- custom', encoding='utf-8') + (custom / 'mine.py').write_text('', encoding='utf-8') + settings = PandocSettings( + command=['pandoc'], + filter_dirs=[custom, bundled_pandoc_filters()], + templates_dir=tmp_path, + ) + pandoc = Pandoc(settings, ['docx-toc.lua', 'docx-pagebreak.lua', 'mine.py'], None) + assert pandoc._extra_args() == [ + '--lua-filter', str(custom / 'docx-toc.lua'), + '--lua-filter', str(bundled_pandoc_filters() / 'docx-pagebreak.lua'), + '--filter', str(custom / 'mine.py'), + ] + with pytest.raises(RuntimeError, match='Pandoc filter "nope.lua" not found'): + Pandoc(settings, ['nope.lua'], None) + with pytest.raises(RuntimeError, match='Pandoc template "nope.tex" not found'): + Pandoc(settings, [], 'nope.tex') + + +def test_missing_pandoc_is_reported(tmp_path): + pandoc = Pandoc(PandocSettings(command=[str(tmp_path / 'no-pandoc')]), [], None) + with pytest.raises(FormatConversionError, match='no-pandoc" not found, is it installed'): + pandoc(source_format=FileFormats.HTML, target_format=FileFormats.DOCX, + data=b'

', metadata={}, workdir=str(tmp_path)) + + +def test_docx_pagebreak_filter(): + panflute = pytest.importorskip('panflute') + from dsw.templating import docx_pagebreak + + doc = panflute.Doc(panflute.Para(panflute.Str(r'\newpage')), + panflute.Para(panflute.Str(r'\toc2')), + panflute.Para(panflute.Str('text')), format='docx') + blocks = docx_pagebreak.main(doc).content + assert blocks[0].text == '' + assert 'TOC \\o "1-2"' in blocks[1].text + assert isinstance(blocks[2], panflute.Para) + + +def test_docx_pagebreak_filter_is_installed(): + import importlib.metadata + + scripts = importlib.metadata.entry_points(group='console_scripts') + assert scripts['pandoc-docx-pagebreakpy'].value == 'dsw.templating.docx_pagebreak:main' + + +def test_docx_pagebreak_filter_without_extra(monkeypatch): + import sys + + from dsw.templating import docx_pagebreak + + monkeypatch.setitem(sys.modules, 'panflute', None) + with pytest.raises(SystemExit, match=r'install dsw-templating\[docx\]'): + docx_pagebreak.main() diff --git a/packages/dsw-document-worker/tests/test_urls.py b/packages/dsw-templating/tests/test_urls.py similarity index 95% rename from packages/dsw-document-worker/tests/test_urls.py rename to packages/dsw-templating/tests/test_urls.py index f4082db7..dddf16b9 100644 --- a/packages/dsw-document-worker/tests/test_urls.py +++ b/packages/dsw-templating/tests/test_urls.py @@ -2,8 +2,8 @@ import pytest -from dsw.document_worker.config import SecurityConfig -from dsw.document_worker.urls import UrlNotAllowedError, UrlPolicy +from dsw.templating.settings import SecuritySettings +from dsw.templating.urls import UrlNotAllowedError, UrlPolicy def make_policy(**kwargs) -> UrlPolicy: @@ -15,7 +15,7 @@ def make_policy(**kwargs) -> UrlPolicy: 'max_redirects': 3, } options.update(kwargs) - return UrlPolicy(SecurityConfig(**options)) + return UrlPolicy(SecuritySettings(**options)) @pytest.mark.parametrize('url', [ diff --git a/pyproject.toml b/pyproject.toml index cd7912d7..c902129a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -23,6 +23,7 @@ dsw-mailer = { workspace = true } dsw-models = { workspace = true } dsw-storage = { workspace = true } dsw-tdk = { workspace = true } +dsw-templating = { workspace = true } [tool.uv.workspace] members = ["packages/*"] diff --git a/uv.lock b/uv.lock index 27c8c3bc..f90cfcc8 100644 --- a/uv.lock +++ b/uv.lock @@ -13,6 +13,7 @@ members = [ "dsw-models", "dsw-storage", "dsw-tdk", + "dsw-templating", ] [manifest.dependency-groups] @@ -502,27 +503,19 @@ requires-dist = [ name = "dsw-document-worker" source = { editable = "packages/dsw-document-worker" } dependencies = [ - { name = "babel" }, { name = "click" }, { name = "dsw-command-queue" }, { name = "dsw-config" }, { name = "dsw-database" }, - { name = "dsw-models", extra = ["rendering"] }, + { name = "dsw-models" }, { name = "dsw-storage" }, - { name = "jinja2" }, - { name = "panflute" }, + { name = "dsw-templating", extra = ["all"] }, { name = "pathvalidate" }, - { name = "pluggy" }, { name = "polib" }, { name = "python-dateutil" }, { name = "python-slugify" }, - { name = "rdflib" }, - { name = "rdflib-jsonld" }, - { name = "requests" }, { name = "sentry-sdk" }, { name = "tenacity" }, - { name = "weasyprint" }, - { name = "xlsxwriter" }, ] [package.optional-dependencies] @@ -532,28 +525,20 @@ test = [ [package.metadata] requires-dist = [ - { name = "babel" }, { name = "click" }, { name = "dsw-command-queue", editable = "packages/dsw-command-queue" }, { name = "dsw-config", editable = "packages/dsw-config" }, { name = "dsw-database", editable = "packages/dsw-database" }, - { name = "dsw-models", extras = ["rendering"], editable = "packages/dsw-models" }, + { name = "dsw-models", editable = "packages/dsw-models" }, { name = "dsw-storage", editable = "packages/dsw-storage" }, - { name = "jinja2" }, - { name = "panflute" }, + { name = "dsw-templating", extras = ["all"], editable = "packages/dsw-templating" }, { name = "pathvalidate" }, - { name = "pluggy" }, { name = "polib" }, { name = "pytest", marker = "extra == 'test'" }, { name = "python-dateutil" }, { name = "python-slugify" }, - { name = "rdflib" }, - { name = "rdflib-jsonld" }, - { name = "requests" }, { name = "sentry-sdk" }, { name = "tenacity" }, - { name = "weasyprint" }, - { name = "xlsxwriter" }, ] provides-extras = ["test"] @@ -655,6 +640,7 @@ dependencies = [ { name = "click" }, { name = "colorama" }, { name = "dsw-models" }, + { name = "dsw-templating" }, { name = "humanize" }, { name = "jinja2" }, { name = "multidict" }, @@ -666,6 +652,9 @@ dependencies = [ ] [package.optional-dependencies] +all = [ + { name = "dsw-templating", extra = ["all"] }, +] test = [ { name = "pytest" }, { name = "pytest-recording" }, @@ -679,6 +668,8 @@ requires-dist = [ { name = "click" }, { name = "colorama" }, { name = "dsw-models", editable = "packages/dsw-models" }, + { name = "dsw-templating", editable = "packages/dsw-templating" }, + { name = "dsw-templating", extras = ["all"], marker = "extra == 'all'", editable = "packages/dsw-templating" }, { name = "humanize" }, { name = "jinja2" }, { name = "multidict" }, @@ -691,7 +682,72 @@ requires-dist = [ { name = "vcrpy", marker = "extra == 'test'" }, { name = "watchfiles" }, ] -provides-extras = ["test"] +provides-extras = ["all", "test"] + +[[package]] +name = "dsw-templating" +source = { editable = "packages/dsw-templating" } +dependencies = [ + { name = "babel" }, + { name = "dsw-models", extra = ["rendering"] }, + { name = "jinja2" }, + { name = "pluggy" }, + { name = "python-dateutil" }, +] + +[package.optional-dependencies] +all = [ + { name = "panflute" }, + { name = "rdflib" }, + { name = "rdflib-jsonld" }, + { name = "requests" }, + { name = "weasyprint" }, + { name = "xlsxwriter" }, +] +docx = [ + { name = "panflute" }, +] +excel = [ + { name = "xlsxwriter" }, +] +http = [ + { name = "requests" }, +] +pdf = [ + { name = "weasyprint" }, +] +rdf = [ + { name = "rdflib" }, + { name = "rdflib-jsonld" }, +] +test = [ + { name = "polib" }, + { name = "pytest" }, +] + +[package.metadata] +requires-dist = [ + { name = "babel" }, + { name = "dsw-models", extras = ["rendering"], editable = "packages/dsw-models" }, + { name = "jinja2" }, + { name = "panflute", marker = "extra == 'all'" }, + { name = "panflute", marker = "extra == 'docx'" }, + { name = "pluggy" }, + { name = "polib", marker = "extra == 'test'" }, + { name = "pytest", marker = "extra == 'test'" }, + { name = "python-dateutil" }, + { name = "rdflib", marker = "extra == 'all'" }, + { name = "rdflib", marker = "extra == 'rdf'" }, + { name = "rdflib-jsonld", marker = "extra == 'all'" }, + { name = "rdflib-jsonld", marker = "extra == 'rdf'" }, + { name = "requests", marker = "extra == 'all'" }, + { name = "requests", marker = "extra == 'http'" }, + { name = "weasyprint", marker = "extra == 'all'" }, + { name = "weasyprint", marker = "extra == 'pdf'" }, + { name = "xlsxwriter", marker = "extra == 'all'" }, + { name = "xlsxwriter", marker = "extra == 'excel'" }, +] +provides-extras = ["all", "docx", "excel", "http", "pdf", "rdf", "test"] [[package]] name = "fonttools"