From 3b23956de2ba48fa7cae3c9a0f5f9367d58235bb Mon Sep 17 00:00:00 2001 From: Jyrki Launonen Date: Wed, 30 Sep 2026 19:17:57 +0300 Subject: [PATCH 1/5] fix(emprinten): Convert url fetcher responses to URLFetcherResponse The default_url_fetcher was deprecated in weasyprint-68 and a new URLFetcher class was introduced to replace it. Unlike what the exception message says, the new API requires the response to be a URLFetcherResponse instead of a dict. AssertionError at /events/.../reference URL fetcher must return either a dict or a URLFetcherResponse instance Traceback (most recent call last): ... File "/usr/src/app/kompassi/labour/views/public_views.py", line 215, in profile_work_reference return render_obj( File "/usr/src/app/kompassi/emprinten/utils.py", line 52, in render_obj return render_pdf( File "/usr/src/app/kompassi/emprinten/renderer.py", line 130, in render_pdf results: list[FileWithData] = wp.compile(sources, result_dir) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "/usr/src/app/kompassi/emprinten/renderer.py", line 324, in compile pdf = pdf_html.write_pdf( File "/usr/src/app/.venv/lib/python3.14/site-packages/weasyprint/__init__.py", line 264, in write_pdf document = self.render( ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "/usr/src/app/.venv/lib/python3.14/site-packages/weasyprint/__init__.py", line 221, in render return Document._render( ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "/usr/src/app/.venv/lib/python3.14/site-packages/weasyprint/document.py", line 201, in _render root_box = build_formatting_structure( ^^^^^^^^^^^ File "/usr/src/app/.venv/lib/python3.14/site-packages/weasyprint/formatting_structure/build.py", line 56, in build_formatting_structure box_list = element_to_box( ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "/usr/src/app/.venv/lib/python3.14/site-packages/weasyprint/formatting_structure/build.py", line 181, in element_to_box child_boxes = element_to_box( ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ ... File "/usr/src/app/.venv/lib/python3.14/site-packages/weasyprint/formatting_structure/build.py", line 276, in element_to_box return html.handle_element(element, box, get_image_from_uri, base_url) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "/usr/src/app/.venv/lib/python3.14/site-packages/weasyprint/html.py", line 186, in handle_element return HTML_HANDLERS[element.tag](element, box, get_image_from_uri, base_url) ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "/usr/src/app/.venv/lib/python3.14/site-packages/weasyprint/html.py", line 223, in handle_img if image := get_image_from_uri(url=src, orientation=orientation): ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ File "/usr/src/app/.venv/lib/python3.14/site-packages/weasyprint/images.py", line 297, in get_image_from_uri with fetch(url_fetcher, url) as response: ^^^^^^^^^^^^^^^^^^^^^^^ File "/usr/local/lib/python3.14/contextlib.py", line 141, in __enter__ return next(self.gen) ^^^^^^^^^^^^^^ File "/usr/src/app/.venv/lib/python3.14/site-packages/weasyprint/urls.py", line 432, in fetch assert isinstance(resource, URLFetcherResponse), ( ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ --- kompassi/emprinten/renderer.py | 54 ++++++++++++++++++++++------------ 1 file changed, 35 insertions(+), 19 deletions(-) diff --git a/kompassi/emprinten/renderer.py b/kompassi/emprinten/renderer.py index 377b487c9..ad56a80e7 100644 --- a/kompassi/emprinten/renderer.py +++ b/kompassi/emprinten/renderer.py @@ -14,6 +14,7 @@ from django.http import FileResponse, HttpResponse, HttpResponseBase from jinja2 import FunctionLoader from jinja2.sandbox import SandboxedEnvironment +from weasyprint.urls import URLFetcher, URLFetcherResponse from . import filters, functions from .files import Lut, NameFactory, make_lut, make_name @@ -306,11 +307,15 @@ def find_stylesheets(files: typing.Iterable[FileVersion]) -> list[FileVersion]: return [file_version for file_version in files if file_version.file.type == ProjectFile.Type.CSS] def compile(self, sources: list[FileWithData], result_dir: str) -> list[FileWithData]: + """ + Compile HTML `sources` into PDF files in `result_dir`. + """ + url_fetcher = VfsFetcher(self.vfs) parsed_sheets = [ weasyprint.CSS( string=sheet_file.data.read(), base_url=LOCAL_FILE_URI_PREFIX, - url_fetcher=self._do_lookup, + url_fetcher=url_fetcher, ) for sheet_file in self.stylesheets ] @@ -319,7 +324,7 @@ def compile(self, sources: list[FileWithData], result_dir: str) -> list[FileWith pdf_html = weasyprint.HTML( filename=source, base_url=LOCAL_FILE_URI_PREFIX, - url_fetcher=self._do_lookup, + url_fetcher=url_fetcher, ) pdf = pdf_html.write_pdf( stylesheets=parsed_sheets, @@ -336,33 +341,44 @@ def compile(self, sources: list[FileWithData], result_dir: str) -> list[FileWith return results - # See `weasyprint.urls.default_url_fetcher` for function signature. - # Note: At least some exceptions are silently ignored by weasyprint. - def _do_lookup(self, url: str, timeout: int = 10, ssl_context=None) -> dict: + +class VfsFetcher(URLFetcher): + def __init__(self, vfs: Vfs) -> None: + super().__init__() + self._vfs = vfs + self._data_director = urllib.request.OpenerDirector() + self._data_director.add_handler(urllib.request.DataHandler()) + + def fetch(self, url, headers=None) -> URLFetcherResponse: + """ + Resolve `data:` and `file:` URLs into content. + Former will be decoded, latter retrieved from VFS (if an exact match exists). + + See `weasyprint.urls.URLFetcher.fetch`. + """ if url.startswith("data:"): - director = urllib.request.OpenerDirector() - director.add_handler(urllib.request.DataHandler()) - data_response = director.open(url) + data_response = self._data_director.open(url) if data_response is None: restricted_url = "Invalid data URL" raise ValueError(restricted_url) - return { - "redirected_url": url, - "mime_type": data_response.headers["content-type"], - "string": data_response.file.read(), - } + return URLFetcherResponse( + url=url, + body=data_response, + headers={"content-type": data_response.headers["content-type"]}, + ) file_url = url.removeprefix(LOCAL_FILE_URI_PREFIX) if file_url == url: restricted_url = "Invalid URL to look up for" raise ValueError(restricted_url) - the_file: FileVersion | None = self.vfs.get(file_url) + the_file: FileVersion | None = self._vfs.get(file_url) if DEBUG: print("Pdf lookup", url, the_file) if the_file is None: raise KeyError - return { - "file_obj": the_file.data.open("rb"), - # Weasyprint requires this to avoid file not found exc with the original filename. - "redirected_url": file_url, - } + return URLFetcherResponse( + # This value is ignored, but return the actual name. + # Any potential relative references won't be found from VFS as only exact matches exist. + url="file:///" + the_file.file.file_name, + body=the_file.data.open("rb"), + ) From 92b8eda744f3bc2f1990e667ff52a99d9ef49a48 Mon Sep 17 00:00:00 2001 From: Jyrki Launonen Date: Wed, 30 Sep 2026 23:03:48 +0300 Subject: [PATCH 2/5] fix(emprinten): Ensure read storage files are closed The .read() shortcut automatically opens the file, but doesn't close it. The _TemplateCompiler._do_lookup didn't leak a handle as it already used the instance as a context manager, which closes the file at __exit__. Technically the CSS file mode changes here from rb to rt, but that should be fine. --- kompassi/emprinten/renderer.py | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/kompassi/emprinten/renderer.py b/kompassi/emprinten/renderer.py index ad56a80e7..f79662b25 100644 --- a/kompassi/emprinten/renderer.py +++ b/kompassi/emprinten/renderer.py @@ -292,11 +292,22 @@ def _do_lookup( ProjectFile.Type.CSS, ): return None - with the_file.data.open("rt") as tpl_file: - src = tpl_file.read() + src = read_and_close(the_file.data) return src, name, lambda: True +def read_and_close(field_file) -> str: + """ + Read given `FieldFile` and then close it. + + A `FileField` of a model class becomes `FieldFile` on a model instance. + Using `FieldFile.read()` (shortcut for `FieldFile.file.read()`) opens the file but doesn't close it. + It is also a context manager that closes it for us automatically. + """ + with field_file.open("rt") as file: + return file.read() + + class _HtmlCompiler: def __init__(self, vfs: Vfs) -> None: self.vfs = vfs @@ -313,7 +324,7 @@ def compile(self, sources: list[FileWithData], result_dir: str) -> list[FileWith url_fetcher = VfsFetcher(self.vfs) parsed_sheets = [ weasyprint.CSS( - string=sheet_file.data.read(), + string=read_and_close(sheet_file.data), base_url=LOCAL_FILE_URI_PREFIX, url_fetcher=url_fetcher, ) From 0358cb4be13b9681f3634149e17e2ebe596abf81 Mon Sep 17 00:00:00 2001 From: Jyrki Launonen Date: Sat, 24 Aug 2024 13:04:33 +0300 Subject: [PATCH 3/5] fix(emprinten): Fix conflicting name check While not an error per Python, getting a semi-random value for the resulting field could confuse user. The check did not collapse the resulting names correctly. --- kompassi/emprinten/files.py | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/kompassi/emprinten/files.py b/kompassi/emprinten/files.py index 6e8a16503..88ac1b2d7 100644 --- a/kompassi/emprinten/files.py +++ b/kompassi/emprinten/files.py @@ -50,8 +50,18 @@ def make_name(name: str) -> str: def parse_header_names(cols: list[str]) -> list[str]: + """ + >>> parse_header_names(["foo", "bar"]) + ['foo', 'bar'] + >>> parse_header_names(["Foo", "Bar!"]) + ['foo', 'bar'] + >>> parse_header_names(["foo", "bar", "bar!"]) + Traceback (most recent call last): + ... + ValueError: Conflicting field names in header + """ names = [make_name(col) for col in cols] - if len(names) != len(cols): + if len(set(names)) != len(cols): fields = "Conflicting field names in header" raise ValueError(fields) return names From 4446ecd01861b8bb7728231801da38f5b71bd065 Mon Sep 17 00:00:00 2001 From: Jyrki Launonen Date: Sun, 13 Apr 2025 09:53:10 +0300 Subject: [PATCH 4/5] chore(emprinten): Clarify FileWithData as TypedDict While the unnamed tuple indices did work, it is clearer to use TypedDict without destructuring. --- kompassi/emprinten/renderer.py | 52 +++++++++++++++++++++++----------- 1 file changed, 36 insertions(+), 16 deletions(-) diff --git a/kompassi/emprinten/renderer.py b/kompassi/emprinten/renderer.py index f79662b25..c06216c66 100644 --- a/kompassi/emprinten/renderer.py +++ b/kompassi/emprinten/renderer.py @@ -22,7 +22,13 @@ DEBUG = False -FileWithData = tuple[str, dict[str, str | dict[str, typing.Any]] | None, bool] + +class FileWithData(typing.TypedDict): + file_name: str + data: dict[str, str | dict[str, typing.Any]] | None + success: bool + + DataRow = dict[str, str | dict[str, typing.Any]] DataSet = list[DataRow] Vfs = dict[str, FileVersion] @@ -89,6 +95,15 @@ def render_pdf( return_archive: bool = False, handle_errors: bool = False, ) -> HttpResponseBase: + """ + Render `data` row(s) with given `files` (templates and resources) into PDF's. + The result is either a `HttpResponse` (any kind of failure) + or a `FileResponse` (for both singular PDF and ZIP with multiple PDF's). + + The `return_archive` controls whether the result is a PDF or a ZIP, both with one or more data rows. + If `handle_errors` is `True`, problems during template compilation will return + a success with error message per problematic row. + """ main = find_main(files) if main is None: return HttpResponse("Main file not found", status=404) @@ -135,14 +150,14 @@ def render_pdf( if return_archive: z_name = os.path.join(tmpdir, "result.zip") with zipfile.ZipFile(z_name, "w") as z: - for pdf_name, row, success in results: - post_format = "{}" if success else RENDER_FAILURE_FILE_NAME_PATTERN + for result in results: + post_format = "{}" if result["success"] else RENDER_FAILURE_FILE_NAME_PATTERN arc_name = name_factory.make( - {"row": row}, - fallback=os.path.basename(pdf_name), + {"row": result["data"]}, + fallback=os.path.basename(result["file_name"]), post_format=post_format, ) - z.write(pdf_name, arcname=arc_name) + z.write(result["file_name"], arcname=arc_name) if DEBUG: ls_r(tmpdir) # FileResponse closes the open file by itself. @@ -152,12 +167,12 @@ def render_pdf( return HttpResponse(status=401) if results: - pdf_name, row, success = results[0] - post_format = "{}" if success else RENDER_FAILURE_FILE_NAME_PATTERN - file_name = name_factory.make({"row": row}, fallback="result.pdf", post_format=post_format) + result = results[0] + post_format = "{}" if result["success"] else RENDER_FAILURE_FILE_NAME_PATTERN + file_name = name_factory.make({"row": result["data"]}, fallback="result.pdf", post_format=post_format) # FileResponse closes the open file by itself. return FileResponse( - open(pdf_name, "rb"), + open(result["file_name"], "rb"), content_type="application/pdf", filename=file_name, ) @@ -228,6 +243,11 @@ def parse(self, source: str, **kwargs) -> jinja2.nodes.Template: def compile( self, main_file_name: str, src_dir: str, data: DataSet, title_pattern: str, *, split_output: bool ) -> list[FileWithData]: + """ + Compile a dataset with Jinja HTML template from `main_file_name` into HTML(s). + `src_dir` should be an empty directory where the resulting HTML files will be saved. + `title_pattern` is compiled with Jinja and is used to fill the head `title` element. + """ lookups = find_lookup_tables(self.vfs.values()) tpl = self.env.get_template(main_file_name) _title_pattern = self.env.from_string(title_pattern) @@ -244,7 +264,7 @@ def compile( of.write(html_header(title=title)) success = self._write_render_or_error(of, tpl, row_copy, idx, lookups) of.write(html_footer()) - sources.append((src_name, row_copy, success)) + sources.append(FileWithData(file_name=src_name, data=row_copy, success=success)) else: # Render title if we have any data, but supply the row only if it is singular. row_copy = dict(data[0]) if len(data) == 1 else None @@ -258,7 +278,7 @@ def compile( for idx, row in enumerate(data, start=1): success &= self._write_render_or_error(of, tpl, dict(row), idx, lookups) of.write(html_footer()) - sources.append((src_name, row_copy, success)) + sources.append(FileWithData(file_name=src_name, data=row_copy, success=success)) return sources def _write_render_or_error(self, of, tpl: jinja2.Template, row: dict, idx: int, lookups: dict) -> bool: @@ -331,9 +351,9 @@ def compile(self, sources: list[FileWithData], result_dir: str) -> list[FileWith for sheet_file in self.stylesheets ] results: list[FileWithData] = [] - for source, row, template_success in sources: + for source in sources: pdf_html = weasyprint.HTML( - filename=source, + filename=source["file_name"], base_url=LOCAL_FILE_URI_PREFIX, url_fetcher=url_fetcher, ) @@ -344,9 +364,9 @@ def compile(self, sources: list[FileWithData], result_dir: str) -> list[FileWith if pdf is None: raise RuntimeError("Unexpectedly None result") - dst_base = os.path.splitext(os.path.basename(source))[0] + dst_base = os.path.splitext(os.path.basename(source["file_name"]))[0] dst_name = os.path.join(result_dir, dst_base + ".pdf") - results.append((dst_name, row, template_success)) + results.append(FileWithData(file_name=dst_name, data=source["data"], success=source["success"])) with open(dst_name, "wb") as of: of.write(pdf) From 15559107683511c8d0f3f8d2a284db58418280e1 Mon Sep 17 00:00:00 2001 From: Jyrki Launonen Date: Thu, 1 Oct 2026 19:43:47 +0300 Subject: [PATCH 5/5] fix(emprinten): Ensure identifiers start with a proper character Numbers cannot start identifiers, so force such names to start with an underscore. --- kompassi/emprinten/files.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/kompassi/emprinten/files.py b/kompassi/emprinten/files.py index 88ac1b2d7..b59984853 100644 --- a/kompassi/emprinten/files.py +++ b/kompassi/emprinten/files.py @@ -31,6 +31,7 @@ def make_lut(file: Pathlike, encoding: str) -> Lut: non_word_char = re.compile(r"\W+") +non_first_char = re.compile(r"^([^a-zA-Z_])") def make_name(name: str) -> str: @@ -45,8 +46,10 @@ def make_name(name: str) -> str: 'foo_bar' >>> make_name("Barb!!") 'barb' + >>> make_name("0foo") + '_0foo' """ - return non_word_char.sub("_", name).strip("_").lower() + return non_first_char.sub(r"_\1", non_word_char.sub("_", name).strip("_")).lower() def parse_header_names(cols: list[str]) -> list[str]: