From 4299d79f31295c6c0bfd8481c9ca9b1dfa071230 Mon Sep 17 00:00:00 2001 From: caufieldjh Date: Tue, 29 Sep 2026 15:56:03 -0400 Subject: [PATCH 01/14] Read a PDF's embedded /Title in PDFExtractor PDFExtractor.extract_title returns the document information /Title, or None when it is absent, unreadable, or a placeholder that authoring tools stamp in ("Microsoft Word - x.doc", bare filenames, "Untitled"). Nothing calls it yet. It is the offline fallback for #93. Co-Authored-By: Claude Opus 5.5 --- .../etl/extract/pdf.py | 65 +++++++++++++++++ tests/test_pdf_url_title.py | 73 +++++++++++++++++++ 2 files changed, 138 insertions(+) create mode 100644 tests/test_pdf_url_title.py diff --git a/src/linkml_reference_validator/etl/extract/pdf.py b/src/linkml_reference_validator/etl/extract/pdf.py index e89e54b..385b8d9 100644 --- a/src/linkml_reference_validator/etl/extract/pdf.py +++ b/src/linkml_reference_validator/etl/extract/pdf.py @@ -6,6 +6,7 @@ import io import logging +import re from typing import Optional, Protocol, Union from linkml_reference_validator.etl.extract.base import Extractor, ExtractorRegistry @@ -20,6 +21,10 @@ def extract_text(self, data: bytes) -> str: """Return extracted plain text for the given PDF bytes.""" ... + def extract_title(self, data: bytes) -> Optional[str]: + """Return the embedded document title, or None if there is none.""" + ... + class PypdfBackend: """Default PDF backend using ``pypdf`` (BSD-licensed, pure-python). @@ -35,6 +40,57 @@ def extract_text(self, data: bytes) -> str: reader = PdfReader(io.BytesIO(data)) return "\n\n".join(page.extract_text() or "" for page in reader.pages) + def extract_title(self, data: bytes) -> Optional[str]: + """Read ``/Title`` from the document information dictionary. + + Examples: + >>> PypdfBackend().extract_title(b"%PDF-1.4 truncated") is None + True + """ + from pypdf import PdfReader + from pypdf.errors import PyPdfError + + try: # external data: a PDF can carry text yet have a broken trailer + metadata = PdfReader(io.BytesIO(data)).metadata + except PyPdfError as e: + logger.debug(f"Could not read PDF metadata: {e}") + return None + if metadata is None or metadata.title is None: + return None + return str(metadata.title) + + +# Authoring tools stamp these into /Title in place of a real one. +_PLACEHOLDER_TITLE = re.compile( + r"^(untitled(\s+document)?|microsoft (word|powerpoint) - .*|.*\.(pdf|docx?|rtf|odt|tex|indd))$", + re.IGNORECASE, +) + + +def clean_pdf_title(title: Optional[str]) -> Optional[str]: + """Return ``title`` stripped, or None if it is empty or a known placeholder. + + Examples: + >>> clean_pdf_title(" Canine Distemper in Dogs ") + 'Canine Distemper in Dogs' + >>> clean_pdf_title("Microsoft Word - draft_v3.doc") is None + True + >>> clean_pdf_title("paper.pdf") is None + True + >>> clean_pdf_title("Untitled") is None + True + >>> clean_pdf_title("") is None + True + >>> clean_pdf_title(None) is None + True + """ + if title is None: + return None + title = " ".join(title.split()) + if not title or _PLACEHOLDER_TITLE.match(title): + return None + return title + _BACKENDS: dict[str, type] = { "pypdf": PypdfBackend, @@ -72,3 +128,12 @@ def extract( raise TypeError("PDF extraction requires bytes, not decoded text") text = self._backend.extract_text(data) return text if text and text.strip() else None + + def extract_title(self, data: bytes) -> Optional[str]: + """Return the PDF's embedded title, or None if absent or a placeholder. + + Examples: + >>> PDFExtractor().extract_title(b"%PDF-1.4 truncated") is None + True + """ + return clean_pdf_title(self._backend.extract_title(data)) diff --git a/tests/test_pdf_url_title.py b/tests/test_pdf_url_title.py new file mode 100644 index 0000000..704cea2 --- /dev/null +++ b/tests/test_pdf_url_title.py @@ -0,0 +1,73 @@ +"""A PDF fetched by URL should carry a real title where one can be found (issue #93). + +Before this, ``URLSource`` set ``title=url`` on every PDF, so a consumer that +reads ``title`` as bibliographic metadata got the address back as the name. +The recovery order is: a publisher landing page found by a documented rule +(``citation_title``), then the PDF's embedded ``/Title``, then the URL. +""" + +import io +from unittest.mock import patch + +import pytest +from pypdf import PdfWriter + +from linkml_reference_validator.etl.extract.pdf import PDFExtractor +from linkml_reference_validator.etl.sources.url import URLSource +from linkml_reference_validator.models import ReferenceValidationConfig + +JSTAGE_PDF = "https://www.jstage.jst.go.jp/article/jvms/64/1/64_1_1/_pdf/-char/ja" +JSTAGE_ARTICLE = "https://www.jstage.jst.go.jp/article/jvms/64/1/64_1_1/_article/-char/ja" + + +def _pdf(title=None) -> bytes: + """Build a blank one-page PDF, optionally with an embedded ``/Title``.""" + writer = PdfWriter() + writer.add_blank_page(width=100, height=100) + if title is not None: + writer.add_metadata({"/Title": title}) + buf = io.BytesIO() + writer.write(buf) + return buf.getvalue() + + +@pytest.fixture +def config(tmp_path): + return ReferenceValidationConfig(cache_dir=tmp_path / "cache", rate_limit_delay=0.0) + + +def _fetch(url, responses, config): + """Fetch ``url`` with the acquirer answering from ``responses`` (url -> (bytes, ctype)).""" + with patch("linkml_reference_validator.etl.sources.url.ContentAcquirer") as MockAcquirer: + MockAcquirer.return_value.fetch_bytes.side_effect = ( + lambda u, _config: responses.get(u, (None, None)) + ) + result = URLSource().fetch(url, config) + requested = [c.args[0] for c in MockAcquirer.return_value.fetch_bytes.call_args_list] + return result, requested + + +# --- embedded PDF metadata ------------------------------------------------- + + +def test_extractor_reads_embedded_title(): + assert PDFExtractor().extract_title(_pdf("Canine Distemper in Dogs")) == ( + "Canine Distemper in Dogs" + ) + + +def test_extractor_title_absent_is_none(): + assert PDFExtractor().extract_title(_pdf()) is None + + +@pytest.mark.parametrize( + "junk", + ["", " ", "untitled", "Untitled", "Microsoft Word - draft_v3.doc", "paper.pdf", "manuscript.docx"], +) +def test_extractor_ignores_placeholder_titles(junk): + """Authoring tools stamp filenames and placeholders into /Title. Those are not titles.""" + assert PDFExtractor().extract_title(_pdf(junk)) is None + + +def test_extractor_title_on_unparseable_bytes_is_none(): + assert PDFExtractor().extract_title(b"%PDF-1.4 not really a pdf") is None From 5d89f71fdfa0bad47b46ab7f4d01ea8701c4dbf5 Mon Sep 17 00:00:00 2001 From: caufieldjh Date: Tue, 29 Sep 2026 15:56:04 -0400 Subject: [PATCH 02/14] Prefer citation_title over for URL pages <title> is often the site or journal name. The citation_title meta tag (Highwire / Google Scholar) names the article itself, so _extract_title now checks it first. Attribute order does not matter and entities are decoded. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> --- .../etl/sources/url.py | 36 ++++++++++++++++++- tests/test_pdf_url_title.py | 13 +++++++ 2 files changed, 48 insertions(+), 1 deletion(-) diff --git a/src/linkml_reference_validator/etl/sources/url.py b/src/linkml_reference_validator/etl/sources/url.py index 384e945..c877b42 100644 --- a/src/linkml_reference_validator/etl/sources/url.py +++ b/src/linkml_reference_validator/etl/sources/url.py @@ -10,6 +10,7 @@ True """ +import html import logging import re from typing import Optional @@ -21,6 +22,10 @@ logger = logging.getLogger(__name__) +_META_TAG = re.compile(r"<meta\b[^>]*>", re.IGNORECASE) +_META_NAME = re.compile(r"""\bname\s*=\s*["']citation_title["']""", re.IGNORECASE) +_META_CONTENT = re.compile(r"""\bcontent\s*=\s*(["'])(.*?)\1""", re.IGNORECASE | re.DOTALL) + @ReferenceSourceRegistry.register class URLSource(ReferenceSource): @@ -101,6 +106,27 @@ def fetch( content_type="url", ) + @staticmethod + def _citation_title(content: str) -> Optional[str]: + """Return the ``citation_title`` meta tag's content, if present. + + Examples: + >>> URLSource._citation_title('<meta name="citation_title" content="A & B">') + 'A & B' + >>> URLSource._citation_title("<meta content='Reversed' name='citation_title'>") + 'Reversed' + >>> URLSource._citation_title("<title>Only a title") is None + True + """ + for tag in _META_TAG.findall(content): + if _META_NAME.search(tag): + match = _META_CONTENT.search(tag) + if match: + title = " ".join(html.unescape(match.group(2)).split()) + if title: + return title + return None + def _decode(self, data: bytes, content_type: str) -> str: """Decode HTML/text bytes using the content-type charset, defaulting to UTF-8. @@ -123,7 +149,7 @@ def _decode(self, data: bytes, content_type: str) -> str: def _extract_title(self, content: str, url: str) -> str: """Extract title from HTML content or use URL. - Looks for tag in HTML. Falls back to URL. + Prefers a ``citation_title`` meta tag, then the <title> tag. Falls back to URL. Args: content: Page content @@ -136,9 +162,17 @@ def _extract_title(self, content: str, url: str) -> str: >>> source = URLSource() >>> source._extract_title("<html><title>Page Title", "https://x.com") 'Page Title' + >>> source._extract_title( + ... 'Journal | Home', + ... "https://x.com") + 'Article' >>> source._extract_title("plain text", "https://example.com/doc.txt") 'https://example.com/doc.txt' """ + citation_title = self._citation_title(content) + if citation_title: + return citation_title + # Look for HTML title tag (simple regex, no BeautifulSoup) match = re.search(r"]*>([^<]+)", content, re.IGNORECASE) if match: diff --git a/tests/test_pdf_url_title.py b/tests/test_pdf_url_title.py index 704cea2..42cbdd5 100644 --- a/tests/test_pdf_url_title.py +++ b/tests/test_pdf_url_title.py @@ -71,3 +71,16 @@ def test_extractor_ignores_placeholder_titles(junk): def test_extractor_title_on_unparseable_bytes_is_none(): assert PDFExtractor().extract_title(b"%PDF-1.4 not really a pdf") is None + + +# --- citation_title on ordinary HTML fetches ------------------------------- + + +def test_html_prefers_citation_title_over_title_tag(config): + url = "https://example.org/article/1" + page = ( + b"Example Journal | Home" + b'' + ) + result, _ = _fetch(url, {url: (page, "text/html")}, config) + assert result.title == "Actual Article Title" From c71b9782ca85429cdea1c63dd9a7a634596041d2 Mon Sep 17 00:00:00 2001 From: caufieldjh Date: Tue, 29 Sep 2026 15:56:05 -0400 Subject: [PATCH 03/14] Recover a title for PDFs fetched by URL URLSource set title=url on every PDF, so consumers that read title as bibliographic metadata got the address back as the paper's name. The PDF branch now tries, in order: the citation_title of a publisher landing page found by LANDING_PAGE_RULES (J-STAGE _pdf -> _article to start), then the PDF's embedded /Title, then the URL. The URL fallback is kept, so title == url still means no title was found. Closes #93 Co-Authored-By: Claude Opus 5.5 --- .../etl/sources/url.py | 47 +++++++++- tests/test_pdf_url_title.py | 88 +++++++++++++++++++ 2 files changed, 134 insertions(+), 1 deletion(-) diff --git a/src/linkml_reference_validator/etl/sources/url.py b/src/linkml_reference_validator/etl/sources/url.py index c877b42..3f617e7 100644 --- a/src/linkml_reference_validator/etl/sources/url.py +++ b/src/linkml_reference_validator/etl/sources/url.py @@ -22,6 +22,14 @@ logger = logging.getLogger(__name__) +# Publishers that serve a PDF at one URL and its metadata page at a predictable +# sibling. Each rule is (pattern, replacement) for ``re.sub`` on the PDF URL. +# The landing page is consulted only for its ``citation_title`` meta tag. +LANDING_PAGE_RULES: list[tuple[str, str]] = [ + # J-STAGE: .../
/_pdf[/-char/ja] -> .../
/_article[/-char/ja] + (r"^(https?://www\.jstage\.jst\.go\.jp/article/.+)/_pdf(/.*)?$", r"\1/_article\2"), +] + _META_TAG = re.compile(r"]*>", re.IGNORECASE) _META_NAME = re.compile(r"""\bname\s*=\s*["']citation_title["']""", re.IGNORECASE) _META_CONTENT = re.compile(r"""\bcontent\s*=\s*(["'])(.*?)\1""", re.IGNORECASE | re.DOTALL) @@ -90,7 +98,7 @@ def fetch( ) return ReferenceContent( reference_id=f"url:{url}", - title=url, + title=self._recover_pdf_title(url, data, config), content=text, content_type="full_text_pdf" if text else "unavailable", full_text_url=url, @@ -106,6 +114,43 @@ def fetch( content_type="url", ) + def _recover_pdf_title( + self, url: str, data: bytes, config: ReferenceValidationConfig + ) -> str: + """Find a real title for a PDF, falling back to its URL. + + Tries, in order: the ``citation_title`` of a landing page found by + ``LANDING_PAGE_RULES``, then the PDF's embedded ``/Title``. When both + fail the URL is returned, so ``title == url`` still means none was found. + """ + landing = self._landing_page_url(url) + if landing is not None: + page, content_type = ContentAcquirer().fetch_bytes(landing, config) + if page is not None: + title = self._citation_title(self._decode(page, (content_type or "").lower())) + if title: + return title + logger.debug(f"No citation_title at landing page {landing} for {url}") + + embedded = PDFExtractor(backend=config.pdf_backend).extract_title(data) + return embedded or url + + @staticmethod + def _landing_page_url(url: str) -> Optional[str]: + """Return the landing page for a PDF URL, if a rule in ``LANDING_PAGE_RULES`` matches. + + Examples: + >>> URLSource._landing_page_url( + ... "https://www.jstage.jst.go.jp/article/jvms/64/1/64_1_1/_pdf/-char/ja") + 'https://www.jstage.jst.go.jp/article/jvms/64/1/64_1_1/_article/-char/ja' + >>> URLSource._landing_page_url("https://example.org/paper.pdf") is None + True + """ + for pattern, replacement in LANDING_PAGE_RULES: + if re.match(pattern, url): + return re.sub(pattern, replacement, url) + return None + @staticmethod def _citation_title(content: str) -> Optional[str]: """Return the ``citation_title`` meta tag's content, if present. diff --git a/tests/test_pdf_url_title.py b/tests/test_pdf_url_title.py index 42cbdd5..ea07530 100644 --- a/tests/test_pdf_url_title.py +++ b/tests/test_pdf_url_title.py @@ -73,6 +73,94 @@ def test_extractor_title_on_unparseable_bytes_is_none(): assert PDFExtractor().extract_title(b"%PDF-1.4 not really a pdf") is None +def test_pdf_url_uses_embedded_title(config): + url = "https://example.org/files/paper.pdf" + result, _ = _fetch(url, {url: (_pdf("A Real Paper Title"), "application/pdf")}, config) + assert result is not None + assert result.title == "A Real Paper Title" + + +def test_pdf_url_without_any_title_falls_back_to_url(config): + """With nothing to recover, the URL stays, so ``title == url`` still means 'none found'.""" + url = "https://example.org/files/paper.pdf" + result, requested = _fetch(url, {url: (_pdf(), "application/pdf")}, config) + assert result is not None + assert result.title == url + assert requested == [url], "no landing-page rule matches, so nothing else is fetched" + + +# --- landing page discovery ------------------------------------------------ + + +@pytest.mark.parametrize( + "pdf_url,landing", + [ + (JSTAGE_PDF, JSTAGE_ARTICLE), + ( + "https://www.jstage.jst.go.jp/article/jvms/64/1/64_1_1/_pdf", + "https://www.jstage.jst.go.jp/article/jvms/64/1/64_1_1/_article", + ), + ("https://example.org/files/paper.pdf", None), + ], +) +def test_landing_page_rule(pdf_url, landing): + assert URLSource._landing_page_url(pdf_url) == landing + + +def test_jstage_pdf_takes_citation_title_from_article_page(config): + article = ( + b'J-STAGE' + b'' + b"" + ) + result, requested = _fetch( + JSTAGE_PDF, + { + JSTAGE_PDF: (_pdf("Microsoft Word - 64_1_1.doc"), "application/pdf"), + JSTAGE_ARTICLE: (article, "text/html; charset=utf-8"), + }, + config, + ) + assert result is not None + assert result.title == "Feline Infectious Peritonitis & Its Diagnosis" + assert result.content_type == "unavailable" # blank page, no text: unchanged behavior + assert requested == [JSTAGE_PDF, JSTAGE_ARTICLE] + + +def test_landing_page_title_beats_embedded_title(config): + """The publisher's page is authoritative. Embedded /Title is often stale.""" + article = b'' + result, _ = _fetch( + JSTAGE_PDF, + { + JSTAGE_PDF: (_pdf("Submitted Draft Title"), "application/pdf"), + JSTAGE_ARTICLE: (article, "text/html"), + }, + config, + ) + assert result.title == "The Published Title" + + +def test_landing_page_without_citation_title_falls_through_to_embedded(config): + """A bare on a landing page is often the site name, so it is not used.""" + article = b"<html><head><title>J-STAGE" + result, _ = _fetch( + JSTAGE_PDF, + { + JSTAGE_PDF: (_pdf("Embedded Title"), "application/pdf"), + JSTAGE_ARTICLE: (article, "text/html"), + }, + config, + ) + assert result.title == "Embedded Title" + + +def test_landing_page_fetch_failure_falls_through_to_url(config): + result, _ = _fetch(JSTAGE_PDF, {JSTAGE_PDF: (_pdf(), "application/pdf")}, config) + assert result is not None + assert result.title == JSTAGE_PDF + + # --- citation_title on ordinary HTML fetches ------------------------------- From 684138f103c33b1f04cac9afeca6aaaa7195fa4e Mon Sep 17 00:00:00 2001 From: caufieldjh Date: Tue, 29 Sep 2026 15:58:04 -0400 Subject: [PATCH 04/14] Sanitize HTML fetched by URLSource before caching URLSource stored the response body verbatim, so page scripts, comments and tag attributes (signed asset URLs, API-key parameters) went into caches that projects commit to public repositories (#92). HTML now passes through sanitize_html: script, style, noscript and template elements and comments are dropped, and every attribute except rowspan, colspan and scope is removed. meta, link and base are dropped too, since they hold nothing but attributes and would remain only as empty tags. Body markup and text are kept. Whitespace left by removed elements is collapsed, outside
, so the result is stable under a
second pass. Plain text and XML are stored as fetched.

The title is read before sanitizing, since citation_title lives in a
 attribute.

Refs #92

Co-Authored-By: Claude Opus 5.5 
---
 .../etl/extract/html.py                       |  46 +++++-
 .../etl/sources/url.py                        |  12 +-
 tests/test_url_html_sanitize.py               | 133 ++++++++++++++++++
 3 files changed, 187 insertions(+), 4 deletions(-)
 create mode 100644 tests/test_url_html_sanitize.py

diff --git a/src/linkml_reference_validator/etl/extract/html.py b/src/linkml_reference_validator/etl/extract/html.py
index 8ee76d4..526abc7 100644
--- a/src/linkml_reference_validator/etl/extract/html.py
+++ b/src/linkml_reference_validator/etl/extract/html.py
@@ -10,7 +10,7 @@
 import re
 from typing import Optional, Union
 
-from bs4 import BeautifulSoup, Tag  # type: ignore
+from bs4 import BeautifulSoup, Comment, Tag  # type: ignore
 
 from linkml_reference_validator.etl.extract.base import Extractor, ExtractorRegistry
 
@@ -35,6 +35,50 @@
 )
 
 
+#: Elements that never hold reference text. Page scripts routinely carry signed
+#: asset URLs and API-key parameters, which must not reach a committed cache.
+#: ``meta``, ``link`` and ``base`` hold nothing but attributes, so once those
+#: are stripped they would remain only as empty tags.
+NON_CONTENT_TAGS = ("script", "style", "noscript", "template", "meta", "link", "base")
+
+#: The only attributes that carry meaning for a quoted excerpt: table structure.
+KEPT_ATTRIBUTES = frozenset({"rowspan", "colspan", "scope"})
+
+
+def sanitize_html(content: str) -> str:
+    """Strip an HTML page to its body markup and text, for caching.
+
+    Drops ``NON_CONTENT_TAGS`` and comments, and every attribute not in
+    ``KEPT_ATTRIBUTES``. Markup is kept, unlike :class:`HTMLExtractor`, which
+    flattens to plain text; this is for ``url:`` pages stored as HTML.
+
+    Examples:
+        >>> sanitize_html('

Hi

') + '

Hi

' + >>> sanitize_html('T') + 'T' + >>> sanitize_html('A') + 'A' + >>> sanitize_html("
x    y
") + '
x    y
' + """ + soup = BeautifulSoup(content, "html.parser") + for tag in soup.find_all(NON_CONTENT_TAGS): + if not tag.decomposed: # already gone with an enclosing non-content tag + tag.decompose() + for comment in soup.find_all(string=lambda text: isinstance(text, Comment)): + comment.extract() + for tag in soup.find_all(True): + tag.attrs = {k: v for k, v in tag.attrs.items() if k in KEPT_ATTRIBUTES} + # Removed elements leave runs of indentation behind. Collapse each + # whitespace-only gap to one break so a second pass changes nothing. + soup.smooth() + for text in soup.find_all(string=True): + if text.strip() or text in ("\n", " ") or text.find_parent("pre"): + continue + text.replace_with("\n" if "\n" in text else " ") + return str(soup) + @ExtractorRegistry.register class HTMLExtractor(Extractor): """Extract readable text from HTML markup, as bytes or as a string. diff --git a/src/linkml_reference_validator/etl/sources/url.py b/src/linkml_reference_validator/etl/sources/url.py index 3f617e7..00a3dce 100644 --- a/src/linkml_reference_validator/etl/sources/url.py +++ b/src/linkml_reference_validator/etl/sources/url.py @@ -17,7 +17,8 @@ from linkml_reference_validator.models import ReferenceContent, ReferenceValidationConfig from linkml_reference_validator.etl.sources.base import ReferenceSource, ReferenceSourceRegistry -from linkml_reference_validator.etl.acquire import ContentAcquirer +from linkml_reference_validator.etl.acquire import ContentAcquirer, sniff_format +from linkml_reference_validator.etl.extract.html import sanitize_html from linkml_reference_validator.etl.extract.pdf import PDFExtractor logger = logging.getLogger(__name__) @@ -39,8 +40,10 @@ class URLSource(ReferenceSource): """Fetch reference content from web URLs. - Fetches HTML and plain text content. HTML is returned as-is (no parsing). - Content is cached to disk like other sources. + Fetches HTML and plain text content. HTML keeps its markup but is + sanitized (see :func:`sanitize_html`) so page scripts and attributes do not + reach the cache. Plain text and XML are stored as fetched. Content is cached + to disk like other sources. Examples: >>> source = URLSource() @@ -105,7 +108,10 @@ def fetch( ) content = self._decode(data, content_type_header) + # Before sanitizing: citation_title lives in a attribute. title = self._extract_title(content, url) + if "html" in content_type_header or sniff_format(data) == "html": + content = sanitize_html(content) return ReferenceContent( reference_id=f"url:{url}", diff --git a/tests/test_url_html_sanitize.py b/tests/test_url_html_sanitize.py new file mode 100644 index 0000000..cac050c --- /dev/null +++ b/tests/test_url_html_sanitize.py @@ -0,0 +1,133 @@ +"""HTML fetched by ``URLSource`` is sanitized before it is cached (issue #92). + +``URLSource`` used to store the response body verbatim, so page scripts, +comments and tag attributes (signed asset URLs, API-key parameters) went +into caches that projects commit to public repositories. Now HTML keeps its +body markup and text, loses ``script``/``style``/``noscript``/``template`` +and comments, drops ``meta``/``link``/``base`` (they hold nothing but +attributes), and keeps only the table-structural attributes ``rowspan``, +``colspan`` and ``scope``. Plain text and XML are stored as before. +""" + +from unittest.mock import patch + +import pytest + +from linkml_reference_validator.etl.extract.html import sanitize_html +from linkml_reference_validator.etl.sources.url import URLSource +from linkml_reference_validator.models import ReferenceValidationConfig + +PAGE = b""" + + + Journal | Home + + + + + + + + + +

Patients showed lactic acidosis.

+ + + + + +
GeneFinding
POLGAB
+ + +""" + + +@pytest.fixture +def config(tmp_path): + return ReferenceValidationConfig(cache_dir=tmp_path / "cache", rate_limit_delay=0.0) + + +def _fetch(url, body, content_type, config): + with patch("linkml_reference_validator.etl.sources.url.ContentAcquirer") as MockAcquirer: + MockAcquirer.return_value.fetch_bytes.return_value = (body, content_type) + return URLSource().fetch(url, config) + + +@pytest.fixture +def cached(config): + result = _fetch("https://example.org/article/1", PAGE, "text/html; charset=utf-8", config) + assert result is not None + return result + + +@pytest.mark.parametrize( + "gone", + [ + "SECRET-KEY-123", + "Signature=abc", + ".hidden", + "tracker.example.org", + "Template body", + "Patients showed lactic acidosis.

" in cached.content + assert "POLG" not in cached.content # rowspan kept on it + assert "POLG" in cached.content + assert cached.content_type == "url" + + +def test_title_is_read_before_attributes_are_stripped(cached): + """citation_title lives in a attribute, so it must be read from the raw page.""" + assert cached.title == "Mitochondrial Disease in Children" + + +def test_html_is_sniffed_without_a_content_type(config): + result = _fetch("https://example.org/a", PAGE, None, config) + assert "SECRET-KEY-123" not in result.content + + +def test_plain_text_is_left_alone(config): + body = b'Plain notes. ' + result = _fetch("https://example.org/notes.txt", body, "text/plain", config) + assert result.content == body.decode() + + +def test_xml_is_left_alone(config): + body = b'T' + result = _fetch("https://example.org/r.xml", body, "application/xml", config) + assert result.content == body.decode() + + +def test_sanitize_html_is_idempotent(): + once = sanitize_html(PAGE.decode()) + assert sanitize_html(once) == once From 9fc8b3f98f35318b297d9dac02b073acc59861e4 Mon Sep 17 00:00:00 2001 From: caufieldjh Date: Tue, 29 Sep 2026 16:39:09 -0400 Subject: [PATCH 05/14] Re-fetch url: entries written before the URLSource fixes Sanitizing HTML (#92) and recovering PDF titles (#93) change what URLSource writes, and neither rewrites what it already wrote. Those entries carry a current extractor_version, so raw page scripts and URL-as-title PDFs would be served forever. url: entries now carry url_source_version, stamped only on a fresh fetch, the way html_full_text_version is. A missing or older stamp makes the entry stale, so it is re-fetched once. Offline, the old entry is still served and not rewritten. Other sources are unaffected. The horizontal-rule frontmatter test saved a url: entry by hand with no stamp, which now reads as stale by design. It now carries the stamp a fresh fetch writes, which still sits below reference_id and so still tests the truncation it was written for. Closes #92 Co-Authored-By: Claude Opus 5.5 --- docs/troubleshooting.md | 10 ++ .../etl/reference_fetcher.py | 26 ++++ tests/test_reference_fetcher.py | 10 +- tests/test_url_source_cache_version.py | 125 ++++++++++++++++++ 4 files changed, 170 insertions(+), 1 deletion(-) create mode 100644 tests/test_url_source_cache_version.py diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index 5488a65..bacbce4 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -902,6 +902,16 @@ rather than as `unavailable`, so it falls outside this stamp and is not re-tested. Clear such entries by hand if you were running that setting before this version. +`url:` entries carry their own stamp, `url_source_version: 1`. Version 1 is +two URLSource changes: HTML is sanitized before caching, so page scripts, +comments and attributes stop reaching the cache (issue #92), and a PDF's title +is recovered from a publisher landing page or its embedded metadata rather than +set to its URL (issue #93). A `url:` entry with a missing or older stamp is +re-fetched once on the next validation fetch. If the page cannot be reached, +the old entry is still served and is not rewritten, so a later run retries. +Other sources are unaffected. With `trust_cached_entries` set, old entries are +served as they are and are not refreshed. + **A cache-wide refresh adds one line to every enriched entry.** Both index providers now set `access_type` where they previously left it unset, so a refreshed public entry gains a `full_text_access_type: open` line. It means the diff --git a/src/linkml_reference_validator/etl/reference_fetcher.py b/src/linkml_reference_validator/etl/reference_fetcher.py index af0a6d1..177205d 100644 --- a/src/linkml_reference_validator/etl/reference_fetcher.py +++ b/src/linkml_reference_validator/etl/reference_fetcher.py @@ -113,6 +113,15 @@ #: fix to any source will want, and it is still a 3% refresh against 100%. ABSENT_CONTENT_CACHE_VERSION = 1 +#: Version what URLSource writes, scoped to ``url:`` entries. Version 1: HTML +#: is sanitized before caching, so page scripts and attributes stop reaching +#: committed caches (#92), and a PDF's title is recovered rather than set to its +#: URL (#93). Neither fix rewrites an entry already written, and both kinds of +#: entry carry a current ``extractor_version``, so without this they would be +#: served as they are forever. Stamped from ``reference.metadata``, like the +#: HTML stamp, and only on a fresh URLSource fetch. +URL_SOURCE_CACHE_VERSION = 1 + #: A cache file's frontmatter delimiter: a line that is exactly ``---``. #: Splitting on the bare string instead lets any *value* containing ``---`` - a #: URL reference_id, a title - truncate the block, which loses every field after @@ -388,6 +397,11 @@ def fetch_with_provenance( content.metadata or {}, xml_extraction_version=XML_EXTRACTION_CACHE_VERSION ) + if content and normalized_reference_id.startswith("url:"): + content.metadata = dict( + content.metadata or {}, url_source_version=URL_SOURCE_CACHE_VERSION + ) + if content and self.config.fetch_full_text and self.needs_full_text(content): content = self._enrich_with_full_text(content) @@ -1349,6 +1363,9 @@ def _save_to_disk( xml_version = (reference.metadata or {}).get("xml_extraction_version") if reference.content_type == "full_text_xml" and type(xml_version) is int: lines.append(f"xml_extraction_version: {xml_version}") + url_version = (reference.metadata or {}).get("url_source_version") + if type(url_version) is int: + lines.append(f"url_source_version: {url_version}") # Written from the constant, unlike the HTML and XML stamps, which come # from `reference.metadata` so a metadata-only rewrite preserves the # original. The invariant that makes that safe: nothing reaches a save @@ -1703,6 +1720,13 @@ def _is_stale_cache_entry(cls, content_text: str) -> bool: if type(xml_version) is not int or xml_version < XML_EXTRACTION_CACHE_VERSION: return True + if isinstance(metadata, dict) and str(metadata.get("reference_id", "")).startswith( + "url:" + ): + url_version = metadata.get("url_source_version") + if type(url_version) is not int or url_version < URL_SOURCE_CACHE_VERSION: + return True + # Scoped to `unavailable`, which leaves one gap: with # `source_extra_fields["PMID"]` configured, a record with no abstract is # stored as `summary` carrying the extra-fields blob, so an @@ -1762,6 +1786,8 @@ def _load_markdown_format( metadata["xml_extraction_version"] = frontmatter["xml_extraction_version"] if "html_full_text_version" in frontmatter: metadata["html_full_text_version"] = frontmatter["html_full_text_version"] + if "url_source_version" in frontmatter: + metadata["url_source_version"] = frontmatter["url_source_version"] if "extra_fields_captured" in frontmatter: metadata["extra_fields_captured"] = frontmatter["extra_fields_captured"] diff --git a/tests/test_reference_fetcher.py b/tests/test_reference_fetcher.py index aa76162..bb47195 100644 --- a/tests/test_reference_fetcher.py +++ b/tests/test_reference_fetcher.py @@ -1384,9 +1384,17 @@ def test_entry_whose_id_contains_a_horizontal_rule_is_not_perpetually_stale(fetc re-fetched, be rewritten with the same id, and read as unstamped again on every single run. """ + from linkml_reference_validator.etl.reference_fetcher import URL_SOURCE_CACHE_VERSION + reference_id = "url:https://example.com/a---b" + # Stamped as a fresh URLSource fetch would be. The stamp is written below + # reference_id, so a truncated block would hide it too. fetcher._save_to_disk( - ReferenceContent(reference_id=reference_id, content="Body text.") + ReferenceContent( + reference_id=reference_id, + content="Body text.", + metadata={"url_source_version": URL_SOURCE_CACHE_VERSION}, + ) ) assert not fetcher._is_stale_cache_entry(_cached_text(fetcher, reference_id)) diff --git a/tests/test_url_source_cache_version.py b/tests/test_url_source_cache_version.py new file mode 100644 index 0000000..34ebb91 --- /dev/null +++ b/tests/test_url_source_cache_version.py @@ -0,0 +1,125 @@ +"""``url:`` entries cached before the URLSource fixes must be re-fetched. + +Sanitizing HTML (#92) and recovering PDF titles (#93) change what URLSource +writes, and neither rewrites what it already wrote: raw page scripts and +URL-as-title entries carry the current ``extractor_version`` and would be +served forever. ``url_source_version`` is their own staleness stamp, scoped +the way ``html_full_text_version`` is, so only ``url:`` entries pay the refresh. +""" + +from __future__ import annotations + +from unittest.mock import patch + +import pytest + +from linkml_reference_validator.etl.reference_fetcher import ( + ABSENT_CONTENT_CACHE_VERSION, + EXTRACTOR_CACHE_VERSION, + URL_SOURCE_CACHE_VERSION, + ReferenceFetcher, +) +from linkml_reference_validator.models import ReferenceContent, ReferenceValidationConfig + +URL = "https://example.org/article/1" +RAW_PAGE = ( + 'Journal | Home' + '' + "

Patients showed lactic acidosis.

" +) + + +def _entry( + reference_id: str, content_type: str, *, url_version: int | None = None, body: str = "" +) -> str: + """A cache entry current in every respect except the stamp a test varies.""" + lines = [ + "---", + f"reference_id: {reference_id}", + f"extractor_version: {EXTRACTOR_CACHE_VERSION}", + ] + if url_version is not None: + lines.append(f"url_source_version: {url_version}") + if content_type == "unavailable": + lines.append(f"absent_content_version: {ABSENT_CONTENT_CACHE_VERSION}") + lines += [f"content_type: {content_type}", "---", "", "## Content", "", body] + return "\n".join(lines) + + +@pytest.mark.parametrize("content_type", ["url", "full_text_pdf", "unavailable"]) +def test_unstamped_url_entry_is_stale(content_type): + assert ReferenceFetcher._is_stale_cache_entry(_entry(f"url:{URL}", content_type)) + + +@pytest.mark.parametrize("content_type", ["url", "full_text_pdf", "unavailable"]) +def test_current_url_entry_is_fresh(content_type): + entry = _entry(f"url:{URL}", content_type, url_version=URL_SOURCE_CACHE_VERSION) + assert not ReferenceFetcher._is_stale_cache_entry(entry) + + +def test_a_newer_stamp_is_not_stale(): + entry = _entry(f"url:{URL}", "url", url_version=URL_SOURCE_CACHE_VERSION + 1) + assert not ReferenceFetcher._is_stale_cache_entry(entry) + + +def test_other_sources_are_untouched(): + assert not ReferenceFetcher._is_stale_cache_entry(_entry("PMID:12345", "abstract_only")) + + +@pytest.fixture +def fetcher(tmp_path): + return ReferenceFetcher(ReferenceValidationConfig(cache_dir=tmp_path, rate_limit_delay=0.0)) + + +def test_a_fresh_fetch_is_stamped_and_round_trips(fetcher, tmp_path): + with patch("linkml_reference_validator.etl.sources.url.ContentAcquirer") as MockAcquirer: + MockAcquirer.return_value.fetch_bytes.return_value = (RAW_PAGE.encode(), "text/html") + fetcher.fetch(f"url:{URL}") + + written = next(tmp_path.glob("*.md")).read_text() + assert f"url_source_version: {URL_SOURCE_CACHE_VERSION}" in written + assert not ReferenceFetcher._is_stale_cache_entry(written) + + reloaded = fetcher._load_from_disk(f"url:{URL}") + assert reloaded is not None + assert reloaded.metadata["url_source_version"] == URL_SOURCE_CACHE_VERSION + + +def test_only_a_fresh_fetch_is_certified(fetcher, tmp_path): + """Saving a url: entry that URLSource did not just produce must not stamp it.""" + fetcher._save_to_disk( + ReferenceContent( + reference_id=f"url:{URL}", content_type="url", title="t", content=RAW_PAGE + ) + ) + written = next(tmp_path.glob("*.md")).read_text() + assert "url_source_version" not in written + + +def test_old_raw_entry_is_refetched_and_sanitized(fetcher): + cache_path = fetcher.get_cache_path(f"url:{URL}") + cache_path.parent.mkdir(parents=True, exist_ok=True) + cache_path.write_text(_entry(f"url:{URL}", "url", body=RAW_PAGE)) + + with patch("linkml_reference_validator.etl.sources.url.ContentAcquirer") as MockAcquirer: + MockAcquirer.return_value.fetch_bytes.return_value = (RAW_PAGE.encode(), "text/html") + result = fetcher.fetch(f"url:{URL}") + + assert result is not None + assert "SECRET-KEY-123" not in result.content + assert "SECRET-KEY-123" not in cache_path.read_text() + + +def test_old_raw_entry_is_still_served_when_offline(fetcher): + """Out of date is better than not found. It is not re-saved, so it stays stale.""" + cache_path = fetcher.get_cache_path(f"url:{URL}") + cache_path.parent.mkdir(parents=True, exist_ok=True) + cache_path.write_text(_entry(f"url:{URL}", "url", body=RAW_PAGE)) + + with patch("linkml_reference_validator.etl.sources.url.ContentAcquirer") as MockAcquirer: + MockAcquirer.return_value.fetch_bytes.return_value = (None, None) + result = fetcher.fetch(f"url:{URL}") + + assert result is not None + assert "lactic acidosis" in result.content + assert ReferenceFetcher._is_stale_cache_entry(cache_path.read_text()) From 3bd8d1b60851bf3d05a82dcc033f1e324a6d79bd Mon Sep 17 00:00:00 2001 From: caufieldjh Date: Tue, 29 Sep 2026 16:58:30 -0400 Subject: [PATCH 06/14] Document URL titles and HTML sanitization validate-urls.md said url: pages were cached as raw content with the title taken from . It now describes sanitization, citation_title, and the PDF title order. use-local-files-and-urls.md said PDFs were unsupported; that holds for file: references only. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> --- docs/how-to/use-local-files-and-urls.md | 4 +- docs/how-to/validate-urls.md | 54 ++++++++++++++++++------- 2 files changed, 42 insertions(+), 16 deletions(-) diff --git a/docs/how-to/use-local-files-and-urls.md b/docs/how-to/use-local-files-and-urls.md index 09dab78..9535944 100644 --- a/docs/how-to/use-local-files-and-urls.md +++ b/docs/how-to/use-local-files-and-urls.md @@ -89,7 +89,7 @@ URLs are cached the same way as PMID and DOI references: ### Title Extraction -For HTML pages, the title is extracted from the `<title>` tag. For other content types, the URL itself is used as the title. +For HTML pages, the title is the `citation_title` meta tag when present, then the `<title>` tag. For PDFs, the validator tries a publisher landing page's `citation_title` and then the PDF's embedded title. When neither is found, and for other content types, the URL itself is used as the title. See [Validating URLs](validate-urls.md#titles). ### Example: Validating Against a Web Page @@ -145,6 +145,6 @@ Both file and URL references work in LinkML data files: ## Limitations -- **PDF files**: Not yet supported (planned for future) +- **Local PDF files**: Not yet supported as `file:` references. PDFs fetched by `url:` are supported. - **Authentication**: URLs requiring login are not supported - **Dynamic content**: JavaScript-rendered pages may not work diff --git a/docs/how-to/validate-urls.md b/docs/how-to/validate-urls.md index 2fb3ccc..79db50a 100644 --- a/docs/how-to/validate-urls.md +++ b/docs/how-to/validate-urls.md @@ -15,8 +15,8 @@ The linkml-reference-validator supports validating references that point to web When a reference field contains a URL, the validator: 1. Fetches the web page content -2. Extracts the page title from `<title>` tag (for HTML) -3. Caches the content for future validations +2. Extracts the page title (see [Titles](#titles)) +3. Sanitizes HTML and caches the content for future validations 4. Validates your supporting text against the page content ## URL Format @@ -79,14 +79,39 @@ When the validator encounters a URL reference, it: The fetcher stores: -- **Title**: Extracted from the `<title>` tag (for HTML pages) -- **Content**: The raw page content as received -- **Content type**: Marked as `url` to distinguish from other reference types +- **Title**: See [Titles](#titles) +- **Content**: Sanitized HTML for HTML pages; plain text and XML as received; + extracted text for PDFs +- **Content type**: `url` for pages, `full_text_pdf` for PDFs with text -Note: The validator stores raw page content without HTML-to-text conversion. -HTML tags remain in the cached file, and tag names can surface during -normalization. If validation fails because tags interrupt the text, consider -extracting plain text and validating against a local `file:` reference instead. +HTML is sanitized before it is cached, because caches are often committed to +public repositories and page scripts routinely carry signed asset URLs and API +keys. `<script>`, `<style>`, `<noscript>` and `<template>` elements and HTML +comments are removed. So are `<meta>`, `<link>` and `<base>`, which hold nothing +but attributes. Every other tag attribute is removed except `rowspan`, +`colspan` and `scope`, which carry table structure. Body markup and text are +kept. There is no HTML-to-text conversion, so tag names can still surface +during normalization. If validation fails because tags interrupt the text, +consider extracting plain text and validating against a local `file:` +reference instead. + +#### Titles + +For HTML pages the title is the `citation_title` meta tag when present (the +Highwire / Google Scholar convention, which names the article rather than the +site), then the `<title>` tag. + +For PDFs the validator tries, in order: + +1. The `citation_title` of the publisher's landing page, where a known rule + maps the PDF URL to it. Currently J-STAGE: `.../_pdf` becomes + `.../_article`. +2. The PDF's embedded `/Title` metadata, ignoring placeholders such as + `Microsoft Word - draft.doc`, bare filenames and `Untitled`. +3. The URL itself. + +So a PDF entry whose `title` equals its URL is one for which no title was +found. ### 3. Caching @@ -104,9 +129,9 @@ content_type: url ## Content <html> - <head> - <title>Chapter 3: Cell Structure and Function - + +Chapter 3: Cell Structure and Function + ... ``` @@ -145,9 +170,10 @@ URL validation is designed for static web pages. It may not work well with: ### Raw Content -The validator stores raw page content. For HTML pages: +The validator stores sanitized page markup, not extracted text. For HTML pages: -- HTML tags are preserved in the cache +- HTML tags are preserved in the cache, without attributes other than + `rowspan`, `colspan` and `scope` - The text normalization during validation handles most cases - Complex HTML layouts may require careful text extraction From 5a297176342d4528439ed89d1178bbfa1473fecc Mon Sep 17 00:00:00 2001 From: caufieldjh Date: Tue, 29 Sep 2026 17:23:13 -0400 Subject: [PATCH 07/14] Keep the PDF when its landing page cannot be fetched fetch_bytes lets requests exceptions through. The landing-page request in _recover_pdf_title was unguarded, so a timeout or refused connection on the J-STAGE article page failed the whole fetch, though the PDF and its text were already in hand. The title is best-effort; it now falls through to the embedded /Title, then the URL. Review of #98. Co-Authored-By: Claude Opus 5.5 --- .../etl/sources/url.py | 23 ++++++++++--- tests/test_pdf_url_title.py | 34 ++++++++++++++++--- 2 files changed, 48 insertions(+), 9 deletions(-) diff --git a/src/linkml_reference_validator/etl/sources/url.py b/src/linkml_reference_validator/etl/sources/url.py index 00a3dce..3a89572 100644 --- a/src/linkml_reference_validator/etl/sources/url.py +++ b/src/linkml_reference_validator/etl/sources/url.py @@ -15,6 +15,8 @@ import re from typing import Optional +import requests + from linkml_reference_validator.models import ReferenceContent, ReferenceValidationConfig from linkml_reference_validator.etl.sources.base import ReferenceSource, ReferenceSourceRegistry from linkml_reference_validator.etl.acquire import ContentAcquirer, sniff_format @@ -131,16 +133,27 @@ def _recover_pdf_title( """ landing = self._landing_page_url(url) if landing is not None: - page, content_type = ContentAcquirer().fetch_bytes(landing, config) - if page is not None: - title = self._citation_title(self._decode(page, (content_type or "").lower())) - if title: - return title + title = self._landing_page_title(landing, config) + if title: + return title logger.debug(f"No citation_title at landing page {landing} for {url}") embedded = PDFExtractor(backend=config.pdf_backend).extract_title(data) return embedded or url + def _landing_page_title( + self, landing: str, config: ReferenceValidationConfig + ) -> Optional[str]: + """Return the ``citation_title`` of a landing page, or None if it cannot be had.""" + try: # external system boundary: the title is best-effort, the PDF is not + page, content_type = ContentAcquirer().fetch_bytes(landing, config) + except requests.RequestException as e: + logger.debug(f"Landing page {landing} could not be fetched: {e}") + return None + if page is None: + return None + return self._citation_title(self._decode(page, (content_type or "").lower())) + @staticmethod def _landing_page_url(url: str) -> Optional[str]: """Return the landing page for a PDF URL, if a rule in ``LANDING_PAGE_RULES`` matches. diff --git a/tests/test_pdf_url_title.py b/tests/test_pdf_url_title.py index ea07530..cbecf1e 100644 --- a/tests/test_pdf_url_title.py +++ b/tests/test_pdf_url_title.py @@ -10,6 +10,7 @@ from unittest.mock import patch import pytest +import requests from pypdf import PdfWriter from linkml_reference_validator.etl.extract.pdf import PDFExtractor @@ -37,11 +38,20 @@ def config(tmp_path): def _fetch(url, responses, config): - """Fetch ``url`` with the acquirer answering from ``responses`` (url -> (bytes, ctype)).""" + """Fetch ``url`` with the acquirer answering from ``responses``. + + Each value is ``(bytes, content_type)``, or an exception to raise, as + ``requests`` does on a timeout or a refused connection. + """ + + def answer(u, _config): + response = responses.get(u, (None, None)) + if isinstance(response, Exception): + raise response + return response + with patch("linkml_reference_validator.etl.sources.url.ContentAcquirer") as MockAcquirer: - MockAcquirer.return_value.fetch_bytes.side_effect = ( - lambda u, _config: responses.get(u, (None, None)) - ) + MockAcquirer.return_value.fetch_bytes.side_effect = answer result = URLSource().fetch(url, config) requested = [c.args[0] for c in MockAcquirer.return_value.fetch_bytes.call_args_list] return result, requested @@ -155,6 +165,22 @@ def test_landing_page_without_citation_title_falls_through_to_embedded(config): assert result.title == "Embedded Title" +@pytest.mark.parametrize( + "error", [requests.Timeout("slow"), requests.ConnectionError("refused")] +) +def test_landing_page_error_keeps_the_pdf(config, error): + """The title is best-effort. A failed landing page must not cost the PDF.""" + result, requested = _fetch( + JSTAGE_PDF, + {JSTAGE_PDF: (_pdf("Embedded Title"), "application/pdf"), JSTAGE_ARTICLE: error}, + config, + ) + assert requested == [JSTAGE_PDF, JSTAGE_ARTICLE] + assert result is not None + assert result.full_text_url == JSTAGE_PDF + assert result.title == "Embedded Title" + + def test_landing_page_fetch_failure_falls_through_to_url(config): result, _ = _fetch(JSTAGE_PDF, {JSTAGE_PDF: (_pdf(), "application/pdf")}, config) assert result is not None From a3620190706ac9c23ae731d90c8df2b12fd4856c Mon Sep 17 00:00:00 2001 From: caufieldjh Date: Tue, 29 Sep 2026 17:23:39 -0400 Subject: [PATCH 08/14] Serve the stale url: entry when the network raises URLSource.fetch let requests exceptions through, so a timeout or a refused connection failed the fetch before _stale_fallback could serve the cached entry. That gap predates #98, but #98 makes every unstamped url: entry go back to the network once, so the first offline run after upgrading would have hit it on every url: reference. The request error now returns None, as clinicaltrials.py already does. The offline test raised nothing before; it now raises ConnectionError and Timeout as well as covering the (None, None) refusal. Review of #98. Co-Authored-By: Claude Opus 5.5 --- .../etl/sources/url.py | 6 ++++- tests/test_url_source_cache_version.py | 22 ++++++++++++++++--- 2 files changed, 24 insertions(+), 4 deletions(-) diff --git a/src/linkml_reference_validator/etl/sources/url.py b/src/linkml_reference_validator/etl/sources/url.py index 3a89572..c2c0ede 100644 --- a/src/linkml_reference_validator/etl/sources/url.py +++ b/src/linkml_reference_validator/etl/sources/url.py @@ -89,7 +89,11 @@ def fetch( # Stream through ContentAcquirer so the size cap, rate-limit delay, and # User-Agent are applied uniformly. A url: pointing at a large PDF would # otherwise be buffered entirely into memory by a plain requests.get. - data, content_type = ContentAcquirer().fetch_bytes(url, config) + try: # external system boundary: requests raises when offline or on timeout + data, content_type = ContentAcquirer().fetch_bytes(url, config) + except requests.RequestException as e: + logger.warning(f"Failed to fetch {url}: {e}") + return None if data is None: # non-200 or the size cap was exceeded (the acquirer logs the reason) return None diff --git a/tests/test_url_source_cache_version.py b/tests/test_url_source_cache_version.py index 34ebb91..0643326 100644 --- a/tests/test_url_source_cache_version.py +++ b/tests/test_url_source_cache_version.py @@ -12,6 +12,7 @@ from unittest.mock import patch import pytest +import requests from linkml_reference_validator.etl.reference_fetcher import ( ABSENT_CONTENT_CACHE_VERSION, @@ -110,14 +111,29 @@ def test_old_raw_entry_is_refetched_and_sanitized(fetcher): assert "SECRET-KEY-123" not in cache_path.read_text() -def test_old_raw_entry_is_still_served_when_offline(fetcher): - """Out of date is better than not found. It is not re-saved, so it stays stale.""" +@pytest.mark.parametrize( + "unreachable", + [ + (None, None), # the acquirer's own refusal: non-200, or over the size cap + requests.ConnectionError("network is unreachable"), # really offline + requests.Timeout("timed out"), + ], +) +def test_old_raw_entry_is_still_served_when_offline(fetcher, unreachable): + """Out of date is better than not found. It is not re-saved, so it stays stale. + + Every unstamped url: entry goes back to the network after this change, so + the first offline run is where this has to hold. + """ cache_path = fetcher.get_cache_path(f"url:{URL}") cache_path.parent.mkdir(parents=True, exist_ok=True) cache_path.write_text(_entry(f"url:{URL}", "url", body=RAW_PAGE)) with patch("linkml_reference_validator.etl.sources.url.ContentAcquirer") as MockAcquirer: - MockAcquirer.return_value.fetch_bytes.return_value = (None, None) + if isinstance(unreachable, Exception): + MockAcquirer.return_value.fetch_bytes.side_effect = unreachable + else: + MockAcquirer.return_value.fetch_bytes.return_value = unreachable result = fetcher.fetch(f"url:{URL}") assert result is not None From 55b17a77c3e96974dd650087cfcc500d5de1a293 Mon Sep 17 00:00:00 2001 From: caufieldjh Date: Tue, 29 Sep 2026 17:24:03 -0400 Subject: [PATCH 09/14] Match citation_title only on the name and content attributes The meta regexes anchored on \b, which also matches after the hyphen in data-name= and data-content=. A page carrying either could have its title read from the wrong attribute. Attribute names are now matched only after whitespace. Review of #98. Co-Authored-By: Claude Opus 5.5 --- src/linkml_reference_validator/etl/sources/url.py | 6 ++++-- tests/test_pdf_url_title.py | 13 +++++++++++++ 2 files changed, 17 insertions(+), 2 deletions(-) diff --git a/src/linkml_reference_validator/etl/sources/url.py b/src/linkml_reference_validator/etl/sources/url.py index c2c0ede..581a7f0 100644 --- a/src/linkml_reference_validator/etl/sources/url.py +++ b/src/linkml_reference_validator/etl/sources/url.py @@ -34,8 +34,10 @@ ] _META_TAG = re.compile(r"]*>", re.IGNORECASE) -_META_NAME = re.compile(r"""\bname\s*=\s*["']citation_title["']""", re.IGNORECASE) -_META_CONTENT = re.compile(r"""\bcontent\s*=\s*(["'])(.*?)\1""", re.IGNORECASE | re.DOTALL) +# Attribute names follow whitespace. A \b boundary would also match inside +# data-name= and data-content=, since "-" is a word boundary. +_META_NAME = re.compile(r"""(?<=\s)name\s*=\s*["']citation_title["']""", re.IGNORECASE) +_META_CONTENT = re.compile(r"""(?<=\s)content\s*=\s*(["'])(.*?)\1""", re.IGNORECASE | re.DOTALL) @ReferenceSourceRegistry.register diff --git a/tests/test_pdf_url_title.py b/tests/test_pdf_url_title.py index cbecf1e..a24a419 100644 --- a/tests/test_pdf_url_title.py +++ b/tests/test_pdf_url_title.py @@ -198,3 +198,16 @@ def test_html_prefers_citation_title_over_title_tag(config): ) result, _ = _fetch(url, {url: (page, "text/html")}, config) assert result.title == "Actual Article Title" + + +@pytest.mark.parametrize( + "page", + [ + '', + '', + '', + ], +) +def test_citation_title_ignores_look_alike_attributes(page): + """``data-name`` and ``data-content`` are other attributes, not ``name`` and ``content``.""" + assert URLSource._citation_title(page) == "Right" From d88cbdca3b6d4c15b6c09b23757b9cec0ac85271 Mon Sep 17 00:00:00 2001 From: caufieldjh Date: Tue, 29 Sep 2026 17:24:50 -0400 Subject: [PATCH 10/14] Recognize HTML past a BOM or leading comments A page served as text/plain or with no content type was sanitized only if sniff_format saw at its start. A byte order mark, a leading comment (saved pages often carry ""), or a page starting at or sent it to the cache raw, scripts and all. URLSource._looks_like_html skips a UTF-8 BOM and leading comments, then also accepts and . sniff_format is unchanged, since other paths use it for format resolution. Text that only mentions a tag is still left alone. Review of #98. Co-Authored-By: Claude Opus 5.5 --- .../etl/sources/url.py | 30 ++++++++++++++++++- tests/test_url_html_sanitize.py | 28 +++++++++++++++++ 2 files changed, 57 insertions(+), 1 deletion(-) diff --git a/src/linkml_reference_validator/etl/sources/url.py b/src/linkml_reference_validator/etl/sources/url.py index 581a7f0..1d20de5 100644 --- a/src/linkml_reference_validator/etl/sources/url.py +++ b/src/linkml_reference_validator/etl/sources/url.py @@ -118,7 +118,7 @@ def fetch( content = self._decode(data, content_type_header) # Before sanitizing: citation_title lives in a attribute. title = self._extract_title(content, url) - if "html" in content_type_header or sniff_format(data) == "html": + if "html" in content_type_header or self._looks_like_html(data): content = sanitize_html(content) return ReferenceContent( @@ -197,6 +197,34 @@ def _citation_title(content: str) -> Optional[str]: return title return None + @staticmethod + def _looks_like_html(data: bytes) -> bool: + """Report whether a body is an HTML page, whatever its content type said. + + Skips a UTF-8 byte order mark and leading comments (a saved page often + begins with one), then accepts a doctype, ````, ```` or + ````. Text that merely mentions a tag is not a page. + + Examples: + >>> URLSource._looks_like_html(b"\\xef\\xbb\\xbf") + True + >>> URLSource._looks_like_html(b"\\n") + True + >>> URLSource._looks_like_html(b"T") + True + >>> URLSource._looks_like_html(b"Notes on the element") + False + >>> URLSource._looks_like_html(b'') + False + """ + head = data[:8192].removeprefix(b"\xef\xbb\xbf").lstrip() + while head.startswith(b"") + if end == -1: + return False + head = head[end + 3 :].lstrip() + return sniff_format(head) == "html" or head[:5].lower() in (b" str: """Decode HTML/text bytes using the content-type charset, defaulting to UTF-8. diff --git a/tests/test_url_html_sanitize.py b/tests/test_url_html_sanitize.py index cac050c..6288113 100644 --- a/tests/test_url_html_sanitize.py +++ b/tests/test_url_html_sanitize.py @@ -131,3 +131,31 @@ def test_xml_is_left_alone(config): def test_sanitize_html_is_idempotent(): once = sanitize_html(PAGE.decode()) assert sanitize_html(once) == once + + +SCRIPTED = b'

Text

' + + +@pytest.mark.parametrize( + "body", + [ + b"\xef\xbb\xbf" + SCRIPTED + b"", + b"\n" + SCRIPTED, + b"\n\n" + SCRIPTED, + b"T" + SCRIPTED, + b"\n " + SCRIPTED, + ], + ids=["bom", "comment", "two-comments", "head-first", "body-first"], +) +@pytest.mark.parametrize("content_type", [None, "text/plain"]) +def test_html_is_recognized_past_a_bom_or_leading_comments(config, body, content_type): + """A mislabelled page is still a page. Its scripts must not reach the cache.""" + result = _fetch("https://example.org/a", body, content_type, config) + assert "SECRET-KEY-123" not in result.content + assert "Text" in result.content + + +def test_text_that_mentions_html_is_left_alone(config): + body = b"Notes on the element and