{body}
',
+ ".article-text": '{body}
',
+ "#artText": '{body}
',
+ ".article-body": '{body}
',
+ ".article__body": '{body}
',
+ ".c-article-body": '{body}
',
+}
+
+#: Markup no selector may take for an article body.
+ARTICLE_BODY_COUNTEREXAMPLES: tuple[str, ...] = (
+ '{body}
',
+ '{body}
',
+)
+
+# --- PMC article containers ------------------------------------------------
+
+#: Classes of PMC's article-body element. PMC moved its article from
+#: ``div.article-body``/``div.tsec`` to ``section.main-article-body``, and the
+#: fallback declined every article until this list caught up
+#: (monarch-initiative/dismech#12672). Deliberately not ``body``, which PMC
+#: pairs with ``main-article-body`` on the same element. Matching it alone
+#: would match every page's ````, and the structural test is the only
+#: thing standing between this fetch and caching a bot-check interstitial
+#: served on an HTTP 200.
+#:
+#: Order is a priority order, not an alphabetical one: the first match wins,
+#: so the current wrapper is tried before the legacy ones. It matters for a
+#: legacy page carrying several ``tsec`` sections, where only the first is
+#: returned. The previous code behaved the same way, so this is a known limit
+#: rather than a regression, but reordering the tuple would change which
+#: section that is.
+PMC_ARTICLE_BODY_CLASSES: tuple[str, ...] = ("main-article-body", "article-body", "tsec")
+
+#: ``class -> markup it must find``. ``{body}`` stands for real article prose.
+PMC_ARTICLE_BODY_EXAMPLES: dict[str, tuple[str, ...]] = {
+ "main-article-body": (
+ '{body}
',), # legacy
+ "tsec": ('{body}
',), # legacy
+}
+
+#: Pages that carry no article and must be declined.
+PMC_ARTICLE_BODY_COUNTEREXAMPLES: tuple[str, ...] = (
+ "Checking your browser before accessing.
", + "", + "", + # is on every page, so class="body" alone must not count. + "{body}", + # PMC's challenge page carries a classed . None of its classes may + # overlap the list, or a bot-check page is cached as full text. + '' + "Checking your browser before accessing
" + "Enable JavaScript and cookies to continue.
" + "", +) + +# --- PMC placeholder notices ----------------------------------------------- + +#: Phrases appearing in the placeholder documents PMC serves in place of an +#: article whose full text it cannot supply. Deliberately broad, because the +#: length gate in :mod:`linkml_reference_validator.etl.extract.xml` +#: (``MAX_STUB_NOTICE_CHARS``), not the wording, is what keeps matching safe. +#: PMC phrases these notices several ways ("access to this article is +#: restricted", "full text is restricted", ...), and an exhaustive list would +#: trade the old false positives for false negatives. +STUB_NOTICE_PHRASES: tuple[str, ...] = ( + "restricted", + "does not allow downloading", + "cannot be obtained", + "not available from pmc", +) + +#: ``phrase -> notices it must catch``. +STUB_NOTICE_EXAMPLES: dict[str, tuple[str, ...]] = { + "restricted": ( + "Access to the full text is restricted by the publisher.", + # Wordings an exhaustive phrase list would miss; the length gate is + # what lets a broad "restricted" catch them safely. + "Access to this article is restricted.", + "Full text is restricted.", + ), + "does not allow downloading": ( + "The publisher of this article does not allow downloading of the full " + "text in XML form from PMC.", + ), + "cannot be obtained": ("The full text of this article cannot be obtained from PMC.",), + "not available from pmc": ("This article is not available from PMC.",), +} diff --git a/src/linkml_reference_validator/etl/sources/url.py b/src/linkml_reference_validator/etl/sources/url.py index 384e945..358830d 100644 --- a/src/linkml_reference_validator/etl/sources/url.py +++ b/src/linkml_reference_validator/etl/sources/url.py @@ -10,24 +10,37 @@ True """ +import html import logging import re from typing import Optional +import requests + from linkml_reference_validator.models import ReferenceContent, ReferenceValidationConfig from linkml_reference_validator.etl.sources.base import ReferenceSource, ReferenceSourceRegistry -from linkml_reference_validator.etl.acquire import ContentAcquirer +from linkml_reference_validator.etl.acquire import ContentAcquirer, sniff_format +from linkml_reference_validator.etl.extract.html import sanitize_html from linkml_reference_validator.etl.extract.pdf import PDFExtractor +from linkml_reference_validator.etl.rules import LANDING_PAGE_RULES logger = logging.getLogger(__name__) +_META_TAG = re.compile(r"]*>", re.IGNORECASE) +# Attribute names follow whitespace. A \b boundary would also match inside +# data-name= and data-content=, since "-" is a word boundary. +_META_NAME = re.compile(r"""(?<=\s)name\s*=\s*["']citation_title["']""", re.IGNORECASE) +_META_CONTENT = re.compile(r"""(?<=\s)content\s*=\s*(["'])(.*?)\1""", re.IGNORECASE | re.DOTALL) + @ReferenceSourceRegistry.register class URLSource(ReferenceSource): """Fetch reference content from web URLs. - Fetches HTML and plain text content. HTML is returned as-is (no parsing). - Content is cached to disk like other sources. + Fetches HTML and plain text content. HTML keeps its markup but is + sanitized (see :func:`sanitize_html`) so page scripts and attributes do not + reach the cache. Plain text and XML are stored as fetched. Content is cached + to disk like other sources. Examples: >>> source = URLSource() @@ -71,7 +84,11 @@ def fetch( # Stream through ContentAcquirer so the size cap, rate-limit delay, and # User-Agent are applied uniformly. A url: pointing at a large PDF would # otherwise be buffered entirely into memory by a plain requests.get. - data, content_type = ContentAcquirer().fetch_bytes(url, config) + try: # external system boundary: requests raises when offline or on timeout + data, content_type = ContentAcquirer().fetch_bytes(url, config) + except requests.RequestException as e: + logger.warning(f"Failed to fetch {url}: {e}") + return None if data is None: # non-200 or the size cap was exceeded (the acquirer logs the reason) return None @@ -85,14 +102,17 @@ def fetch( ) return ReferenceContent( reference_id=f"url:{url}", - title=url, + title=self._recover_pdf_title(url, data, config), content=text, content_type="full_text_pdf" if text else "unavailable", full_text_url=url, ) content = self._decode(data, content_type_header) + # Before sanitizing: citation_title lives in a attribute. title = self._extract_title(content, url) + if "html" in content_type_header or self._looks_like_html(data): + content = sanitize_html(content) return ReferenceContent( reference_id=f"url:{url}", @@ -101,6 +121,103 @@ def fetch( content_type="url", ) + def _recover_pdf_title( + self, url: str, data: bytes, config: ReferenceValidationConfig + ) -> str: + """Find a real title for a PDF, falling back to its URL. + + Tries, in order: the ``citation_title`` of a landing page found by + ``LANDING_PAGE_RULES``, then the PDF's embedded ``/Title``. When both + fail the URL is returned, so ``title == url`` still means none was found. + """ + landing = self._landing_page_url(url) + if landing is not None: + title = self._landing_page_title(landing, config) + if title: + return title + logger.debug(f"No citation_title at landing page {landing} for {url}") + + embedded = PDFExtractor(backend=config.pdf_backend).extract_title(data) + return embedded or url + + def _landing_page_title( + self, landing: str, config: ReferenceValidationConfig + ) -> Optional[str]: + """Return the ``citation_title`` of a landing page, or None if it cannot be had.""" + try: # external system boundary: the title is best-effort, the PDF is not + page, content_type = ContentAcquirer().fetch_bytes(landing, config) + except requests.RequestException as e: + logger.debug(f"Landing page {landing} could not be fetched: {e}") + return None + if page is None: + return None + return self._citation_title(self._decode(page, (content_type or "").lower())) + + @staticmethod + def _landing_page_url(url: str) -> Optional[str]: + """Return the landing page for a PDF URL, if a rule in ``LANDING_PAGE_RULES`` matches. + + Examples: + >>> URLSource._landing_page_url( + ... "https://www.jstage.jst.go.jp/article/jvms/64/1/64_1_1/_pdf/-char/ja") + 'https://www.jstage.jst.go.jp/article/jvms/64/1/64_1_1/_article/-char/ja' + >>> URLSource._landing_page_url("https://example.org/paper.pdf") is None + True + """ + for pattern, replacement in LANDING_PAGE_RULES.values(): + if re.match(pattern, url): + return re.sub(pattern, replacement, url) + return None + + @staticmethod + def _citation_title(content: str) -> Optional[str]: + """Return the ``citation_title`` meta tag's content, if present. + + Examples: + >>> URLSource._citation_title('') + 'A & B' + >>> URLSource._citation_title("") + 'Reversed' + >>> URLSource._citation_title("Text cannot be obtained from PMC.
| Data |
" + ("Real article prose. " * 40) + "
" - - -@pytest.mark.parametrize( - "markup,label", - [ - (f'{BODY}
', "legacy div.article-body"),
- (f'{BODY}
', "legacy div.tsec"),
- (f'Checking your browser before accessing.
", - "bot-check interstitial"), - ("", "navigation page"), - ("", - "banner chrome only"), - ], -) -def test_pages_without_an_article_container_are_declined(markup, label): - soup = BeautifulSoup(markup, "html.parser") - assert find_pmc_article_body(soup) is None, f"{label} must not be accepted" - - -def test_a_bare_body_element_is_not_an_article_container(): - """```` is on every page, so matching class="body" alone would accept - an interstitial. Only the article-body classes count.""" - soup = BeautifulSoup(f"{BODY}", "html.parser") - assert find_pmc_article_body(soup) is None - - -def test_a_challenge_page_carrying_a_classed_body_is_still_declined(): - """The interstitial arrives on an HTTP 200, so this selector is the only check. - - PMC's challenge page carries a classed ````. None of its classes may - overlap the article-body list, or a bot-check page gets cached as full text. - """ - markup = ( - '' - "Checking your browser before accessing
" - "Enable JavaScript and cookies to continue.
" - "" - ) - soup = BeautifulSoup(markup, "html.parser") - assert find_pmc_article_body(soup) is None diff --git a/tests/test_reference_fetcher.py b/tests/test_reference_fetcher.py index aa76162..bb47195 100644 --- a/tests/test_reference_fetcher.py +++ b/tests/test_reference_fetcher.py @@ -1384,9 +1384,17 @@ def test_entry_whose_id_contains_a_horizontal_rule_is_not_perpetually_stale(fetc re-fetched, be rewritten with the same id, and read as unstamped again on every single run. """ + from linkml_reference_validator.etl.reference_fetcher import URL_SOURCE_CACHE_VERSION + reference_id = "url:https://example.com/a---b" + # Stamped as a fresh URLSource fetch would be. The stamp is written below + # reference_id, so a truncated block would hide it too. fetcher._save_to_disk( - ReferenceContent(reference_id=reference_id, content="Body text.") + ReferenceContent( + reference_id=reference_id, + content="Body text.", + metadata={"url_source_version": URL_SOURCE_CACHE_VERSION}, + ) ) assert not fetcher._is_stale_cache_entry(_cached_text(fetcher, reference_id)) diff --git a/tests/test_rules.py b/tests/test_rules.py new file mode 100644 index 0000000..e77bfea --- /dev/null +++ b/tests/test_rules.py @@ -0,0 +1,129 @@ +"""Every publisher rule in ``etl.rules`` has examples, and the code honours them. + +The examples live in ``etl/rules.py`` beside their rules. This file only runs +them, through the code that consumes each rule rather than the rule alone, so +an example that passes here passes where it matters. +""" + +import pytest +from bs4 import BeautifulSoup + +from linkml_reference_validator.etl import rules +from linkml_reference_validator.etl.extract import html, pdf, xml +from linkml_reference_validator.etl.extract.pdf import clean_pdf_title +from linkml_reference_validator.etl.extract.xml import is_stub_notice +from linkml_reference_validator.etl.fulltext import pmc +from linkml_reference_validator.etl.fulltext.pmc import find_pmc_article_body +from linkml_reference_validator.etl.sources import url +from linkml_reference_validator.etl.sources.url import URLSource + +#: Stands in for ``{body}`` in markup examples: long enough to be an article. +BODY = "" + ("Real article prose. " * 40) + "
" + + +# --- every rule has an example --------------------------------------------- + + +@pytest.mark.parametrize( + "rule_names,examples", + [ + (set(rules.LANDING_PAGE_RULES), rules.LANDING_PAGE_EXAMPLES), + (set(rules.ARTICLE_BODY_SELECTORS), rules.ARTICLE_BODY_EXAMPLES), + (set(rules.PMC_ARTICLE_BODY_CLASSES), rules.PMC_ARTICLE_BODY_EXAMPLES), + (set(rules.STUB_NOTICE_PHRASES), rules.STUB_NOTICE_EXAMPLES), + ], + ids=["landing-page", "article-body", "pmc-article-body", "stub-notice"], +) +def test_every_rule_has_an_example(rule_names, examples): + assert set(examples) == rule_names + assert all(examples.values()) + + +def test_placeholder_titles_have_examples(): + assert rules.PDF_PLACEHOLDER_TITLE_EXAMPLES + assert rules.PDF_PLACEHOLDER_TITLE_COUNTEREXAMPLES + + +# --- consumers use the rules, not copies ------------------------------------ + + +def test_consumers_import_the_rules(): + assert url.LANDING_PAGE_RULES is rules.LANDING_PAGE_RULES + assert pdf.PDF_PLACEHOLDER_TITLE is rules.PDF_PLACEHOLDER_TITLE + assert pmc.PMC_ARTICLE_BODY_CLASSES is rules.PMC_ARTICLE_BODY_CLASSES + assert xml.STUB_NOTICE_PHRASES is rules.STUB_NOTICE_PHRASES + assert html.ARTICLE_BODY_SELECTOR == ", ".join(rules.ARTICLE_BODY_SELECTORS) + + +# --- landing pages ----------------------------------------------------------- + + +@pytest.mark.parametrize( + "pdf_url,landing", + [pair for pairs in rules.LANDING_PAGE_EXAMPLES.values() for pair in pairs], +) +def test_landing_page_example(pdf_url, landing): + assert URLSource._landing_page_url(pdf_url) == landing + + +@pytest.mark.parametrize("pdf_url", rules.LANDING_PAGE_COUNTEREXAMPLES) +def test_landing_page_counterexample(pdf_url): + assert URLSource._landing_page_url(pdf_url) is None + + +# --- placeholder PDF titles ------------------------------------------------- + + +@pytest.mark.parametrize("title", rules.PDF_PLACEHOLDER_TITLE_EXAMPLES) +def test_placeholder_title_example(title): + assert clean_pdf_title(title) is None + + +@pytest.mark.parametrize("title", rules.PDF_PLACEHOLDER_TITLE_COUNTEREXAMPLES) +def test_placeholder_title_counterexample(title): + assert clean_pdf_title(title) == title + + +# --- article body containers ------------------------------------------------ + + +@pytest.mark.parametrize("selector,markup", sorted(rules.ARTICLE_BODY_EXAMPLES.items())) +def test_article_body_example(selector, markup): + soup = BeautifulSoup(markup.format(body=BODY), "html.parser") + assert soup.select_one(html.ARTICLE_BODY_SELECTOR) is not None, selector + + +@pytest.mark.parametrize("markup", rules.ARTICLE_BODY_COUNTEREXAMPLES) +def test_article_body_counterexample(markup): + soup = BeautifulSoup(markup.format(body=BODY), "html.parser") + assert soup.select_one(html.ARTICLE_BODY_SELECTOR) is None + + +# --- PMC article containers -------------------------------------------------- + + +@pytest.mark.parametrize( + "markup", + [m for examples in rules.PMC_ARTICLE_BODY_EXAMPLES.values() for m in examples], +) +def test_pmc_article_body_example(markup): + soup = BeautifulSoup(markup.format(body=BODY), "html.parser") + assert find_pmc_article_body(soup) is not None + + +@pytest.mark.parametrize("markup", rules.PMC_ARTICLE_BODY_COUNTEREXAMPLES) +def test_pmc_article_body_counterexample(markup): + soup = BeautifulSoup(markup.format(body=BODY), "html.parser") + assert find_pmc_article_body(soup) is None + + +# --- PMC placeholder notices -------------------------------------------------- + + +@pytest.mark.parametrize( + "phrase,notice", + [(p, n) for p, notices in rules.STUB_NOTICE_EXAMPLES.items() for n in notices], +) +def test_stub_notice_example(phrase, notice): + assert phrase in notice.lower(), "an example must show the phrase it is filed under" + assert is_stub_notice(notice) diff --git a/tests/test_url_html_sanitize.py b/tests/test_url_html_sanitize.py new file mode 100644 index 0000000..6288113 --- /dev/null +++ b/tests/test_url_html_sanitize.py @@ -0,0 +1,161 @@ +"""HTML fetched by ``URLSource`` is sanitized before it is cached (issue #92). + +``URLSource`` used to store the response body verbatim, so page scripts, +comments and tag attributes (signed asset URLs, API-key parameters) went +into caches that projects commit to public repositories. Now HTML keeps its +body markup and text, loses ``script``/``style``/``noscript``/``template`` +and comments, drops ``meta``/``link``/``base`` (they hold nothing but +attributes), and keeps only the table-structural attributes ``rowspan``, +``colspan`` and ``scope``. Plain text and XML are stored as before. +""" + +from unittest.mock import patch + +import pytest + +from linkml_reference_validator.etl.extract.html import sanitize_html +from linkml_reference_validator.etl.sources.url import URLSource +from linkml_reference_validator.models import ReferenceValidationConfig + +PAGE = b""" + + +Patients showed lactic acidosis.
+ +Template body
+| Gene | Finding | |
|---|---|---|
| POLG | A | B |
Text
' + + +@pytest.mark.parametrize( + "body", + [ + b"\xef\xbb\xbf" + SCRIPTED + b"", + b"\n" + SCRIPTED, + b"\n\n" + SCRIPTED, + b"Patients showed lactic acidosis.
" +) + + +def _entry( + reference_id: str, content_type: str, *, url_version: int | None = None, body: str = "" +) -> str: + """A cache entry current in every respect except the stamp a test varies.""" + lines = [ + "---", + f"reference_id: {reference_id}", + f"extractor_version: {EXTRACTOR_CACHE_VERSION}", + ] + if url_version is not None: + lines.append(f"url_source_version: {url_version}") + if content_type == "unavailable": + lines.append(f"absent_content_version: {ABSENT_CONTENT_CACHE_VERSION}") + lines += [f"content_type: {content_type}", "---", "", "## Content", "", body] + return "\n".join(lines) + + +@pytest.mark.parametrize("content_type", ["url", "full_text_pdf", "unavailable"]) +def test_unstamped_url_entry_is_stale(content_type): + assert ReferenceFetcher._is_stale_cache_entry(_entry(f"url:{URL}", content_type)) + + +@pytest.mark.parametrize("content_type", ["url", "full_text_pdf", "unavailable"]) +def test_current_url_entry_is_fresh(content_type): + entry = _entry(f"url:{URL}", content_type, url_version=URL_SOURCE_CACHE_VERSION) + assert not ReferenceFetcher._is_stale_cache_entry(entry) + + +def test_a_newer_stamp_is_not_stale(): + entry = _entry(f"url:{URL}", "url", url_version=URL_SOURCE_CACHE_VERSION + 1) + assert not ReferenceFetcher._is_stale_cache_entry(entry) + + +def test_other_sources_are_untouched(): + assert not ReferenceFetcher._is_stale_cache_entry(_entry("PMID:12345", "abstract_only")) + + +@pytest.fixture +def fetcher(tmp_path): + return ReferenceFetcher(ReferenceValidationConfig(cache_dir=tmp_path, rate_limit_delay=0.0)) + + +def test_a_fresh_fetch_is_stamped_and_round_trips(fetcher, tmp_path): + with patch("linkml_reference_validator.etl.sources.url.ContentAcquirer") as MockAcquirer: + MockAcquirer.return_value.fetch_bytes.return_value = (RAW_PAGE.encode(), "text/html") + fetcher.fetch(f"url:{URL}") + + written = next(tmp_path.glob("*.md")).read_text() + assert f"url_source_version: {URL_SOURCE_CACHE_VERSION}" in written + assert not ReferenceFetcher._is_stale_cache_entry(written) + + reloaded = fetcher._load_from_disk(f"url:{URL}") + assert reloaded is not None + assert reloaded.metadata["url_source_version"] == URL_SOURCE_CACHE_VERSION + + +def test_only_a_fresh_fetch_is_certified(fetcher, tmp_path): + """Saving a url: entry that URLSource did not just produce must not stamp it.""" + fetcher._save_to_disk( + ReferenceContent( + reference_id=f"url:{URL}", content_type="url", title="t", content=RAW_PAGE + ) + ) + written = next(tmp_path.glob("*.md")).read_text() + assert "url_source_version" not in written + + +def test_old_raw_entry_is_refetched_and_sanitized(fetcher): + cache_path = fetcher.get_cache_path(f"url:{URL}") + cache_path.parent.mkdir(parents=True, exist_ok=True) + cache_path.write_text(_entry(f"url:{URL}", "url", body=RAW_PAGE)) + + with patch("linkml_reference_validator.etl.sources.url.ContentAcquirer") as MockAcquirer: + MockAcquirer.return_value.fetch_bytes.return_value = (RAW_PAGE.encode(), "text/html") + result = fetcher.fetch(f"url:{URL}") + + assert result is not None + assert "SECRET-KEY-123" not in result.content + assert "SECRET-KEY-123" not in cache_path.read_text() + + +@pytest.mark.parametrize( + "unreachable", + [ + (None, None), # the acquirer's own refusal: non-200, or over the size cap + requests.ConnectionError("network is unreachable"), # really offline + requests.Timeout("timed out"), + ], +) +def test_old_raw_entry_is_still_served_when_offline(fetcher, unreachable): + """Out of date is better than not found. It is not re-saved, so it stays stale. + + Every unstamped url: entry goes back to the network after this change, so + the first offline run is where this has to hold. + """ + cache_path = fetcher.get_cache_path(f"url:{URL}") + cache_path.parent.mkdir(parents=True, exist_ok=True) + cache_path.write_text(_entry(f"url:{URL}", "url", body=RAW_PAGE)) + + with patch("linkml_reference_validator.etl.sources.url.ContentAcquirer") as MockAcquirer: + if isinstance(unreachable, Exception): + MockAcquirer.return_value.fetch_bytes.side_effect = unreachable + else: + MockAcquirer.return_value.fetch_bytes.return_value = unreachable + result = fetcher.fetch(f"url:{URL}") + + assert result is not None + assert "lactic acidosis" in result.content + assert ReferenceFetcher._is_stale_cache_entry(cache_path.read_text())