diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md
index 15d2bec..8fbf1e3 100644
--- a/docs/troubleshooting.md
+++ b/docs/troubleshooting.md
@@ -785,3 +785,51 @@ Run through this checklist when encountering issues:
- [CLI Reference](reference/cli.md) - Complete command documentation
- [How to Repair Validation Errors](how-to/repair-validation-errors.md) - Fixing common issues
- [GitHub Issues](https://github.com/linkml/linkml-reference-validator/issues) - Report bugs
+
+### JATS tables and XML cache refresh
+
+JATS/PMC XML extraction appends pipe-delimited tables after the existing body
+paragraphs. Tables are found throughout the document, including `floats-group`
+when there is no body. Labels and captions form headings; table paragraphs do
+not also appear as body prose. Other existing body paragraphs, such as table
+attribution or alternative descriptions, are retained. Abstract extraction remains the source's job.
+Restricted body notices are checked before any tables are appended.
+
+Each actual table is rendered once, including tables inside nested wrappers.
+Nested tables without their own wrapper use a generic `Nested table` heading.
+Inline text and symbols are retained, block/line breaks become spaces, and
+literal backslashes and pipes in cells are escaped. Rows preserve source cell
+order, including empty cells. A span is printed as `[rowspan=2]` or
+`[colspan=2]` on its source cell: values are not copied into other rows or
+columns. These are quotable source rows, not a reconstructed rectangular grid;
+interpret spanned rows using the original table. Images and non-HTML table
+encodings are not transcribed, but their labels/captions are retained.
+As in existing body extraction, superscript/subscript text is flattened:
+`109` becomes `109`, not exponent notation. Consult the original
+for numeric interpretation; superscript styling is not preserved.
+
+The first **200 source rows per table**, including header and empty rows, are
+kept. Larger tables end with `[Table truncated after 200 rows.]`; later rows
+cannot be validated from this cache. Table footnotes in `table-wrap-foot` are retained once in document order,
+including notes in `floats-group`. The cap is per table, not per document.
+
+`full_text_xml` entries now carry `xml_extraction_version: 1`, independently of
+`extractor_version` and `html_full_text_version`. Missing/older XML stamps cause
+refresh on the next validation fetch; current PDF and HTML entries need no
+refresh for this change. Fresh source/provider XML is stamped after acquisition.
+Inventory and metadata-only rewrites preserve the original XML stamp, including
+future versions, and never certify old text. If refresh is unavailable, legacy
+XML remains available with the existing stale-cache warning and is not rewritten;
+it may still lack table rows. A later process retries the refresh. Existing
+stale HTML rejection remains unchanged.
+
+A successful source refresh that returns only an abstract can replace the old
+full text if no provider supplies a body. This is existing refresh behavior;
+the stale fallback applies when the source returns no record, not when it
+returns an abstract-only record. Keep a backup if retaining older full text is
+necessary.
+
+This pass targets JATS `table-wrap` content and searches the whole document;
+tables and notes inside embedded `sub-article` or `response` elements are
+excluded so reviewer/reply findings are not attributed to the main paper. Bare tables
+without a `table-wrap` remain outside this JATS extraction pass.
diff --git a/src/linkml_reference_validator/etl/extract/__init__.py b/src/linkml_reference_validator/etl/extract/__init__.py
index 41bb10d..290deab 100644
--- a/src/linkml_reference_validator/etl/extract/__init__.py
+++ b/src/linkml_reference_validator/etl/extract/__init__.py
@@ -5,7 +5,10 @@
When a change means previously cached text is *wrong* rather than merely older,
bump ``EXTRACTOR_CACHE_VERSION`` in
:mod:`linkml_reference_validator.etl.reference_fetcher` so existing entries are
-re-fetched instead of keeping the old output forever.
+re-fetched instead of keeping the old output forever. For changes confined to
+XML table extraction, bump ``XML_EXTRACTION_CACHE_VERSION`` instead; HTML
+acceptance has its own ``HTML_FULL_TEXT_CACHE_VERSION``. These format-specific
+stamps avoid refreshing unchanged formats.
"""
from linkml_reference_validator.etl.extract.base import Extractor, ExtractorRegistry
diff --git a/src/linkml_reference_validator/etl/extract/xml.py b/src/linkml_reference_validator/etl/extract/xml.py
index 42c2719..463fa25 100644
--- a/src/linkml_reference_validator/etl/extract/xml.py
+++ b/src/linkml_reference_validator/etl/extract/xml.py
@@ -2,13 +2,14 @@
Changing what this yields for the same input may make already-cached text wrong
rather than merely older; see :mod:`linkml_reference_validator.etl.extract` for
-when to bump ``EXTRACTOR_CACHE_VERSION``.
+when to bump cache versions. Table changes use ``XML_EXTRACTION_CACHE_VERSION``
+so PDF and HTML entries remain current.
"""
import logging
from typing import Optional, Union
-from bs4 import BeautifulSoup # type: ignore
+from bs4 import BeautifulSoup, CData, NavigableString, Tag # type: ignore
from linkml_reference_validator.etl.extract.base import Extractor, ExtractorRegistry
@@ -83,18 +84,117 @@ def is_stub_notice(text: str) -> bool:
return any(phrase in lowered for phrase in STUB_NOTICE_PHRASES)
+#: Bound cached table evidence; headers count and truncation is always explicit.
+MAX_TABLE_ROWS = 200
+
+
+def _table_text(node: Tag) -> str:
+ """Keep inline text contiguous, separating blocks and excluding nested tables."""
+
+ def walk(tag: Tag) -> str:
+ """Render descendants without mutating the parsed document."""
+ parts = []
+ for child in tag.children:
+ if type(child) in (NavigableString, CData):
+ parts.append(str(child))
+ elif isinstance(child, Tag) and child.name not in {"table", "table-wrap"}:
+ value = walk(child)
+ if child.name in {"p", "break", "br", "list-item", "title"}:
+ value = " " + value + " "
+ parts.append(value)
+ return "".join(parts)
+
+ return " ".join(walk(node).split())
+
+
+def _table_heading(wrap: Tag) -> str:
+ """Read only the label/caption owned by this wrapper, not nested wrappers."""
+ parts = []
+ for name in ("label", "caption"):
+ node = next(
+ (n for n in wrap.find_all(name) if n.find_parent("table-wrap") is wrap),
+ None,
+ )
+ if node is not None:
+ parts.append(_table_text(node))
+ return " ".join(part for part in parts if part)
+
+
+def _tables_as_text(soup: BeautifulSoup) -> list[str]:
+ """Render each table once in document order, using source cells, not a grid.
+
+ Row/column spans are explicit annotations; values are never replicated or
+ assigned to inferred columns. Header rows count toward the 200-row limit.
+ """
+ sections = []
+ for table in soup.find_all(["table-wrap", "table", "table-wrap-foot"]):
+ if table.find_parent(["sub-article", "response"]):
+ continue
+ if table.name == "table-wrap":
+ heading = _table_heading(table)
+ if heading:
+ sections.append("## " + heading)
+ continue
+ wrap = table.find_parent("table-wrap")
+ if wrap is None:
+ # This pass targets JATS tables, not arbitrary XML layout tables.
+ continue
+ if table.name == "table-wrap-foot":
+ note = _table_text(table)
+ if note:
+ sections.append(note)
+ continue
+ # Named wrapper headings were emitted at their own document position.
+ heading = "" if _table_heading(wrap) else "Table"
+ if table.find_parent(["table", "table-wrap"]) is not wrap:
+ heading = "Nested table"
+ rows = [
+ row for row in table.find_all("tr") if row.find_parent("table") is table
+ ]
+ rendered = []
+ for row in rows[:MAX_TABLE_ROWS]:
+ cells = []
+ for cell in row.find_all(["th", "td"]):
+ if (
+ cell.find_parent("tr") is not row
+ or cell.find_parent("table") is not table
+ ):
+ continue
+ value = _table_text(cell).replace("\\", "\\\\").replace("|", "\\|")
+ for span in ("rowspan", "colspan"):
+ if cell.has_attr(span) and str(cell[span]) != "1":
+ value += f" [{span}={cell[span]}]"
+ cells.append(value)
+ if cells:
+ rendered.append("| " + " | ".join(cells) + " |")
+ if len(rows) > MAX_TABLE_ROWS:
+ rendered.append(f"[Table truncated after {MAX_TABLE_ROWS} rows.]")
+ if rendered:
+ sections.append(
+ ("## " + heading + "\n\n" if heading else "") + "\n".join(rendered)
+ )
+ return sections
+
+
@ExtractorRegistry.register
class XMLExtractor(Extractor):
"""Extract body text from JATS/PMC article XML.
- Returns the concatenated text of paragraphs within the article ``
``.
- Returns None when there is no body content, and when the body holds one of
+ Returns main-article body paragraphs followed by labeled tables and notes,
+ including floats-group content. Sub-article and response content is excluded.
+ Table cells retain inline text and explicitly annotate spans; the first 200
+ source rows per table are retained, with a notice when rows are omitted.
+ Returns None when neither body paragraphs nor table content exists, and when
+ the body holds one of
PMC's placeholder notices instead of the article itself.
Examples:
>>> xml = b"
"
>>> XMLExtractor().extract(stub) is None
True
@@ -112,17 +212,28 @@ def extract(
# re-encoding str here would leave that declaration contradicting the
# bytes. The parser gets both cases right on its own.
soup = BeautifulSoup(data, "xml")
- body = soup.find("body")
- if not body:
- return None
-
- paragraphs = body.find_all("p")
- if not paragraphs:
- return None
-
- text = "\n\n".join(p.get_text() for p in paragraphs if p.get_text().strip())
- if not text.strip():
- return None
+ body = next(
+ (
+ node
+ for node in soup.find_all("body")
+ if not node.find_parent(["sub-article", "response"])
+ ),
+ None,
+ )
+ paragraphs = (
+ [
+ p
+ for p in body.find_all("p")
+ if not p.find_parent(["sub-article", "response"])
+ ]
+ if body
+ else []
+ )
+ text = "\n\n".join(
+ p.get_text()
+ for p in paragraphs
+ if not p.find_parent("table-wrap") and p.get_text().strip()
+ )
# Judged on the extracted body, not the raw markup: a citation title or
# a methods sentence elsewhere in the document says nothing about
@@ -137,4 +248,17 @@ def extract(
)
return None
- return text
+ # Keep existing attribution/alternative prose that the renderer does not
+ # emit. Table content must not inflate the stub check above.
+ text = "\n\n".join(
+ p.get_text()
+ for p in paragraphs
+ if p.get_text().strip()
+ and not (
+ p.find_parent("table-wrap")
+ and p.find_parent(["label", "caption", "table", "table-wrap-foot"])
+ )
+ )
+ return (
+ "\n\n".join(part for part in [text, *_tables_as_text(soup)] if part) or None
+ )
diff --git a/src/linkml_reference_validator/etl/reference_fetcher.py b/src/linkml_reference_validator/etl/reference_fetcher.py
index a68a66e..68ec64a 100644
--- a/src/linkml_reference_validator/etl/reference_fetcher.py
+++ b/src/linkml_reference_validator/etl/reference_fetcher.py
@@ -64,6 +64,9 @@
#: PMC or a configured text provider is trusted under FullTextLocation's contract.
HTML_FULL_TEXT_CACHE_VERSION = 1
+#: XML table extraction changes only XML caches, independent of HTML acceptance.
+XML_EXTRACTION_CACHE_VERSION = 1
+
#: A cache file's frontmatter delimiter: a line that is exactly ``---``.
#: Splitting on the bare string instead lets any *value* containing ``---`` - a
#: URL reference_id, a title - truncate the block, which loses every field after
@@ -184,6 +187,11 @@ def fetch(
content.metadata or {}, html_full_text_version=HTML_FULL_TEXT_CACHE_VERSION
)
+ if content and content.content_type == "full_text_xml":
+ content.metadata = dict(
+ content.metadata or {}, xml_extraction_version=XML_EXTRACTION_CACHE_VERSION
+ )
+
if content and self.config.fetch_full_text and self.needs_full_text(content):
content = self._enrich_with_full_text(content)
@@ -403,6 +411,10 @@ def _apply_full_text_location(
content.metadata = dict(
content.metadata or {}, html_full_text_version=HTML_FULL_TEXT_CACHE_VERSION
)
+ if fmt == "xml":
+ content.metadata = dict(
+ content.metadata or {}, xml_extraction_version=XML_EXTRACTION_CACHE_VERSION
+ )
content.full_text_provider = location.provider or provider_name
# Non-public endpoints are not durable provenance and may contain
# session-specific access information.
@@ -711,6 +723,10 @@ def _save_to_disk(
html_version = (reference.metadata or {}).get("html_full_text_version")
if reference.content_type == "full_text_html" and isinstance(html_version, int):
lines.append(f"html_full_text_version: {html_version}")
+ # YAML booleans are not extraction versions, despite bool subclassing int.
+ xml_version = (reference.metadata or {}).get("xml_extraction_version")
+ if reference.content_type == "full_text_xml" and type(xml_version) is int:
+ lines.append(f"xml_extraction_version: {xml_version}")
if reference.title:
lines.append(f"title: {self._quote_yaml_value(reference.title)}")
if reference.authors:
@@ -976,7 +992,7 @@ def _split_frontmatter(content_text: str) -> Optional[tuple[str, str]]:
@classmethod
def _is_stale_cache_entry(cls, content_text: str) -> bool:
- """Report whether extraction or HTML full-text acceptance needs refreshing.
+ """Report whether extraction or format-specific full-text processing needs refreshing.
Deliberately not applied by :meth:`iter_cached_references`: export and
enrichment walk the cache as a record of what was fetched, and dropping
@@ -1017,6 +1033,11 @@ def _is_stale_cache_entry(cls, content_text: str) -> bool:
):
return True
+ if isinstance(metadata, dict) and metadata.get("content_type") == "full_text_xml":
+ xml_version = metadata.get("xml_extraction_version")
+ if type(xml_version) is not int or xml_version < XML_EXTRACTION_CACHE_VERSION:
+ return True
+
# A newer stamp is not stale: an older tool reading a cache written by a
# newer one should leave it alone rather than re-fetch it on every run.
return False
@@ -1056,6 +1077,8 @@ def _load_markdown_format(
)
metadata: dict = {}
+ if "xml_extraction_version" in frontmatter:
+ metadata["xml_extraction_version"] = frontmatter["xml_extraction_version"]
if "html_full_text_version" in frontmatter:
metadata["html_full_text_version"] = frontmatter["html_full_text_version"]
if "extra_fields_captured" in frontmatter:
diff --git a/tests/fixtures/jats/PMC5593426.xml b/tests/fixtures/jats/PMC5593426.xml
new file mode 100644
index 0000000..f30cccb
--- /dev/null
+++ b/tests/fixtures/jats/PMC5593426.xml
@@ -0,0 +1,5 @@
+
+PMC5593426PMC5593426.155934265593426NIHMS8714592853071310.1038/ni.3753NIHMS871459NIHPA871459BACH2 immunodeficiency illustrates an association between super-enhancers and haploinsufficiencyUsers may view, print, copy, and download text and data-mine the content in such documents, for the purposes of academic research, subject always to the full Conditions of use:
+http://www.nature.com/authors/editorial_policies/license.html#terms
Summary clinical characteristics of patients with missense mutations in BACH2.
positive by immunofluorescence but negative for myeloperoxidase and proteinase III antibodies by ELISA.
diff --git a/tests/fixtures/jats/README.md b/tests/fixtures/jats/README.md
new file mode 100644
index 0000000..e4354f8
--- /dev/null
+++ b/tests/fixtures/jats/README.md
@@ -0,0 +1,21 @@
+# JATS regression fixture
+
+`PMC5593426.xml` retains Table 1, article identifiers/title and the original
+permissions from the Europe PMC response for PMID:28530713 / PMC5593426,
+downloaded 2026-09-15 from:
+https://www.ebi.ac.uk/europepmc/webservices/rest/PMC5593426/fullTextXML
+
+Full downloaded response SHA-256:
+`f9512138d416f28467b81373fd1157766c868ca90bbf6129933ea3b968912ae7`
+
+The article is *BACH2 immunodeficiency illustrates an association between
+super-enhancers and haploinsufficiency*, DOI:10.1038/ni.3753.
+Table 1 contains the immunoglobulin findings reported in issue #68. Its source
+`table-wrap` (T1), including caption, rows and notes, is preserved inside a
+minimal article with the original `floats-group` placement. BeautifulSoup's XML
+serializer normalized markup; table text and attributes were not edited.
+Article prose, other figures, references and supplementary material are omitted.
+
+The original permissions remain in the excerpt. Only the relevant table and
+provenance metadata are retained for regression testing; the complete manuscript
+is not included and the excerpt is not relicensed.
diff --git a/tests/test_jats_tables.py b/tests/test_jats_tables.py
new file mode 100644
index 0000000..d2c6a9a
--- /dev/null
+++ b/tests/test_jats_tables.py
@@ -0,0 +1,344 @@
+"""JATS table extraction and targeted cache migration regressions for #68."""
+
+from pathlib import Path
+
+import pytest
+
+from linkml_reference_validator.etl.extract.xml import XMLExtractor
+from linkml_reference_validator.etl.reference_fetcher import ReferenceFetcher
+from linkml_reference_validator.models import (
+ FullTextLocation,
+ ReferenceContent,
+ ReferenceValidationConfig,
+)
+from linkml_reference_validator.validation.supporting_text_validator import (
+ SupportingTextValidator,
+)
+
+FIXTURE = Path(__file__).parent / "fixtures/jats/PMC5593426.xml"
+
+
+def test_real_clinical_rows():
+ """The reported table-only findings are directly quotable from the real paper."""
+ text = XMLExtractor().extract(FIXTURE.read_bytes())
+ assert text is not None
+ assert (
+ "## Table 1 Summary clinical characteristics of patients with missense mutations in BACH2."
+ in text
+ )
+ assert "IvIg, intravenous immunoglobulin" in text
+ for row in [
+ "On IvIg treatment | Yes | Yes | No",
+ "IgM | Low | Low | High",
+ "IgG | Low | Low | High*",
+ ]:
+ assert f"| {row} |" in text
+ ref = ReferenceContent(reference_id="PMID:28530713", content=text)
+ assert (
+ SupportingTextValidator(ReferenceValidationConfig())
+ .find_text_in_reference(row, ref)
+ .found
+ )
+
+
+@pytest.mark.parametrize("placement", ["body", "floats-group"])
+def test_structural_tables(placement):
+ """Wrappers, blocks and inline tags retain boundaries without duplicate captions."""
+ xml = f"""
Abstract stays separate.
<{placement}>
+
Clinical features.
+
Patient
A
B
+
CD4+ cells
IgG
+
Yesagain
A | B
+ {placement}>"""
+ text = XMLExtractor().extract(xml)
+ assert (
+ text
+ == "## Table 2 Clinical features.\n\n| Patient | A | B |\n| CD4+ cells IgG | Yes again | A \\| B |"
+ )
+
+
+def test_nested_tables_and_wraps_once():
+ """Each actual table owns its rows; descendants never leak into ancestor cells."""
+ xml = """
Body.
+
Outer cell
Inner cell
+
"""
+ text = XMLExtractor().extract(xml)
+ assert text == "Body.\n\n## Outer\n\n| Outer cell |\n\n## Inner\n\n| Inner cell |"
+
+
+def test_spans_are_explicit_source_cells():
+ """Spanned cells are annotated, never propagated into invented patient values."""
+ xml = """
Group
Shared
+
A
B
"""
+ assert (
+ XMLExtractor().extract(xml)
+ == "## Table\n\n| Group [rowspan=2] | Shared [colspan=2] |\n| A | B |"
+ )
+
+
+@pytest.mark.parametrize("count", [200, 201])
+def test_row_cap(count):
+ """Only the first 200 source rows survive, with an explicit truncation notice."""
+ rows = "".join(f"
Row {i}
" for i in range(count))
+ text = XMLExtractor().extract(
+ f"
{rows}
"
+ )
+ assert text.count("\n| Row ") == 200
+ assert "| Row 199 |" in text
+ assert "Row 200" not in text
+ assert ("[Table truncated after 200 rows.]" in text) == (count > 200)
+
+
+def test_stub_is_checked_before_tables():
+ """Table size cannot disguise a restricted body notice."""
+ xml = "
Text cannot be obtained from PMC.
"
+ xml += "
Data
" * 200 + "
"
+ assert XMLExtractor().extract(xml) is None
+
+
+def make_fetcher(tmp_path):
+ """Use a fresh in-memory cache on each call."""
+ return ReferenceFetcher(
+ ReferenceValidationConfig(cache_dir=tmp_path, rate_limit_delay=0)
+ )
+
+
+@pytest.mark.parametrize("version", [None, 0, 1, 99])
+def test_xml_version_roundtrip(tmp_path, version):
+ """Metadata-only rewrites neither certify old XML nor downgrade future stamps."""
+ fetcher = make_fetcher(tmp_path)
+ path = fetcher.get_cache_path("unknown:xml")
+ stamp = "" if version is None else f"xml_extraction_version: {version}\n"
+ path.write_text(
+ "---\nreference_id: unknown:xml\nextractor_version: 1\ncontent_type: full_text_xml\n"
+ + stamp
+ + "---\nOld body"
+ )
+ ref = fetcher._load_from_disk("unknown:xml", allow_stale=True)
+ fetcher._save_to_disk(ref)
+ assert ReferenceFetcher._is_stale_cache_entry(path.read_text()) == (
+ version in (None, 0)
+ )
+ assert ref.metadata.get("xml_extraction_version") == version
+ if version is None:
+ assert "xml_extraction_version" not in path.read_text()
+
+
+def test_xml_offline_stays_stale(tmp_path, caplog):
+ """Useful old XML remains available offline without falsely certifying its text."""
+ fetcher = make_fetcher(tmp_path)
+ path = fetcher.get_cache_path("unknown:xml")
+ old = "---\nreference_id: unknown:xml\nextractor_version: 1\ncontent_type: full_text_xml\n---\nOld body"
+ path.write_text(old)
+ assert fetcher.fetch("unknown:xml").content == "Old body"
+ assert path.read_text() == old
+ assert "older extractor" in caplog.text
+ assert fetcher._load_from_disk("unknown:xml") is None
+
+
+def test_provider_xml_stamp_and_abstract(tmp_path):
+ """Fresh extracted provider XML certifies its version and keeps the abstract."""
+ fetcher = make_fetcher(tmp_path)
+ ref = ReferenceContent(
+ reference_id="unknown:xml",
+ content="Original abstract",
+ content_type="abstract_only",
+ )
+ text = XMLExtractor().extract(FIXTURE.read_bytes())
+ assert fetcher.apply_full_text_location(
+ ref, FullTextLocation(text=text, format_hint="xml"), "pmc"
+ )
+ assert ref.content.startswith("Original abstract\n\n")
+ assert ref.metadata["xml_extraction_version"] == 1
+ assert make_fetcher(tmp_path)._load_from_disk(ref.reference_id) is not None
+
+
+def test_warm_legacy_xml_refreshes_from_real_source(tmp_path, monkeypatch):
+ """A registered fixture-backed source replaces legacy text once, then stays warm."""
+ from linkml_reference_validator.etl.sources.base import (
+ ReferenceSource,
+ ReferenceSourceRegistry,
+ )
+
+ reads = []
+
+ class FixtureXMLSource(ReferenceSource):
+ """Read and extract the actual PMC response using the source contract."""
+
+ @classmethod
+ def prefix(cls):
+ """Use an isolated reference namespace."""
+ return "JATSFIXTURE"
+
+ def fetch(self, identifier, config):
+ """Perform real extraction, tracking disk acquisitions."""
+ reads.append(identifier)
+ return ReferenceContent(
+ reference_id=f"JATSFIXTURE:{identifier}",
+ content=XMLExtractor().extract(FIXTURE.read_bytes()),
+ content_type="full_text_xml",
+ )
+
+ monkeypatch.setattr(ReferenceSourceRegistry, "_sources", [FixtureXMLSource])
+ fetcher = make_fetcher(tmp_path)
+ path = fetcher.get_cache_path("JATSFIXTURE:28530713")
+ path.write_text(
+ "---\nreference_id: JATSFIXTURE:28530713\nextractor_version: 1\ncontent_type: full_text_xml\n---\nLegacy prose only"
+ )
+ fresh = fetcher.fetch("JATSFIXTURE:28530713")
+ assert "| On IvIg treatment | Yes | Yes | No |" in fresh.content
+ assert fresh.metadata["xml_extraction_version"] == 1
+ assert "xml_extraction_version: 1" in path.read_text()
+ assert make_fetcher(tmp_path).fetch("JATSFIXTURE:28530713").content == fresh.content
+ assert reads == ["28530713"]
+
+
+@pytest.mark.parametrize(
+ "kind,stamp",
+ [("full_text_pdf", ""), ("full_text_html", "html_full_text_version: 1\n")],
+)
+def test_xml_migration_does_not_refresh_other_formats(tmp_path, kind, stamp):
+ """Current HTML and PDF caches remain usable without an XML version."""
+ fetcher = make_fetcher(tmp_path)
+ path = fetcher.get_cache_path("unknown:other")
+ path.write_text(
+ f"---\nreference_id: unknown:other\nextractor_version: 1\ncontent_type: {kind}\n{stamp}---\nCurrent body"
+ )
+ assert fetcher._load_from_disk("unknown:other").content == "Current body"
+
+
+def test_downloaded_xml_provider_uses_tables(tmp_path):
+ """Actual HTTP acquisition extracts the PMC fixture and stamps its cache."""
+ from functools import partial
+ from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer
+ from threading import Thread
+
+ server = ThreadingHTTPServer(
+ ("127.0.0.1", 0),
+ partial(SimpleHTTPRequestHandler, directory=str(FIXTURE.parent)),
+ )
+ thread = Thread(target=server.serve_forever, daemon=True)
+ thread.start()
+ try:
+ ref = ReferenceContent(
+ reference_id="unknown:download",
+ content_type="abstract_only",
+ content="Abstract",
+ )
+ fetcher = make_fetcher(tmp_path)
+ assert fetcher.apply_full_text_location(
+ ref,
+ FullTextLocation(
+ url=f"http://127.0.0.1:{server.server_port}/{FIXTURE.name}",
+ format_hint="xml",
+ ),
+ "local-test",
+ )
+ assert ref.content.startswith("Abstract\n\n")
+ assert "| On IvIg treatment | Yes | Yes | No |" in ref.content
+ assert ref.metadata["xml_extraction_version"] == 1
+ assert not fetcher._is_stale_cache_entry(
+ fetcher.get_cache_path(ref.reference_id).read_text()
+ )
+ finally:
+ server.shutdown()
+ server.server_close()
+ thread.join()
+
+
+def test_hidden_xml_nodes_are_not_quotable():
+ """Comments and processing instructions are metadata, while CDATA is content."""
+ xml = """
Clinical
+
Yes again
+
"""
+ assert (
+ XMLExtractor().extract(xml) == "## Clinical\n\n| Yes again | Useful text | |"
+ )
+
+
+@pytest.mark.parametrize("placement", ["body", "floats-group"])
+def test_table_footnotes_preserved_once(placement):
+ """Footnotes defining table markers remain quotable, including floats-group notes."""
+ xml = f"""<{placement}>
+
IgG
High*
+
* Measured before treatment.
+
IvIg, intravenous immunoglobulin.
+ {placement}>"""
+ assert XMLExtractor().extract(xml) == (
+ "## Table 1\n\n| IgG | High* |\n\n"
+ "* Measured before treatment. IvIg, intravenous immunoglobulin."
+ )
+
+
+def test_nested_wrapper_footnotes_are_not_duplicated():
+ """Inner and outer notes are rendered once, under their own wrappers."""
+ xml = """
Outer
+
Inner
Inner note.
+
Outer note.
"""
+ text = XMLExtractor().extract(xml)
+ assert text.count("Inner note.") == 1
+ assert text.count("Outer note.") == 1
+
+
+@pytest.mark.parametrize("tag", ["sub-article", "response"])
+def test_other_article_tables_are_not_attributed_to_main_paper(tag):
+ """Review/reply tables cannot become evidence attributed to the main article."""
+ xml = f"""
Main prose.
+
Main finding
+ <{tag}>
Reviewer prose.
Reviewer finding
+
Reviewer note.
{tag}>"""
+ text = XMLExtractor().extract(xml)
+ assert text == "Main prose.\n\n## Table\n\n| Main finding |"
+
+
+def test_nested_table_without_own_wrapper_has_own_heading():
+ """A nested physical table must not repeat the outer table's label/caption."""
+ xml = """
"]
+)
+def test_unrenderable_table_keeps_caption(table):
+ """Existing caption evidence survives even when no table cells can be extracted."""
+ xml = f"
Baseline characteristics.
{table}"
+ assert XMLExtractor().extract(xml) == "## Table 1 Baseline characteristics."
+
+
+@pytest.mark.parametrize("tag", ["sub-article", "response"])
+def test_floats_only_article_does_not_take_reviewer_body(tag):
+ """A missing main body must not select a reviewer body as fallback prose."""
+ xml = f"
Main finding
<{tag}>
Reviewer prose.
{tag}>"
+ assert XMLExtractor().extract(xml) == "## Table\n\n| Main finding |"
+
+
+def test_cell_boundary_contracts():
+ """Unit spans are implicit, backslashes escaped, and sup/sub text stays flattened."""
+ xml = r'