From e5f7143cbdd543e80ac2b925445bdcae73cb4fdb Mon Sep 17 00:00:00 2001 From: James Kent Date: Thu, 1 Oct 2026 23:50:43 -0500 Subject: [PATCH] Add keep_tables option to keep tables in the article text Readability keeps only paragraphs and headings, so every table was dropped from Article.text. With keep_tables=True each innermost table is swapped for a placeholder before cleaning and replaced by its rows afterwards (caption, then one tab-separated line per row, spanned cells repeated). A table whose placeholder readability discarded is appended at the end. table_text is public so callers can render tables ACE fetched from linked pages the same way. Co-Authored-By: Claude Opus 5.5 (1M context) --- ace/sources.py | 95 ++++++++++++++++++++++++++++++++++++++++++- ace/tests/test_ace.py | 38 +++++++++++++++++ 2 files changed, 131 insertions(+), 2 deletions(-) diff --git a/ace/sources.py b/ace/sources.py index df68bf2..f5a04f4 100644 --- a/ace/sources.py +++ b/ace/sources.py @@ -5,6 +5,7 @@ import os import json import abc +import copy import importlib from glob import glob from urllib.parse import urljoin, urlparse @@ -51,6 +52,51 @@ r"(? [text, rows still covered] + for tr in table.find_all("tr"): + row = [] + + def _fill_spans(): + while len(row) in spans: + column = len(row) + row.append(spans[column][0]) + spans[column][1] -= 1 + if spans[column][1] == 0: + del spans[column] + + for cell in tr.find_all(["td", "th"], recursive=False): + _fill_spans() + text = " ".join(cell.get_text(" ").split()) + colspan = _span(cell, "colspan") + rowspan = _span(cell, "rowspan") + for _ in range(colspan): + if rowspan > 1: + spans[len(row)] = [text, rowspan - 1] + row.append(text) + _fill_spans() + if any(row): + rows.append("\t".join(row)) + caption = table.find("caption") + caption_text = " ".join(caption.get_text(" ").split()) if caption else "" + return "\n".join(part for part in [caption_text] + rows if part) + + +def _span(cell, attribute): + try: + return max(1, min(int(cell.get(attribute, 1)), 100)) + except (TypeError, ValueError): + return 1 + # Try to import readabilipy for enhanced HTML cleaning try: from readabilipy import simple_json_from_html_string @@ -179,6 +225,46 @@ def _clean_html_with_readability(self, html): logger.warning(f"Error using readabilipy, falling back to basic HTML cleaning: {e}") return self._safe_clean_html(html) + def _text_with_tables(self, soup): + """Article text with every table kept, at its place where possible. + + Readability keeps only paragraphs and headings, so tables are lost + from the text. Each innermost table is swapped for a placeholder + paragraph before cleaning, then replaced by its rows. A table whose + placeholder readability discarded is appended at the end, unless + an identical table was already placed (pages often repeat one in a + hidden pop-up). + """ + work = copy.copy(soup) + tables = {} + for index, table in enumerate( + [t for t in work.find_all("table") if not t.find("table")]): + key = str(index) + tables[key] = table_text(table) + placeholder = work.new_tag("p") + placeholder.string = "[ace-table-%s]" % key + table.replace_with(placeholder) + + text = self._clean_html_with_readability(str(work)) + + placed = set() + + def _substitute(match): + placed.add(match.group(1)) + return tables.get(match.group(1), "") + + text = _TABLE_PLACEHOLDER.sub(_substitute, text) + seen = {tables[key] for key in placed} + leftover = [] + for key, rendered in tables.items(): + if key in placed or not rendered or rendered in seen: + continue + seen.add(rendered) + leftover.append(rendered) + if leftover: + text = "\n\n".join([text.rstrip()] + leftover) + return text + def _safe_clean_html(self, html): """ Clean HTML content using BeautifulSoup as a fallback. @@ -240,11 +326,13 @@ def parse_article( pmid=None, metadata_dir=None, skip_metadata=False, + keep_tables=False, **kwargs, ): ''' Takes HTML article as input and returns an Article. PMID Can also be passed, which prevents having to scrape it from the article and/or look it - up in PubMed. ''' + up in PubMed. If keep_tables is True, each table is kept in the article + text as tab-separated rows; see `_text_with_tables`. ''' html = self.decode_html_entities(html) soup = BeautifulSoup(html, "lxml") @@ -264,7 +352,10 @@ def parse_article( script.extract() # Get text using readability - text = self._clean_html_with_readability(str(soup)) + if keep_tables: + text = self._text_with_tables(soup) + else: + text = self._clean_html_with_readability(str(soup)) self.article = database.Article(text, pmid=pmid, metadata=metadata) self.extract_neurovault(soup) diff --git a/ace/tests/test_ace.py b/ace/tests/test_ace.py index d18a613..928b8f3 100644 --- a/ace/tests/test_ace.py +++ b/ace/tests/test_ace.py @@ -3,6 +3,7 @@ from os.path import dirname, join, exists, sep as pathsep import pytest +from bs4 import BeautifulSoup from ace import sources, database, export, scrape, ingest from ace import tableparser @@ -723,3 +724,40 @@ def test_additional_missed_in_main_text_regressions(test_weird_data_path, source assert article is not None assert len(article.tables) >= 1 assert _count_valid_activations(article.tables) >= 1 + + +def test_table_text_repeats_spanned_cells(): + table = BeautifulSoup( + "" + "" + "" + "
Peaks
RegionMNI
xyz
IFG-42186
", + "lxml", + ).table + assert sources.table_text(table) == ( + "Peaks\nRegion\tMNI\tMNI\tMNI\nRegion\tx\ty\tz\nIFG\t-42\t18\t6" + ) + + +def test_keep_tables_puts_tables_in_the_text(test_data_path, source_manager): + html = open(join(test_data_path, 'pmc.html')).read() + source = source_manager.identify_source(html) + plain = source.parse_article(html, pmid='1', skip_metadata=True).text + kept = source.parse_article( + html, pmid='1', skip_metadata=True, keep_tables=True).text + rows = [line for line in kept.splitlines() if line.count('\t') >= 2] + assert rows + assert not any(line.count('\t') >= 2 for line in plain.splitlines()) + assert '[ace-table-' not in kept + + +def test_a_table_readability_drops_is_appended(source_manager): + source = source_manager.sources['Default'] if 'Default' in source_manager.sources \ + else next(iter(source_manager.sources.values())) + source._clean_html_with_readability = lambda html: "Body text." + soup = BeautifulSoup( + "

Body text.

" + "
ab
" + "" + "", "lxml") + assert source._text_with_tables(soup) == "Body text.\n\na\tb"