Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
95 changes: 93 additions & 2 deletions ace/sources.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
import os
import json
import abc
import copy
import importlib
from glob import glob
from urllib.parse import urljoin, urlparse
Expand Down Expand Up @@ -51,6 +52,51 @@
r"(?<!\d)[+\-−–—]?\d{1,3}\s*[,;/|\t ]\s*[+\-−–—]?\d{1,3}\s*[,;/|\t ]\s*[+\-−–—]?\d{1,3}(?!\d)"
)

_TABLE_PLACEHOLDER = re.compile(r"\[ace-table-(\d+)\]")


def table_text(table):
"""A table as text: its caption, then one tab-separated line per row.

Spanned cells are repeated across the columns and rows they cover, as
pandas.read_html does, so a value stays in its column.
"""
rows = []
spans = {} # column -> [text, rows still covered]
for tr in table.find_all("tr"):
row = []

def _fill_spans():
while len(row) in spans:
column = len(row)
row.append(spans[column][0])
spans[column][1] -= 1
if spans[column][1] == 0:
del spans[column]

for cell in tr.find_all(["td", "th"], recursive=False):
_fill_spans()
text = " ".join(cell.get_text(" ").split())
colspan = _span(cell, "colspan")
rowspan = _span(cell, "rowspan")
for _ in range(colspan):
if rowspan > 1:
spans[len(row)] = [text, rowspan - 1]
row.append(text)
_fill_spans()
if any(row):
rows.append("\t".join(row))
caption = table.find("caption")
caption_text = " ".join(caption.get_text(" ").split()) if caption else ""
return "\n".join(part for part in [caption_text] + rows if part)


def _span(cell, attribute):
try:
return max(1, min(int(cell.get(attribute, 1)), 100))
except (TypeError, ValueError):
return 1

# Try to import readabilipy for enhanced HTML cleaning
try:
from readabilipy import simple_json_from_html_string
Expand Down Expand Up @@ -179,6 +225,46 @@ def _clean_html_with_readability(self, html):
logger.warning(f"Error using readabilipy, falling back to basic HTML cleaning: {e}")
return self._safe_clean_html(html)

def _text_with_tables(self, soup):
"""Article text with every table kept, at its place where possible.

Readability keeps only paragraphs and headings, so tables are lost
from the text. Each innermost table is swapped for a placeholder
paragraph before cleaning, then replaced by its rows. A table whose
placeholder readability discarded is appended at the end, unless
an identical table was already placed (pages often repeat one in a
hidden pop-up).
"""
work = copy.copy(soup)
tables = {}
for index, table in enumerate(
[t for t in work.find_all("table") if not t.find("table")]):
key = str(index)
tables[key] = table_text(table)
placeholder = work.new_tag("p")
placeholder.string = "[ace-table-%s]" % key
table.replace_with(placeholder)

text = self._clean_html_with_readability(str(work))

placed = set()

def _substitute(match):
placed.add(match.group(1))
return tables.get(match.group(1), "")

text = _TABLE_PLACEHOLDER.sub(_substitute, text)
seen = {tables[key] for key in placed}
leftover = []
for key, rendered in tables.items():
if key in placed or not rendered or rendered in seen:
continue
seen.add(rendered)
leftover.append(rendered)
if leftover:
text = "\n\n".join([text.rstrip()] + leftover)
return text

def _safe_clean_html(self, html):
"""
Clean HTML content using BeautifulSoup as a fallback.
Expand Down Expand Up @@ -240,11 +326,13 @@ def parse_article(
pmid=None,
metadata_dir=None,
skip_metadata=False,
keep_tables=False,
**kwargs,
):
''' Takes HTML article as input and returns an Article. PMID Can also be
passed, which prevents having to scrape it from the article and/or look it
up in PubMed. '''
up in PubMed. If keep_tables is True, each table is kept in the article
text as tab-separated rows; see `_text_with_tables`. '''

html = self.decode_html_entities(html)
soup = BeautifulSoup(html, "lxml")
Expand All @@ -264,7 +352,10 @@ def parse_article(
script.extract()

# Get text using readability
text = self._clean_html_with_readability(str(soup))
if keep_tables:
text = self._text_with_tables(soup)
else:
text = self._clean_html_with_readability(str(soup))

self.article = database.Article(text, pmid=pmid, metadata=metadata)
self.extract_neurovault(soup)
Expand Down
38 changes: 38 additions & 0 deletions ace/tests/test_ace.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
from os.path import dirname, join, exists, sep as pathsep

import pytest
from bs4 import BeautifulSoup

from ace import sources, database, export, scrape, ingest
from ace import tableparser
Expand Down Expand Up @@ -723,3 +724,40 @@ def test_additional_missed_in_main_text_regressions(test_weird_data_path, source
assert article is not None
assert len(article.tables) >= 1
assert _count_valid_activations(article.tables) >= 1


def test_table_text_repeats_spanned_cells():
table = BeautifulSoup(
"<table><caption>Peaks</caption>"
"<tr><th rowspan=2>Region</th><th colspan=3>MNI</th></tr>"
"<tr><td>x</td><td>y</td><td>z</td></tr>"
"<tr><td>IFG</td><td>-42</td><td>18</td><td>6</td></tr></table>",
"lxml",
).table
assert sources.table_text(table) == (
"Peaks\nRegion\tMNI\tMNI\tMNI\nRegion\tx\ty\tz\nIFG\t-42\t18\t6"
)


def test_keep_tables_puts_tables_in_the_text(test_data_path, source_manager):
html = open(join(test_data_path, 'pmc.html')).read()
source = source_manager.identify_source(html)
plain = source.parse_article(html, pmid='1', skip_metadata=True).text
kept = source.parse_article(
html, pmid='1', skip_metadata=True, keep_tables=True).text
rows = [line for line in kept.splitlines() if line.count('\t') >= 2]
assert rows
assert not any(line.count('\t') >= 2 for line in plain.splitlines())
assert '[ace-table-' not in kept


def test_a_table_readability_drops_is_appended(source_manager):
source = source_manager.sources['Default'] if 'Default' in source_manager.sources \
else next(iter(source_manager.sources.values()))
source._clean_html_with_readability = lambda html: "Body text."
soup = BeautifulSoup(
"<html><body><p>Body text.</p>"
"<table><tr><td>a</td><td>b</td></tr></table>"
"<div hidden><table><tr><td>a</td><td>b</td></tr></table></div>"
"</body></html>", "lxml")
assert source._text_with_tables(soup) == "Body text.\n\na\tb"
Loading