diff --git a/zeeguu/core/model/article.py b/zeeguu/core/model/article.py index bc754b1cc..53c5f1bf3 100644 --- a/zeeguu/core/model/article.py +++ b/zeeguu/core/model/article.py @@ -427,7 +427,7 @@ def create_article_fragments(self, session): This preserves the structure and formatting from the original article. """ from zeeguu.core.model.article_fragment import ArticleFragment - from bs4 import BeautifulSoup + from bs4 import BeautifulSoup, Tag # Get HTML content - use htmlContent if available, otherwise fall back to plain text html_content = getattr(self, "htmlContent", None) or self.source.get_content() @@ -444,30 +444,50 @@ def create_article_fragments(self, session): # Note: We skip ul/ol containers to avoid duplication, only process individual li items # Note: We skip inline elements like strong, em here as they should be preserved within their parent blocks block_elements = ["p", "h1", "h2", "h3", "h4", "h5", "h6", "li", "blockquote"] + text_blocks = [b for b in block_elements if b not in ("li", "blockquote")] + + # A list item may hold its text directly (
  • X
  • ) or in blocks + # (
  • X

  • , the shape the teacher editor saves; or a heading + # followed by paragraphs). In the latter case the blocks are emitted one + # by one, and the first of them carries the bullet. + bullet_carriers = set() for element in soup.find_all(block_elements): # Skip blockquote containers - we'll process their paragraph children instead if element.name == "blockquote": continue - # For paragraphs inside blockquotes, use special formatting to indicate they're part of a quote - if element.name == "p" and element.find_parent("blockquote"): - tag_name = "blockquote" - text_content = element.get_text().strip() - # For list items, get direct text content (not nested lists) - elif element.name == "li": - # Get only direct text content, excluding nested ul/ol + if element.name == "li": + own_blocks = [ + b + for b in element.find_all(text_blocks) + if b.find_parent("li") is element + ] + own_block_ids = {id(b) for b in own_blocks} + # Direct text content, excluding nested ul/ol and the blocks text_parts = [] for content in element.contents: - if hasattr(content, "get_text"): - # Skip nested lists - if content.name not in ["ul", "ol"]: - text_parts.append(content.get_text().strip()) - else: - # Direct text node + if not isinstance(content, Tag): text_parts.append(str(content).strip()) + elif content.name in ["ul", "ol"]: + continue + elif id(content) in own_block_ids or any( + id(b) in own_block_ids for b in content.find_all(text_blocks) + ): + continue + else: + text_parts.append(content.get_text().strip()) text_content = " ".join(text_parts).strip() tag_name = element.name + if not text_content and own_blocks: + bullet_carriers.add(id(own_blocks[0])) + elif id(element) in bullet_carriers: + text_content = element.get_text().strip() + tag_name = "li" + # For paragraphs inside blockquotes, use special formatting to indicate they're part of a quote + elif element.name == "p" and element.find_parent("blockquote"): + tag_name = "blockquote" + text_content = element.get_text().strip() else: # For other elements, preserve some inline formatting by getting inner HTML # and then extracting text, but keep track of formatting diff --git a/zeeguu/core/test/test_article_fragments.py b/zeeguu/core/test/test_article_fragments.py new file mode 100644 index 000000000..51d4cd55f --- /dev/null +++ b/zeeguu/core/test/test_article_fragments.py @@ -0,0 +1,59 @@ +from unittest import TestCase + +import zeeguu.core +from zeeguu.core.model.article_fragment import ArticleFragment +from zeeguu.core.test.model_test_mixin import ModelTestMixIn +from zeeguu.core.test.rules.article_rule import ArticleRule + +db_session = zeeguu.core.model.db.session + + +class ArticleFragmentsTest(ModelTestMixIn, TestCase): + """Each block of the article's HTML becomes exactly one reader fragment.""" + + def _fragments(self, html): + article = ArticleRule().article + ArticleFragment.query.filter_by(article_id=article.id).delete() + article.htmlContent = html + article.create_article_fragments(db_session) + db_session.commit() + return [ + (f.formatting, f.text.content) + for f in ArticleFragment.get_all_article_fragments_in_order(article.id) + ] + + def test_paragraph_inside_list_item_is_not_repeated(self): + # The shape the teacher editor (TipTap) saves every list item in + html = "

    Vessels:

    " + self.assertEqual( + [("p", "Vessels:"), ("li", "Container ships"), ("li", "Tankers")], + self._fragments(html), + ) + + def test_nested_list_items_each_appear_once(self): + html = "" + self.assertEqual( + [("li", "Ships"), ("li", "Tankers")], + self._fragments(html), + ) + + def test_list_item_with_several_blocks_keeps_them_apart(self): + html = "" + self.assertEqual( + [("li", "Container ships"), ("p", "Carry boxes."), ("p", "Largest class.")], + self._fragments(html), + ) + + def test_loose_text_before_blocks_carries_the_bullet(self): + html = "" + self.assertEqual( + [("li", "Tankers"), ("p", "Carry oil.")], + self._fragments(html), + ) + + def test_plain_list_items_and_quotes_unchanged(self): + html = "
    1. One

    Quoted

    " + self.assertEqual( + [("li", "One"), ("blockquote", "Quoted")], + self._fragments(html), + )