From 3f9b2bfc60647199cc4af478fb6063aeae95f7ec Mon Sep 17 00:00:00 2001 From: Mircea Lungu Date: Tue, 29 Sep 2026 10:34:48 +0200 Subject: [PATCH 1/2] Show a list item once when its text is wrapped in a paragraph The teacher editor (TipTap) saves every list item as
  • X

  • . create_article_fragments walks all block elements recursively, so it emitted the li (which already gathers its children's text) and then the nested p again: students saw each bullet followed by the same text as a paragraph. Any block inside a list item is now left to its nearest li. Existing articles keep their duplicated fragments until rebuilt. Co-Authored-By: Claude Opus 5.5 --- zeeguu/core/model/article.py | 5 +++ zeeguu/core/test/test_article_fragments.py | 45 ++++++++++++++++++++++ 2 files changed, 50 insertions(+) create mode 100644 zeeguu/core/test/test_article_fragments.py diff --git a/zeeguu/core/model/article.py b/zeeguu/core/model/article.py index bc754b1cc..9773a81ee 100644 --- a/zeeguu/core/model/article.py +++ b/zeeguu/core/model/article.py @@ -450,6 +450,11 @@ def create_article_fragments(self, session): if element.name == "blockquote": continue + # A block nested in a list item (e.g.
  • X

  • , the shape the + # teacher editor saves) was already emitted as part of its nearest li + if element.name != "li" and element.find_parent("li"): + continue + # For paragraphs inside blockquotes, use special formatting to indicate they're part of a quote if element.name == "p" and element.find_parent("blockquote"): tag_name = "blockquote" diff --git a/zeeguu/core/test/test_article_fragments.py b/zeeguu/core/test/test_article_fragments.py new file mode 100644 index 000000000..381780c94 --- /dev/null +++ b/zeeguu/core/test/test_article_fragments.py @@ -0,0 +1,45 @@ +from unittest import TestCase + +import zeeguu.core +from zeeguu.core.model.article_fragment import ArticleFragment +from zeeguu.core.test.model_test_mixin import ModelTestMixIn +from zeeguu.core.test.rules.article_rule import ArticleRule + +db_session = zeeguu.core.model.db.session + + +class ArticleFragmentsTest(ModelTestMixIn, TestCase): + """Each block of the article's HTML becomes exactly one reader fragment.""" + + def _fragments(self, html): + article = ArticleRule().article + ArticleFragment.query.filter_by(article_id=article.id).delete() + article.htmlContent = html + article.create_article_fragments(db_session) + db_session.commit() + return [ + (f.formatting, f.text.content) + for f in ArticleFragment.get_all_article_fragments_in_order(article.id) + ] + + def test_paragraph_inside_list_item_is_not_repeated(self): + # The shape the teacher editor (TipTap) saves every list item in + html = "

    Vessels:

    " + self.assertEqual( + [("p", "Vessels:"), ("li", "Container ships"), ("li", "Tankers")], + self._fragments(html), + ) + + def test_nested_list_items_each_appear_once(self): + html = "" + self.assertEqual( + [("li", "Ships"), ("li", "Tankers")], + self._fragments(html), + ) + + def test_plain_list_items_and_quotes_unchanged(self): + html = "
    1. One

    Quoted

    " + self.assertEqual( + [("li", "One"), ("blockquote", "Quoted")], + self._fragments(html), + ) From c4923c4591feaa20c2282d93cebe3614dbb20645 Mon Sep 17 00:00:00 2001 From: Mircea Lungu Date: Tue, 29 Sep 2026 10:42:40 +0200 Subject: [PATCH 2/2] Keep the blocks of a multi-block list item apart Skipping every block inside a list item merged an item like
  • Title

    A

    B

  • into one bullet 'Title A B'. Now a list item that holds its text in blocks emits those blocks one by one, and the first carries the bullet. Loose text before the blocks, if any, is the bullet instead. Co-Authored-By: Claude Opus 5.5 --- zeeguu/core/model/article.py | 53 ++++++++++++++-------- zeeguu/core/test/test_article_fragments.py | 14 ++++++ 2 files changed, 48 insertions(+), 19 deletions(-) diff --git a/zeeguu/core/model/article.py b/zeeguu/core/model/article.py index 9773a81ee..53c5f1bf3 100644 --- a/zeeguu/core/model/article.py +++ b/zeeguu/core/model/article.py @@ -427,7 +427,7 @@ def create_article_fragments(self, session): This preserves the structure and formatting from the original article. """ from zeeguu.core.model.article_fragment import ArticleFragment - from bs4 import BeautifulSoup + from bs4 import BeautifulSoup, Tag # Get HTML content - use htmlContent if available, otherwise fall back to plain text html_content = getattr(self, "htmlContent", None) or self.source.get_content() @@ -444,35 +444,50 @@ def create_article_fragments(self, session): # Note: We skip ul/ol containers to avoid duplication, only process individual li items # Note: We skip inline elements like strong, em here as they should be preserved within their parent blocks block_elements = ["p", "h1", "h2", "h3", "h4", "h5", "h6", "li", "blockquote"] + text_blocks = [b for b in block_elements if b not in ("li", "blockquote")] + + # A list item may hold its text directly (
  • X
  • ) or in blocks + # (
  • X

  • , the shape the teacher editor saves; or a heading + # followed by paragraphs). In the latter case the blocks are emitted one + # by one, and the first of them carries the bullet. + bullet_carriers = set() for element in soup.find_all(block_elements): # Skip blockquote containers - we'll process their paragraph children instead if element.name == "blockquote": continue - # A block nested in a list item (e.g.
  • X

  • , the shape the - # teacher editor saves) was already emitted as part of its nearest li - if element.name != "li" and element.find_parent("li"): - continue - - # For paragraphs inside blockquotes, use special formatting to indicate they're part of a quote - if element.name == "p" and element.find_parent("blockquote"): - tag_name = "blockquote" - text_content = element.get_text().strip() - # For list items, get direct text content (not nested lists) - elif element.name == "li": - # Get only direct text content, excluding nested ul/ol + if element.name == "li": + own_blocks = [ + b + for b in element.find_all(text_blocks) + if b.find_parent("li") is element + ] + own_block_ids = {id(b) for b in own_blocks} + # Direct text content, excluding nested ul/ol and the blocks text_parts = [] for content in element.contents: - if hasattr(content, "get_text"): - # Skip nested lists - if content.name not in ["ul", "ol"]: - text_parts.append(content.get_text().strip()) - else: - # Direct text node + if not isinstance(content, Tag): text_parts.append(str(content).strip()) + elif content.name in ["ul", "ol"]: + continue + elif id(content) in own_block_ids or any( + id(b) in own_block_ids for b in content.find_all(text_blocks) + ): + continue + else: + text_parts.append(content.get_text().strip()) text_content = " ".join(text_parts).strip() tag_name = element.name + if not text_content and own_blocks: + bullet_carriers.add(id(own_blocks[0])) + elif id(element) in bullet_carriers: + text_content = element.get_text().strip() + tag_name = "li" + # For paragraphs inside blockquotes, use special formatting to indicate they're part of a quote + elif element.name == "p" and element.find_parent("blockquote"): + tag_name = "blockquote" + text_content = element.get_text().strip() else: # For other elements, preserve some inline formatting by getting inner HTML # and then extracting text, but keep track of formatting diff --git a/zeeguu/core/test/test_article_fragments.py b/zeeguu/core/test/test_article_fragments.py index 381780c94..51d4cd55f 100644 --- a/zeeguu/core/test/test_article_fragments.py +++ b/zeeguu/core/test/test_article_fragments.py @@ -37,6 +37,20 @@ def test_nested_list_items_each_appear_once(self): self._fragments(html), ) + def test_list_item_with_several_blocks_keeps_them_apart(self): + html = "
    • Container ships

      Carry boxes.

      Largest class.

    " + self.assertEqual( + [("li", "Container ships"), ("p", "Carry boxes."), ("p", "Largest class.")], + self._fragments(html), + ) + + def test_loose_text_before_blocks_carries_the_bullet(self): + html = "
    • Tankers

      Carry oil.

    " + self.assertEqual( + [("li", "Tankers"), ("p", "Carry oil.")], + self._fragments(html), + ) + def test_plain_list_items_and_quotes_unchanged(self): html = "
    1. One

    Quoted

    " self.assertEqual(