diff --git a/zeeguu/core/model/article.py b/zeeguu/core/model/article.py
index bc754b1cc..53c5f1bf3 100644
--- a/zeeguu/core/model/article.py
+++ b/zeeguu/core/model/article.py
@@ -427,7 +427,7 @@ def create_article_fragments(self, session):
This preserves the structure and formatting from the original article.
"""
from zeeguu.core.model.article_fragment import ArticleFragment
- from bs4 import BeautifulSoup
+ from bs4 import BeautifulSoup, Tag
# Get HTML content - use htmlContent if available, otherwise fall back to plain text
html_content = getattr(self, "htmlContent", None) or self.source.get_content()
@@ -444,30 +444,50 @@ def create_article_fragments(self, session):
# Note: We skip ul/ol containers to avoid duplication, only process individual li items
# Note: We skip inline elements like strong, em here as they should be preserved within their parent blocks
block_elements = ["p", "h1", "h2", "h3", "h4", "h5", "h6", "li", "blockquote"]
+ text_blocks = [b for b in block_elements if b not in ("li", "blockquote")]
+
+ # A list item may hold its text directly (
X) or in blocks
+ # (X
, the shape the teacher editor saves; or a heading
+ # followed by paragraphs). In the latter case the blocks are emitted one
+ # by one, and the first of them carries the bullet.
+ bullet_carriers = set()
for element in soup.find_all(block_elements):
# Skip blockquote containers - we'll process their paragraph children instead
if element.name == "blockquote":
continue
- # For paragraphs inside blockquotes, use special formatting to indicate they're part of a quote
- if element.name == "p" and element.find_parent("blockquote"):
- tag_name = "blockquote"
- text_content = element.get_text().strip()
- # For list items, get direct text content (not nested lists)
- elif element.name == "li":
- # Get only direct text content, excluding nested ul/ol
+ if element.name == "li":
+ own_blocks = [
+ b
+ for b in element.find_all(text_blocks)
+ if b.find_parent("li") is element
+ ]
+ own_block_ids = {id(b) for b in own_blocks}
+ # Direct text content, excluding nested ul/ol and the blocks
text_parts = []
for content in element.contents:
- if hasattr(content, "get_text"):
- # Skip nested lists
- if content.name not in ["ul", "ol"]:
- text_parts.append(content.get_text().strip())
- else:
- # Direct text node
+ if not isinstance(content, Tag):
text_parts.append(str(content).strip())
+ elif content.name in ["ul", "ol"]:
+ continue
+ elif id(content) in own_block_ids or any(
+ id(b) in own_block_ids for b in content.find_all(text_blocks)
+ ):
+ continue
+ else:
+ text_parts.append(content.get_text().strip())
text_content = " ".join(text_parts).strip()
tag_name = element.name
+ if not text_content and own_blocks:
+ bullet_carriers.add(id(own_blocks[0]))
+ elif id(element) in bullet_carriers:
+ text_content = element.get_text().strip()
+ tag_name = "li"
+ # For paragraphs inside blockquotes, use special formatting to indicate they're part of a quote
+ elif element.name == "p" and element.find_parent("blockquote"):
+ tag_name = "blockquote"
+ text_content = element.get_text().strip()
else:
# For other elements, preserve some inline formatting by getting inner HTML
# and then extracting text, but keep track of formatting
diff --git a/zeeguu/core/test/test_article_fragments.py b/zeeguu/core/test/test_article_fragments.py
new file mode 100644
index 000000000..51d4cd55f
--- /dev/null
+++ b/zeeguu/core/test/test_article_fragments.py
@@ -0,0 +1,59 @@
+from unittest import TestCase
+
+import zeeguu.core
+from zeeguu.core.model.article_fragment import ArticleFragment
+from zeeguu.core.test.model_test_mixin import ModelTestMixIn
+from zeeguu.core.test.rules.article_rule import ArticleRule
+
+db_session = zeeguu.core.model.db.session
+
+
+class ArticleFragmentsTest(ModelTestMixIn, TestCase):
+ """Each block of the article's HTML becomes exactly one reader fragment."""
+
+ def _fragments(self, html):
+ article = ArticleRule().article
+ ArticleFragment.query.filter_by(article_id=article.id).delete()
+ article.htmlContent = html
+ article.create_article_fragments(db_session)
+ db_session.commit()
+ return [
+ (f.formatting, f.text.content)
+ for f in ArticleFragment.get_all_article_fragments_in_order(article.id)
+ ]
+
+ def test_paragraph_inside_list_item_is_not_repeated(self):
+ # The shape the teacher editor (TipTap) saves every list item in
+ html = "Vessels:
"
+ self.assertEqual(
+ [("p", "Vessels:"), ("li", "Container ships"), ("li", "Tankers")],
+ self._fragments(html),
+ )
+
+ def test_nested_list_items_each_appear_once(self):
+ html = ""
+ self.assertEqual(
+ [("li", "Ships"), ("li", "Tankers")],
+ self._fragments(html),
+ )
+
+ def test_list_item_with_several_blocks_keeps_them_apart(self):
+ html = "Container ships
Carry boxes.
Largest class.
"
+ self.assertEqual(
+ [("li", "Container ships"), ("p", "Carry boxes."), ("p", "Largest class.")],
+ self._fragments(html),
+ )
+
+ def test_loose_text_before_blocks_carries_the_bullet(self):
+ html = ""
+ self.assertEqual(
+ [("li", "Tankers"), ("p", "Carry oil.")],
+ self._fragments(html),
+ )
+
+ def test_plain_list_items_and_quotes_unchanged(self):
+ html = "- One
Quoted
"
+ self.assertEqual(
+ [("li", "One"), ("blockquote", "Quoted")],
+ self._fragments(html),
+ )