Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
48 changes: 34 additions & 14 deletions zeeguu/core/model/article.py
Original file line number Diff line number Diff line change
Expand Up @@ -427,7 +427,7 @@ def create_article_fragments(self, session):
This preserves the structure and formatting from the original article.
"""
from zeeguu.core.model.article_fragment import ArticleFragment
from bs4 import BeautifulSoup
from bs4 import BeautifulSoup, Tag

# Get HTML content - use htmlContent if available, otherwise fall back to plain text
html_content = getattr(self, "htmlContent", None) or self.source.get_content()
Expand All @@ -444,30 +444,50 @@ def create_article_fragments(self, session):
# Note: We skip ul/ol containers to avoid duplication, only process individual li items
# Note: We skip inline elements like strong, em here as they should be preserved within their parent blocks
block_elements = ["p", "h1", "h2", "h3", "h4", "h5", "h6", "li", "blockquote"]
text_blocks = [b for b in block_elements if b not in ("li", "blockquote")]

# A list item may hold its text directly (<li>X</li>) or in blocks
# (<li><p>X</p></li>, the shape the teacher editor saves; or a heading
# followed by paragraphs). In the latter case the blocks are emitted one
# by one, and the first of them carries the bullet.
bullet_carriers = set()

for element in soup.find_all(block_elements):
# Skip blockquote containers - we'll process their paragraph children instead
if element.name == "blockquote":
continue

# For paragraphs inside blockquotes, use special formatting to indicate they're part of a quote
if element.name == "p" and element.find_parent("blockquote"):
tag_name = "blockquote"
text_content = element.get_text().strip()
# For list items, get direct text content (not nested lists)
elif element.name == "li":
# Get only direct text content, excluding nested ul/ol
if element.name == "li":
own_blocks = [
b
for b in element.find_all(text_blocks)
if b.find_parent("li") is element
]
own_block_ids = {id(b) for b in own_blocks}
# Direct text content, excluding nested ul/ol and the blocks
text_parts = []
for content in element.contents:
if hasattr(content, "get_text"):
# Skip nested lists
if content.name not in ["ul", "ol"]:
text_parts.append(content.get_text().strip())
else:
# Direct text node
if not isinstance(content, Tag):
text_parts.append(str(content).strip())
elif content.name in ["ul", "ol"]:
continue
elif id(content) in own_block_ids or any(
id(b) in own_block_ids for b in content.find_all(text_blocks)
):
continue
else:
text_parts.append(content.get_text().strip())
text_content = " ".join(text_parts).strip()
tag_name = element.name
if not text_content and own_blocks:
bullet_carriers.add(id(own_blocks[0]))
elif id(element) in bullet_carriers:
text_content = element.get_text().strip()
tag_name = "li"
# For paragraphs inside blockquotes, use special formatting to indicate they're part of a quote
elif element.name == "p" and element.find_parent("blockquote"):
tag_name = "blockquote"
text_content = element.get_text().strip()
else:
# For other elements, preserve some inline formatting by getting inner HTML
# and then extracting text, but keep track of formatting
Expand Down
59 changes: 59 additions & 0 deletions zeeguu/core/test/test_article_fragments.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,59 @@
from unittest import TestCase

import zeeguu.core
from zeeguu.core.model.article_fragment import ArticleFragment
from zeeguu.core.test.model_test_mixin import ModelTestMixIn
from zeeguu.core.test.rules.article_rule import ArticleRule

db_session = zeeguu.core.model.db.session


class ArticleFragmentsTest(ModelTestMixIn, TestCase):
"""Each block of the article's HTML becomes exactly one reader fragment."""

def _fragments(self, html):
article = ArticleRule().article
ArticleFragment.query.filter_by(article_id=article.id).delete()
article.htmlContent = html
article.create_article_fragments(db_session)
db_session.commit()
return [
(f.formatting, f.text.content)
for f in ArticleFragment.get_all_article_fragments_in_order(article.id)
]

def test_paragraph_inside_list_item_is_not_repeated(self):
# The shape the teacher editor (TipTap) saves every list item in
html = "<p>Vessels:</p><ul><li><p>Container ships</p></li><li><p>Tankers</p></li></ul>"
self.assertEqual(
[("p", "Vessels:"), ("li", "Container ships"), ("li", "Tankers")],
self._fragments(html),
)

def test_nested_list_items_each_appear_once(self):
html = "<ul><li><p>Ships</p><ul><li><p>Tankers</p></li></ul></li></ul>"
self.assertEqual(
[("li", "Ships"), ("li", "Tankers")],
self._fragments(html),
)

def test_list_item_with_several_blocks_keeps_them_apart(self):
html = "<ul><li><h3>Container ships</h3><p>Carry boxes.</p><p>Largest class.</p></li></ul>"
self.assertEqual(
[("li", "Container ships"), ("p", "Carry boxes."), ("p", "Largest class.")],
self._fragments(html),
)

def test_loose_text_before_blocks_carries_the_bullet(self):
html = "<ul><li><strong>Tankers</strong><p>Carry oil.</p></li></ul>"
self.assertEqual(
[("li", "Tankers"), ("p", "Carry oil.")],
self._fragments(html),
)

def test_plain_list_items_and_quotes_unchanged(self):
html = "<ol><li>One</li></ol><blockquote><p>Quoted</p></blockquote>"
self.assertEqual(
[("li", "One"), ("blockquote", "Quoted")],
self._fragments(html),
)
Loading