Skip to content

Commit 6204ebf

Browse files
committed
feat: move away from Kindlegen to Kindling
All `<br>` tags are forced to `<br/>` in order to comply with Kindle XHTML requirements.
1 parent 3da1631 commit 6204ebf

8 files changed

Lines changed: 50 additions & 171 deletions

File tree

tests/test_3_convert.py

Lines changed: 13 additions & 23 deletions
Original file line numberDiff line numberDiff line change
@@ -60,7 +60,7 @@ def setup_logging(*args: str, **kwargs: str) -> None:
6060
# Ensure summaries are properly handled
6161
assert (
6262
len([record for record in log_records if "Effective words + variants" in record])
63-
== 3 * 2 # (KoboFormat + DictFileFormat + DictFileFormatForMobi) * (etym + noetym)
63+
== 2 * 2 # (KoboFormat + DictFileFormat) * (etym + noetym)
6464
)
6565

6666
# Check Mobi warnings
@@ -227,12 +227,10 @@ def setup_logging(*args: str, **kwargs: str) -> None:
227227
files = sorted(path.relative_to(tempdir).as_posix() for path in Path(tempdir).glob("**/*"))
228228
expected_files = [
229229
"HDImages",
230-
"kindlegenbuild.log",
231230
"mobi7",
232231
"mobi7/Images",
233-
"mobi7/Images/cover00022.jpeg",
234-
"mobi7/Images/image00021.gif",
235-
"mobi7/Images/image00024.jpeg",
232+
"mobi7/Images/cover00009.png",
233+
"mobi7/Images/image00010.gif",
236234
"mobi7/book.html",
237235
"mobi7/content.opf",
238236
"mobi7/toc.ncx",
@@ -509,7 +507,7 @@ def test_kindle_format_variants_from_uppercase_only_word(tmp_path: Path) -> None
509507
"""See issue #2623."""
510508
words = WORDS_VARIANTS_RU
511509
variants = convert.make_variants(words)
512-
formatter = convert.DictFileFormatForMobi("ru", tmp_path, words, variants, "20260122")
510+
formatter = convert.MobiFormat("ru", tmp_path, words, variants, "20260122")
513511

514512
ФСБ = "".join(formatter.handle_word("ФСБ", words))
515513
assert "@ ФСБ" in ФСБ
@@ -622,7 +620,6 @@ def test_sublang(locale: str, lang_src: str, lang_dst: str, tmp_path: Path) -> N
622620
patch.object(convert, "load") as mocked_l,
623621
patch.object(convert, "make_variants") as mocked_mv,
624622
patch.object(convert, "distribute_workload") as mocked_dw,
625-
patch.object(convert, "run_mobi_formatter") as mocked_rmf,
626623
):
627624
mocked_gljf.return_value = pages
628625
mocked_l.return_value = words
@@ -638,45 +635,38 @@ def test_sublang(locale: str, lang_src: str, lang_dst: str, tmp_path: Path) -> N
638635
for include_etymology in [False, True]:
639636
mocked_dw.assert_any_call(convert.get_primary_formatters(), *args, include_etymology=include_etymology)
640637
mocked_dw.assert_any_call(convert.get_secondary_formatters(), *args, include_etymology=False)
641-
mocked_rmf.assert_any_call(*args, include_etymology=False)
642638
assert mocked_dw.call_count == 4
643-
assert mocked_rmf.call_count == 2
644639

645640

646641
@pytest.mark.parametrize("format", list(convert.FORMATTERS.keys()))
647642
def test_format(format: str) -> None:
648-
primary, secondary, mobi_run = convert.get_formatters(format)
643+
primary, secondary = convert.get_formatters(format)
649644
assert primary == {convert.FORMATTERS[format][0]}
650645
if secondary:
651646
assert secondary == {convert.FORMATTERS[format][1]}
652-
assert not mobi_run
653647

654648

655649
@pytest.mark.parametrize("format", ["mobi", "kindle"])
656650
def test_format_mobi(format: str) -> None:
657-
primary, secondary, mobi_run = convert.get_formatters(format)
658-
assert not primary
659-
assert not secondary
660-
assert mobi_run
651+
primary, secondary = convert.get_formatters(format)
652+
assert primary == {convert.FORMATTERS["mobi"][0]}
653+
assert secondary == {convert.FORMATTERS["mobi"][1]}
661654

662655

663656
@pytest.mark.parametrize("format", ["", "all"])
664657
def test_format_all(format: str) -> None:
665-
primary, secondary, mobi_run = convert.get_formatters(format)
658+
primary, secondary = convert.get_formatters(format)
666659
assert primary == convert.get_primary_formatters()
667660
assert secondary == convert.get_secondary_formatters()
668-
assert mobi_run
669661

670662

671663
def test_format_unknown() -> None:
672-
primary, secondary, mobi_run = convert.get_formatters("unknown")
664+
primary, secondary = convert.get_formatters("unknown")
673665
assert not primary
674666
assert not secondary
675-
assert not mobi_run
676667

677668

678669
def test_formats() -> None:
679-
primary, secondary, mobi_run = convert.get_formatters("df,mobi")
680-
assert primary == {convert.FORMATTERS["df"][0]}
681-
assert secondary == {convert.FORMATTERS["df"][1]}
682-
assert mobi_run
670+
primary, secondary = convert.get_formatters("df,mobi")
671+
assert primary == {convert.FORMATTERS["df"][0], convert.FORMATTERS["mobi"][0]}
672+
assert secondary == {convert.FORMATTERS["df"][1], convert.FORMATTERS["mobi"][1]}

tests/test_zh.py

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -68,7 +68,7 @@ def setup_lua_ctx() -> None:
6868
"如《禮記》所述:",
6969
"<dl>前有摯獸,則載<b>貔貅</b>。 &#91;文言文,繁體&#93;<br>前有挚兽,则载<b>貔貅</b>。 &#91;文言文,簡體&#93;<dd><small>出自:《禮記》,約公元前4 – 前2世紀</small></dd><dd><i>Qián yǒu zhìshòu, zé zǎi <b>píxiū</b>.</i> &#91;漢語拼音&#93;</dd><dd>當前面有兇猛的野獸(獵物)時,應懸掛披有<b>貔貅</b>(豹皮)的旗幟。</dd></dl>",
7070
"如《逸周书》所述:",
71-
"<dl>山之深也,虎豹<b>貔貅</b>何為可服? &#91;文言文,繁體&#93;<br>山之深也,虎豹<b>貔貅</b>何为可服? &#91;文言文,簡體&#93;<dd><small>出自:《逸周書》,約公元前4 – 前1世紀</small></dd><dd><i>Shān zhī shēn yě, hǔbào <b>píxiū</b> héwèi kě fú?</i> &#91;漢語拼音&#93;</dd><dd>山如此之深,虎、豹、<b>貔貅</b>如何馴服?</dd></dl>",
71+
"<dl>山之深也,虎豹<b>貔貅</b>何為可服? &#91;文言文,繁體&#93;<br/>山之深也,虎豹<b>貔貅</b>何为可服? &#91;文言文,簡體&#93;<dd><small>出自:《逸周書》,約公元前4 – 前1世紀</small></dd><dd><i>Shān zhī shēn yě, hǔbào <b>píxiū</b> héwèi kě fú?</i> &#91;漢語拼音&#93;</dd><dd>山如此之深,虎、豹、<b>貔貅</b>如何馴服?</dd></dl>",
7272
],
7373
{
7474
"名詞": ["(<i>中國神話</i>) 傳說的一種瑞獸,能帶來歡樂及好運", "(<i>比喻義</i>) 勇猛的戰士"],
@@ -85,7 +85,7 @@ def setup_lua_ctx() -> None:
8585
"最早見於東晉孫盛《晉陽秋》(4世紀)中記載,桓溫紀念譙秀而作的一則上表(347年),後來又被南朝宋史學家裴松之引用在《三國志注》(5世紀早期)中。",
8686
"<dl>於時皇極遘道消之會,群黎蹈顛沛之艱,<b>中華</b>有顧瞻之哀,幽谷無遷喬之望。 &#91;文言文,繁體&#93;<br>于时皇极遘道消之会,群黎蹈颠沛之艰,<b>中华</b>有顾瞻之哀,幽谷无迁乔之望。 &#91;文言文,簡體&#93;<dd><small>出自:裴松之,《三国志注》,約公元5世紀</small></dd><dd><i>Yú shí huángjí gòu dàoxiāo zhī huì, qúnlí dǎo diānpèi zhī jiān, <b>zhōnghuá</b> yǒu gùzhān zhī āi, yōugǔ wú qiānqiáo zhī wàng.</i> &#91;漢語拼音&#93;</dd><dd>這時朝廷遇上衰落,民眾生活流離艱苦,<b>中原</b>[國家]有敗亡的憂慮,百姓沒有出頭高升的希望。</dd></dl>",
8787
"裴松之在為《諸葛亮傳》作注時,也使用了中華一詞。",
88-
"<dl>若使游步<b>中華</b>,騁其龍光,豈夫多士所能沈翳哉! &#91;文言文,繁體&#93;<br>若使游步<b>中华</b>,骋其龙光,岂夫多士所能沈翳哉! &#91;文言文,簡體&#93;<dd><small>出自:裴松之,《三国志注》,約公元5世紀</small></dd><dd><i>Ruò shǐ yóubù <b>Zhōnghuá</b>, chěng qí lóngguāng, qǐ fū duō shì suǒ néng shěnyì zāi!</i> &#91;漢語拼音&#93;</dd></dl>",
88+
"<dl>若使游步<b>中華</b>,騁其龍光,豈夫多士所能沈翳哉! &#91;文言文,繁體&#93;<br/>若使游步<b>中华</b>,骋其龙光,岂夫多士所能沈翳哉! &#91;文言文,簡體&#93;<dd><small>出自:裴松之,《三国志注》,約公元5世紀</small></dd><dd><i>Ruò shǐ yóubù <b>Zhōnghuá</b>, chěng qí lóngguāng, qǐ fū duō shì suǒ néng shěnyì zāi!</i> &#91;漢語拼音&#93;</dd></dl>",
8989
],
9090
{"專有名詞": ["(正式,詩歌,exalted) 中國(多指文化、文明、民族等方面)", "(~里) 位於臺灣臺北松山區的里"]},
9191
[],

wikidict/constants.py

Lines changed: 1 addition & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -35,14 +35,9 @@
3535
# Syntax: "locale": "origin locale"
3636
LOCALE_ORIGIN = {"fro": "fr"}
3737

38-
# Dictionaries known to be problematic about the number of chars in MobiPocket
39-
MOBI_CLEANUP = {"en", "en:en", "fr", "fr:fr"}
40-
# Dictionaries known to be problematic about the file size in MobiPocket
41-
MOBI_SKIP: set[str] = {"ja", "ja:ja"}
42-
4338
# Mobi
4439
COVER_FILE = Path(__file__).parent / "cover.png"
45-
KINDLEGEN_FILE = Path.home() / ".local" / "bin" / "kindlegen"
40+
MOBIPOCKET_TOOL = Path.home() / ".local" / "bin" / "kindling"
4641

4742
# HTTP requests
4843
SESSION = requests.Session()

wikidict/context.py

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -520,7 +520,7 @@ def clean_html_output(html: str, locale: str) -> str:
520520

521521
# Remove those tags
522522
html = re.sub(r"</?(?:a|bdi|cite|div|em|li|ol|p|span|strong|templatestyles|ul)[^>]*>", "", html)
523-
html = html.replace("<hr>", "<br>")
523+
html = html.replace("<hr>", "<br/>")
524524

525525
# Clean-up attributes from those tags
526526
html = re.sub(r"<(b|dl|i|small|sub|sup)\s+[^>]+>", r"<\1>", html)

wikidict/convert.py

Lines changed: 22 additions & 128 deletions
Original file line numberDiff line numberDiff line change
@@ -323,7 +323,7 @@ def handle_word(self, word: str, words: Words) -> Generator[str]:
323323
variants.add(lowercase_word)
324324

325325
# Russian on Kindle must provide a lowercase variant for uppercase-only words (see #2623)
326-
elif is_russian and isinstance(self, DictFileFormatForMobi) and current_word.isupper():
326+
elif is_russian and isinstance(self, MobiFormat) and current_word.isupper():
327327
variants.add(current_word.lower())
328328

329329
yield self.render_word(
@@ -503,18 +503,11 @@ def process(self) -> None:
503503
self.summary(file)
504504

505505

506-
class DictFileFormatForMobi(DictFileFormat):
507-
"""Save the data into a *.df* DictFile."""
508-
509-
output_file = f"altered-{DictFileFormat.output_file}"
510-
511-
512506
class ConverterFromDictFile(DictFileFormat):
513507
target_format = ""
514508
target_suffix = ""
515509
final_file = ""
516510
zip_glob_files = "dict-data.*"
517-
dictfile_format_cls = DictFileFormat
518511
glossary_options: dict[str, str | bool] = {}
519512

520513
def _patch_gc(self) -> None:
@@ -569,10 +562,21 @@ def get_bookname(cls) -> str: # type: ignore[no-untyped-def]
569562
glos.sourceLangName = self.effective_lang_src()
570563
glos.targetLangName = self.effective_lang_dst()
571564

565+
if isinstance(self, MobiFormat):
566+
# Alter the generated word title to fix this Kindling warning:
567+
# [warning R6.1] section 6.1 (p.22): Content is not well-formed XHTML. Kindle requires well-formed HTML documents for reliable conversion. Parse error: ill-formed document: expected `</br>`, but `</idx:orth>` was found (g000002.xhtml)
568+
wordTitleStr_original = glos.wordTitleStr
569+
570+
def wordTitleStr(word: str, **kwargs: str) -> str:
571+
# Do not end with `<br>` but `<br/>`
572+
return str(wordTitleStr_original(word, **kwargs)).replace("<br>", "<br/>")
573+
574+
glos.wordTitleStr = wordTitleStr
575+
572576
self.output_dir_tmp.mkdir()
573577
glos.convert(
574578
ConvertArgs(
575-
inputFilename=str(self.dictionary_file(self.dictfile_format_cls.output_file)),
579+
inputFilename=str(self.dictionary_file(DictFileFormat.output_file)),
576580
outputFilename=str(self.output_dir_tmp / f"dict-data.{self.target_suffix}"),
577581
writeOptions=self.glossary_options,
578582
)
@@ -619,33 +623,16 @@ class DictOrgFormat(ConverterFromDictFile):
619623

620624

621625
class MobiFormat(ConverterFromDictFile):
622-
"""Save the data into a Mobi file.
623-
624-
Incompatibility issues:
625-
626-
1) No support for multiple HTML tags, they will be ignored:
627-
628-
Warning(inputpreprocessor):W29007: Rejected unknown tag: <bdi>
629-
630-
2) Most locales are not fully supported:
631-
632-
Warning(index build):W15008: language not supported. Using default phonetics for spellchecker: english.
633-
634-
3) Greek (EL), and Russian (RU), locales might be incorrectly displayed:
635-
636-
Error(core):E1008: Failed conversion to unicode. The resulting string may contain wrong characters.
637-
638-
"""
626+
"""Save the data into a MobiPocket file."""
639627

640628
target_format = "mobi"
641629
target_suffix = "mobi"
642630
final_file = "dict-{lang_src}-{lang_dst}{etym_suffix}.mobi.zip"
643631
zip_glob_files = "" # Will be set in `_compress()`
644-
dictfile_format_cls = DictFileFormatForMobi
645632
glossary_options = {
646633
"cover_path": str(constants.COVER_FILE),
647634
"keep": True,
648-
"kindlegen_path": str(constants.KINDLEGEN_FILE),
635+
"kindlegen_path": str(constants.MOBIPOCKET_TOOL),
649636
}
650637

651638
def _compress(self) -> Path:
@@ -913,16 +900,18 @@ def summary(self, file: Path) -> None:
913900

914901

915902
PRIMARY_FORMATTERS = {KoboFormat, DictFileFormat, JSONVolumeFormat}
916-
SECONDARY_FORMATTERS = {BZ2DictFileFormat, DictOrgFormat, StarDictFormat}
903+
SECONDARY_FORMATTERS = {BZ2DictFileFormat, DictOrgFormat, MobiFormat, StarDictFormat}
917904
FORMATTERS: dict[str, tuple[type[BaseFormat], type[BaseFormat] | None]] = {
918905
# "format": (primary formatter class, secondary formatter class)
919906
"dictfile": (DictFileFormat, BZ2DictFileFormat),
907+
"dicthtml": (KoboFormat, None),
920908
"dictorg": (DictFileFormat, DictOrgFormat),
921909
"jsonvolume": (JSONVolumeFormat, None),
922-
"dicthtml": (KoboFormat, None),
910+
"mobi": (DictFileFormat, MobiFormat),
923911
"stardict": (DictFileFormat, StarDictFormat),
924912
}
925913
FORMATTERS["df"] = FORMATTERS["dictfile"]
914+
FORMATTERS["kindle"] = FORMATTERS["mobi"]
926915
FORMATTERS["kobo"] = FORMATTERS["dicthtml"]
927916

928917

@@ -935,93 +924,6 @@ def get_secondary_formatters() -> set[type[BaseFormat]]:
935924
return SECONDARY_FORMATTERS
936925

937926

938-
def run_mobi_formatter(
939-
output_dir: Path,
940-
snapshot: str,
941-
locale: str,
942-
words: Words,
943-
variants: Variants,
944-
*,
945-
include_etymology: bool = True,
946-
) -> None:
947-
"""Mobi formatter.
948-
949-
For multiple languages, we need to delete words if the total number of unique unicode characters is greater than 256.
950-
To do this, we delete words using the least-used characters until we meet this condition.
951-
"""
952-
953-
if locale in constants.MOBI_SKIP:
954-
log.info("[Mobi %s] Skipping as the final file size would be > 650 MiB", locale.upper())
955-
return
956-
957-
def all_chars(word: str, details: Word) -> set[str]:
958-
chars = set(word)
959-
if definitions := details.definitions:
960-
if isinstance(definitions, str):
961-
chars.update(definitions)
962-
elif isinstance(definitions, tuple):
963-
chars.update(utils.flatten(definitions))
964-
if etymology := details.etymology:
965-
if isinstance(etymology, str):
966-
chars.update(etymology)
967-
elif isinstance(etymology, tuple):
968-
chars.update(utils.flatten(etymology))
969-
return chars
970-
971-
stats = defaultdict(list)
972-
for word, details in words.copy().items():
973-
if len(word) > 127:
974-
log.info("[Mobi %s] Truncated word too long: %r", locale.upper(), word)
975-
truncated = word[:127]
976-
words[truncated] = words.pop(word)
977-
word = truncated
978-
for char in all_chars(word, details):
979-
stats[char].append(word)
980-
981-
if locale in constants.MOBI_CLEANUP and len(stats) > 256:
982-
new_words = words.copy()
983-
threshold = 1
984-
while len(stats) > 256:
985-
log.info(
986-
"[Mobi %s] Removing words with unique characters count at %d (total is %d)",
987-
locale.upper(),
988-
threshold,
989-
len(stats),
990-
)
991-
for char, related_words in sorted(stats.copy().items(), key=lambda v: (char, len(v[1]))):
992-
if len(related_words) == threshold:
993-
for w in related_words:
994-
new_words.pop(w, None)
995-
stats.pop(char)
996-
if len(stats) <= 256:
997-
break
998-
threshold += 1
999-
1000-
log.info(
1001-
"[Mobi %s] Removed %s words from .mobi (total words count is %s, unique characters count is %d)",
1002-
locale.upper(),
1003-
f"{len(words) - len(new_words):,}",
1004-
f"{len(new_words):,}",
1005-
len(stats),
1006-
)
1007-
words = new_words
1008-
variants = make_variants(words)
1009-
else:
1010-
log.info(
1011-
"[Mobi %s] Untouched words for .mobi (total words count is %s, unique characters count is %d)",
1012-
locale.upper(),
1013-
f"{len(words):,}",
1014-
len(stats),
1015-
)
1016-
1017-
args = (locale, output_dir, words, variants, snapshot)
1018-
run_formatter(DictFileFormatForMobi, *args, include_etymology=include_etymology)
1019-
try:
1020-
run_formatter(MobiFormat, *args, include_etymology=include_etymology)
1021-
except Exception:
1022-
log.exception("[Mobi %s] Error with the Mobi conversion", locale.upper())
1023-
1024-
1025927
def run_formatter(
1026928
cls: type[BaseFormat],
1027929
locale: str,
@@ -1097,33 +999,28 @@ def get_latest_json_file(source_dir: Path) -> Path | None:
1097999
return sorted(files)[-1] if files else None
10981000

10991001

1100-
def get_formatters(formats: str) -> tuple[set[type[BaseFormat]], set[type[BaseFormat]], bool]:
1002+
def get_formatters(formats: str) -> tuple[set[type[BaseFormat]], set[type[BaseFormat]]]:
11011003
primary_formatters: set[type[BaseFormat]] = set()
11021004
secondary_formatters: set[type[BaseFormat]] = set()
1103-
mobi_run = False
11041005
for fmt in (formats or "all").split(","):
11051006
match fmt:
11061007
case _ if fmt in FORMATTERS:
11071008
primary, secondary = FORMATTERS[fmt]
11081009
primary_formatters.add(primary)
11091010
if secondary:
11101011
secondary_formatters.add(secondary)
1111-
case "kindle" | "mobi":
1112-
mobi_run = True
11131012
case "all":
11141013
primary_formatters = get_primary_formatters()
11151014
secondary_formatters = get_secondary_formatters()
1116-
mobi_run = True
11171015
break
11181016
case _:
11191017
print(f"Unknown format: {fmt!r}")
1120-
return primary_formatters, secondary_formatters, mobi_run
1018+
return primary_formatters, secondary_formatters
11211019

11221020

11231021
def convert(
11241022
primary_formatters: set[type[BaseFormat]],
11251023
secondary_formatters: set[type[BaseFormat]],
1126-
mobi_run: bool,
11271024
output_dir: Path,
11281025
snapshot: str,
11291026
locale: str,
@@ -1137,8 +1034,6 @@ def convert(
11371034
for include_etymology in include_etymologies:
11381035
distribute_workload(primary_formatters, *args, include_etymology=include_etymology)
11391036
distribute_workload(secondary_formatters, *args, include_etymology=include_etymology)
1140-
if mobi_run:
1141-
run_mobi_formatter(*args, include_etymology=include_etymology)
11421037

11431038

11441039
def main(locale: str, format: str = "all", with_etym_only: bool = False) -> int:
@@ -1159,12 +1054,11 @@ def main(locale: str, format: str = "all", with_etym_only: bool = False) -> int:
11591054
output_dir = source_dir / "output"
11601055
output_dir.mkdir(exist_ok=True, parents=True)
11611056

1162-
primary_formatters, secondary_formatters, mobi_run = get_formatters(format)
1057+
primary_formatters, secondary_formatters = get_formatters(format)
11631058
start = monotonic()
11641059
convert(
11651060
primary_formatters,
11661061
secondary_formatters,
1167-
mobi_run,
11681062
output_dir,
11691063
input_file.stem.split("-")[-1],
11701064
locale,

0 commit comments

Comments
 (0)