1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85
|
# This file is part of EbookLib.
# Copyright (c) 2013 Aleksandar Erkalovic <aerkalov@gmail.com>
#
# EbookLib is free software: you can redistribute it and/or modify
# it under the terms of the GNU Affero General Public License as published by
# the Free Software Foundation, either version 3 of the License, or
# (at your option) any later version.
#
# EbookLib is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# GNU Affero General Public License for more details.
#
# You should have received a copy of the GNU Affero General Public License
# along with EbookLib. If not, see <http://www.gnu.org/licenses/>.
import subprocess
import six
from ebooklib.plugins.base import BasePlugin
# Recommend usage of
# - https://github.com/w3c/tidy-html5
def tidy_cleanup(content, **extra):
cmd = []
for k, v in six.iteritems(extra):
if v:
cmd.append("--{k}".format(k=k)) # noqa: UP032
cmd.append(v)
else:
cmd.append("-{k}".format(k=k)) # noqa: UP032
# must parse all other extra arguments
try:
p = subprocess.Popen(
["tidy"] + cmd,
shell=False,
stdin=subprocess.PIPE,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
close_fds=True,
)
except OSError:
return (3, None)
p.stdin.write(content)
(cont, p_err) = p.communicate()
# 0 - all ok
# 1 - there were warnings
# 2 - there were errors
# 3 - exception
return (p.returncode, cont)
class TidyPlugin(BasePlugin):
NAME = "Tidy HTML"
OPTIONS = {"char-encoding": "utf8", "tidy-mark": "no"}
def __init__(self, extra=None):
self.options = dict(self.OPTIONS)
if extra is not None:
self.options.update(extra)
def html_before_write(self, book, chapter):
if not chapter.content:
return None
(_, chapter.content) = tidy_cleanup(chapter.content, **self.options)
return chapter.content
def html_after_read(self, book, chapter):
if not chapter.content:
return None
(_, chapter.content) = tidy_cleanup(chapter.content, **self.options)
return chapter.content
|