Some cleanups and fixes.
This commit is contained in:
parent
676d73181e
commit
01633ed147
4 changed files with 19 additions and 30 deletions
32
README.md
32
README.md
|
|
@ -2,9 +2,9 @@
|
|||
|
||||
A tool to convert all installed man pages to a simple ebook or other formats with contents in the head.
|
||||
|
||||
Maybe this could be only a small shell script but if not I will use Python.
|
||||
~~Maybe this could be only a small shell script but if not I will use Python.~~ - It is Python now... :)
|
||||
|
||||
## Why man pages as ebook?
|
||||
## Why man pages as book?
|
||||
|
||||
Because I want to read them comfortable on my ebook reader when traveling but all man pages in one book.
|
||||
Not more... ;)
|
||||
|
|
@ -15,23 +15,23 @@ Not more... ;)
|
|||
https://pypi.org/project/pypandoc/ or https://pypi.org/project/pandoc/.
|
||||
(very mature features for document conversion like direct manpage input and epub output,
|
||||
but I don't know much about pandoc, so I need to discover this.)
|
||||
- https://www.gnu.org/software/groff/
|
||||
(man -Thtml df > df.html)
|
||||
- https://linux.die.net/man/1/man2html
|
||||
(is nice because it must not generate html, head and body)
|
||||
- https://pypi.org/project/epubmaker/ - Python
|
||||
(maybe nice but unmaintained)
|
||||
- https://docutils.sourceforge.io/ - Python
|
||||
(as I can see now it only converts from .rst files.)
|
||||
- https://github.com/hanez/aov-html2epub - Bash
|
||||
- ~~https://www.gnu.org/software/groff/
|
||||
(man -Thtml df > df.html)~~
|
||||
- ~~https://linux.die.net/man/1/man2html
|
||||
(is nice because it must not generate html, head and body)~~
|
||||
- ~~https://pypi.org/project/epubmaker/ - Python
|
||||
(maybe nice but unmaintained)~~
|
||||
- ~~https://docutils.sourceforge.io/ - Python
|
||||
(as I can see now it only converts from .rst files.)~~
|
||||
- ~~https://github.com/hanez/aov-html2epub - Bash~~
|
||||
|
||||
## Workflow
|
||||
|
||||
1. Read manpage sections
|
||||
2. Create HTML file for each manpage in each section
|
||||
3. Read and store title(metadata) for in each file
|
||||
4. Remove all unneeded stuff like html, head and body in each file
|
||||
5. Create TOC for the entire ebook
|
||||
1. ~~Read manpage sections~~
|
||||
2. ~~Create HTML file for each manpage in each section~~
|
||||
3. ~~Read and store title(metadata) for in each file~~
|
||||
4. ~~Remove all unneeded stuff like html, head and body in each file~~
|
||||
5. ~~Create TOC for the entire ebook~~
|
||||
6. Create a header HTML file for the ebook containing the head tag
|
||||
7. Merge toc and content of all files for preparing the ebook
|
||||
8. Merge the header file with the prepared ebook while inserting html and body tags which should result in a valid HTML file
|
||||
|
|
|
|||
13
man2book
13
man2book
|
|
@ -6,21 +6,15 @@ import re
|
|||
from bs4 import BeautifulSoup
|
||||
|
||||
__author__ = 'Johannes Findeisen <you@hanez.org>'
|
||||
__version__ = '0.0.3'
|
||||
__version__ = '0.0.4'
|
||||
|
||||
DIR = '/home/hanez/code/man2book/'
|
||||
# the limit is just for development to limit the number of man pages in each section. set to 0 to
|
||||
# have no limit.
|
||||
LIMIT = 0
|
||||
MANPAGE_PATH = '/usr/share/man/'
|
||||
OUTPUT_DIR = DIR + 'output/'
|
||||
TMP_DIR = '/tmp/man2book/'
|
||||
|
||||
sections = ['1', '2', '3', '4', '5', '6', '7', '8']
|
||||
# the toc will be a nested dictionary: https://www.geeksforgeeks.org/python-nested-dictionary/
|
||||
# toc = {'section1': {'title': 'title', 'anchor': 'anchor'},
|
||||
# 'section1': {'title': 'title', 'anchor': 'anchor'},
|
||||
# 'section2': {'title': 'title', 'anchor': 'anchor'}}
|
||||
toc = {}
|
||||
|
||||
parser = argparse.ArgumentParser(
|
||||
|
|
@ -29,9 +23,6 @@ parser = argparse.ArgumentParser(
|
|||
epilog='author: ' + __author__,
|
||||
prog='man2book')
|
||||
|
||||
# if using this feature there must be a way to set the section for each manpage since the name
|
||||
# can be used in more than one section. maybe make it optional like df:1 or so. so this feature
|
||||
# is not reliable at the moment.
|
||||
parser.add_argument('-m', '--manpages', metavar='MANPAGES', help='limit only to a subset of '
|
||||
'man pages. e.g. cd or a list '
|
||||
'like cd,df,mv. this feature '
|
||||
|
|
@ -90,7 +81,7 @@ for section in sections:
|
|||
os.system('rm -rf ' + TMP_DIR + manpage_file)
|
||||
pass
|
||||
os.system('pandoc --from man --to html < ' + TMP_DIR + manpage +
|
||||
' > ' + TMP_DIR + manpage + '.html')
|
||||
' > ' + TMP_DIR + section + '.' + os.path.splitext(manpage)[0] + '.html')
|
||||
|
||||
os.system('rm -rf ' + TMP_DIR + manpage)
|
||||
|
||||
|
|
|
|||
|
|
@ -1,3 +1 @@
|
|||
beautifulsoup4~=4.11.1
|
||||
lxml~=4.9.1
|
||||
pandocfilters~=1.5.0
|
||||
beautifulsoup4~=4.11.1
|
||||
Loading…
Add table
Add a link
Reference in a new issue