Initial commit
This commit is contained in:
Vendored
BIN
Binary file not shown.
@@ -0,0 +1,25 @@
|
||||
from ebooklib import epub
|
||||
from ebooklib import ITEM_DOCUMENT
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
CHAPTER_TAGS = {"h1", "h2", "h3", "h4"}
|
||||
|
||||
def read_epub(path):
|
||||
book = epub.read_epub(path)
|
||||
chapters = []
|
||||
for item in book.get_items():
|
||||
if item.get_type() == ITEM_DOCUMENT:
|
||||
soup = BeautifulSoup(item.get_content(), "html.parser")
|
||||
chapters.append(soup.get_text())
|
||||
return chapters
|
||||
|
||||
def read_epub_flat(path):
|
||||
return "\n\n".join(read_epub(path))
|
||||
|
||||
def read_epub_metadata(path):
|
||||
book = epub.read_epub(path)
|
||||
title = book.get_metadata('DC', 'title')
|
||||
author = book.get_metadata('DC', 'creator')
|
||||
title_str = title[0][0] if title else None
|
||||
author_str = author[0][0] if author else None
|
||||
return title_str, author_str
|
||||
@@ -0,0 +1,62 @@
|
||||
import re
|
||||
|
||||
JUNK_PATTERNS = [
|
||||
r'all rights reserved',
|
||||
r'copyright\s*©',
|
||||
r'isbn[\s\-]',
|
||||
r'published by',
|
||||
r'first published',
|
||||
r'printed in',
|
||||
r'no part of this',
|
||||
r'table of contents',
|
||||
r'this is a work of fiction',
|
||||
r'any resemblance to',
|
||||
]
|
||||
JUNK_RE = re.compile('|'.join(JUNK_PATTERNS), re.IGNORECASE)
|
||||
|
||||
def _is_junk(text):
|
||||
if len(text.split()) < 30:
|
||||
return True
|
||||
if JUNK_RE.search(text):
|
||||
return True
|
||||
return False
|
||||
|
||||
def split_scenes(chapters, min_sentences=3, max_sentences=6):
|
||||
scenes = []
|
||||
for chapter_text in chapters:
|
||||
paragraphs = [p.strip() for p in re.split(r'\n{2,}', chapter_text) if p.strip()]
|
||||
buf = []
|
||||
buf_sentences = 0
|
||||
|
||||
for para in paragraphs:
|
||||
sentences = re.split(r'(?<=[.!?])\s+', para.strip())
|
||||
sentences = [s for s in sentences if s]
|
||||
|
||||
for sentence in sentences:
|
||||
buf.append(sentence)
|
||||
buf_sentences += 1
|
||||
|
||||
if buf_sentences >= max_sentences:
|
||||
candidate = " ".join(buf)
|
||||
if not _is_junk(candidate):
|
||||
scenes.append(candidate)
|
||||
buf = []
|
||||
buf_sentences = 0
|
||||
|
||||
if buf_sentences >= min_sentences:
|
||||
candidate = " ".join(buf)
|
||||
if not _is_junk(candidate):
|
||||
scenes.append(candidate)
|
||||
buf = []
|
||||
buf_sentences = 0
|
||||
|
||||
if buf:
|
||||
candidate = " ".join(buf)
|
||||
if scenes and _is_junk(candidate):
|
||||
pass
|
||||
elif scenes:
|
||||
scenes[-1] = scenes[-1] + " " + candidate
|
||||
elif not _is_junk(candidate):
|
||||
scenes.append(candidate)
|
||||
|
||||
return [s for s in scenes if len(s.strip()) > 20]
|
||||
Reference in New Issue
Block a user