move abx plugins inside vendor dir

2025-06-04 16:53:53 -04:00 · 2024-10-28 04:07:35 -07:00 · 2024-10-28 04:07:35 -07:00 · b3c1cb716e
commit b3c1cb716e
parent 5d9a32c364
242 changed files with 2153 additions and 2700 deletions
--- a/archivebox/vendor/abx-plugin-htmltotext/abx_plugin_htmltotext/init.py
+++ b/archivebox/vendor/abx-plugin-htmltotext/abx_plugin_htmltotext/init.py
@ -0,0 +1,22 @@
+__package__ = 'abx_plugin_htmltotext'
+__label__ = 'HTML-to-Text'
+
+import abx
+
+
+@abx.hookimpl
+def get_CONFIG():
+    from .config import HTMLTOTEXT_CONFIG
+    
+    return {
+        'HTMLTOTEXT_CONFIG': HTMLTOTEXT_CONFIG
+    }
+
+
+# @abx.hookimpl
+# def get_EXTRACTORS():
+#     from .extractors import FAVICON_EXTRACTOR
+    
+#     return {
+#         'htmltotext': FAVICON_EXTRACTOR,
+#     }
--- a/archivebox/vendor/abx-plugin-htmltotext/abx_plugin_htmltotext/config.py
+++ b/archivebox/vendor/abx-plugin-htmltotext/abx_plugin_htmltotext/config.py
@ -0,0 +1,8 @@
+from abx_spec_config.base_configset import BaseConfigSet
+
+
+class HtmltotextConfig(BaseConfigSet):
+    SAVE_HTMLTOTEXT: bool = True
+
+
+HTMLTOTEXT_CONFIG = HtmltotextConfig()
--- a/archivebox/vendor/abx-plugin-htmltotext/abx_plugin_htmltotext/htmltotext.py
+++ b/archivebox/vendor/abx-plugin-htmltotext/abx_plugin_htmltotext/htmltotext.py
@ -0,0 +1,158 @@
+__package__ = 'archivebox.extractors'
+
+from html.parser import HTMLParser
+import io
+from pathlib import Path
+from typing import Optional
+
+from archivebox.config import VERSION
+from archivebox.config.common import ARCHIVING_CONFIG
+from archivebox.misc.system import atomic_write
+from archivebox.misc.util import enforce_types, is_static_file
+
+from archivebox.plugins_extractor.htmltotext.config import HTMLTOTEXT_CONFIG
+
+from ..logging_util import TimedProgress
+from ..index.schema import Link, ArchiveResult, ArchiveError
+from .title import get_html
+
+
+def get_output_path():
+    return "htmltotext.txt"
+
+
+
+class HTMLTextExtractor(HTMLParser):
+    TEXT_ATTRS = [
+        "alt", "cite", "href", "label",
+        "list", "placeholder", "title", "value"
+    ]
+    NOTEXT_TAGS = ["script", "style", "template"]
+    NOTEXT_HREF = ["data:", "javascript:", "#"]
+
+    def __init__(self):
+        super().__init__()
+
+        self.output = io.StringIO()
+        self._tag_stack = []
+
+    def _is_text_attr(self, name, value):
+        if not isinstance(value, str):
+            return False
+        if name == "href" and any(map(lambda p: value.startswith(p), self.NOTEXT_HREF)):
+            return False
+
+        if name in self.TEXT_ATTRS:
+            return True
+
+        return False
+
+    def _parent_tag(self):
+        try:
+            return self._tag_stack[-1]
+        except IndexError:
+            return None
+
+    def _in_notext_tag(self):
+        return any([t in self._tag_stack for t in self.NOTEXT_TAGS])
+
+    def handle_starttag(self, tag, attrs):
+        self._tag_stack.append(tag)
+
+        # Don't write out attribute values if any ancestor
+        # is in NOTEXT_TAGS
+        if self._in_notext_tag():
+            return
+
+        for name, value in attrs:
+            if self._is_text_attr(name, value):
+                self.output.write(f"({value.strip()}) ")
+
+    def handle_endtag(self, tag):
+        orig_stack = self._tag_stack.copy()
+        try:
+            # Keep popping tags until we find the nearest
+            # ancestor matching this end tag
+            while tag != self._tag_stack.pop():
+                pass
+            # Write a space after every tag, to ensure that tokens
+            # in tag text aren't concatenated. This may result in
+            # excess spaces, which should be ignored by search tokenizers.
+            if not self._in_notext_tag() and tag not in self.NOTEXT_TAGS:
+                self.output.write(" ")
+        except IndexError:
+            # Got to the top of the stack, but somehow missed
+            # this end tag -- maybe malformed markup -- restore the
+            # stack
+            self._tag_stack = orig_stack
+
+    def handle_data(self, data):
+        # Don't output text data if any ancestor is in NOTEXT_TAGS
+        if self._in_notext_tag():
+            return
+
+        data = data.lstrip()
+        len_before_rstrip = len(data)
+        data = data.rstrip()
+        spaces_rstripped = len_before_rstrip - len(data)
+        if data:
+            self.output.write(data)
+            if spaces_rstripped:
+                # Add back a single space if 1 or more
+                # whitespace characters were stripped
+                self.output.write(' ')
+
+    def __str__(self):
+        return self.output.getvalue()
+
+
+@enforce_types
+def should_save_htmltotext(link: Link, out_dir: Optional[Path]=None, overwrite: Optional[bool]=False) -> bool:
+    if is_static_file(link.url):
+        return False
+
+    out_dir = out_dir or Path(link.link_dir)
+    if not overwrite and (out_dir / get_output_path()).exists():
+        return False
+
+    return HTMLTOTEXT_CONFIG.SAVE_HTMLTOTEXT
+
+
+@enforce_types
+def save_htmltotext(link: Link, out_dir: Optional[Path]=None, timeout: int=ARCHIVING_CONFIG.TIMEOUT) -> ArchiveResult:
+    """extract search-indexing-friendly text from an HTML document"""
+
+    out_dir = Path(out_dir or link.link_dir)
+    output = get_output_path()
+    cmd = ['(internal) archivebox.extractors.htmltotext', './{singlefile,dom}.html']
+
+    timer = TimedProgress(timeout, prefix='      ')
+    extracted_text = None
+    status = 'failed'
+    try:
+        extractor = HTMLTextExtractor()
+        document = get_html(link, out_dir)
+
+        if not document:
+            raise ArchiveError('htmltotext could not find HTML to parse for article text')
+
+        extractor.feed(document)
+        extractor.close()
+        extracted_text = str(extractor)
+
+        atomic_write(str(out_dir / output), extracted_text)
+        status = 'succeeded'
+    except (Exception, OSError) as err:
+        output = err
+    finally:
+        timer.end()
+
+    return ArchiveResult(
+        cmd=cmd,
+        pwd=str(out_dir),
+        cmd_version=VERSION,
+        output=output,
+        status=status,
+        index_texts=[extracted_text] if extracted_text else [],
+        **timer.stats,  
+    )