Add htmltotext extractor

Saves HTML text nodes and selected element attributes in `htmltotext.txt` for each Snapshot. Primarily intended to be used for search indexing.
2025-05-12 22:25:44 -04:00 · 2023-10-23 21:42:25 -04:00 · 2023-10-23 21:42:25 -04:00 · 310b4d1242
commit 310b4d1242
parent 6555719489
9 changed files with 203 additions and 104 deletions
--- a/archivebox/index/html.py
+++ b/archivebox/index/html.py
@ -143,7 +143,7 @@ def snapshot_icons(snapshot) -> str:
            "mercury": "🅼",
            "warc": "📦"
        }
-        exclude = ["favicon", "title", "headers", "archive_org"]
+        exclude = ["favicon", "title", "headers", "htmltotext", "archive_org"]
        # Missing specific entry for WARC

        extractor_outputs = defaultdict(lambda: None)