ArchiveBox/tests/test_extractors.py

from .fixtures import *
import json as pyjson
from archivebox.extractors import ignore_methods, get_default_archive_methods, should_save_title

def test_wget_broken_pipe(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"USE_WGET": "true"})
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],
                                 capture_output=True, env=disable_extractors_dict)
    assert "TypeError chmod_file(..., path: str) got unexpected NoneType argument path=None" not in add_process.stdout.decode("utf-8")

def test_ignore_methods():
    """
    Takes the passed method out of the default methods list and returns that value
    """
    ignored = ignore_methods(['title'])
    assert "title" not in ignored

def test_save_allowdenylist_works(tmp_path, process, disable_extractors_dict):
    allow_list = {
        r'/static': ["headers", "singlefile"],
        r'example\.com\.html$': ["headers"],
    }
    deny_list = {
        "/static": ["singlefile"],
    }
    disable_extractors_dict.update({
        "SAVE_HEADERS": "true",
        "USE_SINGLEFILE": "true",
        "SAVE_ALLOWLIST": pyjson.dumps(allow_list),
        "SAVE_DENYLIST": pyjson.dumps(deny_list),
    })
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict) 
    archived_item_path = list(tmp_path.glob('archive/**/*'))[0]
    singlefile_file = archived_item_path / "singlefile.html"
    assert not singlefile_file.exists()
    headers_file = archived_item_path / "headers.json"
    assert headers_file.exists()

def test_save_denylist_works(tmp_path, process, disable_extractors_dict):
    deny_list = {
        "/static": ["singlefile"],
    }
    disable_extractors_dict.update({
        "SAVE_HEADERS": "true",
        "USE_SINGLEFILE": "true",
        "SAVE_DENYLIST": pyjson.dumps(deny_list),
    })
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict) 
    archived_item_path = list(tmp_path.glob('archive/**/*'))[0]
    singlefile_file = archived_item_path / "singlefile.html"
    assert not singlefile_file.exists()
    headers_file = archived_item_path / "headers.json"
    assert headers_file.exists()

def test_singlefile_works(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"USE_SINGLEFILE": "true"})
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    archived_item_path = list(tmp_path.glob('archive/**/*'))[0]
    output_file = archived_item_path / "singlefile.html" 
    assert output_file.exists()

def test_readability_works(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"USE_READABILITY": "true"})
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    archived_item_path = list(tmp_path.glob("archive/**/*"))[0]
    output_file = archived_item_path / "readability" / "content.html"
    assert output_file.exists()

def test_mercury_works(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"USE_MERCURY": "true"})
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    archived_item_path = list(tmp_path.glob("archive/**/*"))[0]
    output_file = archived_item_path / "mercury" / "content.html"
    assert output_file.exists()

def test_htmltotext_works(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"SAVE_HTMLTOTEXT": "true"})
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    archived_item_path = list(tmp_path.glob("archive/**/*"))[0]
    output_file = archived_item_path / "htmltotext.txt"
    assert output_file.exists()

def test_readability_works_with_wget(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"USE_READABILITY": "true", "USE_WGET": "true"})
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    archived_item_path = list(tmp_path.glob("archive/**/*"))[0]
    output_file = archived_item_path / "readability" / "content.html"
    assert output_file.exists()

def test_readability_works_with_singlefile(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"USE_READABILITY": "true", "USE_SINGLEFILE": "true"})
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    archived_item_path = list(tmp_path.glob("archive/**/*"))[0]
    output_file = archived_item_path / "readability" / "content.html"
    assert output_file.exists()

def test_readability_works_with_dom(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"USE_READABILITY": "true", "SAVE_DOM": "true"})
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    archived_item_path = list(tmp_path.glob("archive/**/*"))[0]
    output_file = archived_item_path / "readability" / "content.html"
    assert output_file.exists()

def test_use_node_false_disables_readability_and_singlefile(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"USE_READABILITY": "true", "SAVE_DOM": "true", "USE_SINGLEFILE": "true", "USE_NODE": "false"}) 
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    output_str = add_process.stdout.decode("utf-8")
    assert "> singlefile" not in output_str
    assert "> readability" not in output_str

def test_headers_ignored(tmp_path, process, disable_extractors_dict):
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/headers/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    archived_item_path = list(tmp_path.glob("archive/**/*"))[0]
    output_file = archived_item_path / "headers.json"
    assert not output_file.exists()

def test_headers_retrieved(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"SAVE_HEADERS": "true"})
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/headers/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    archived_item_path = list(tmp_path.glob("archive/**/*"))[0]
    output_file = archived_item_path / "headers.json"
    assert output_file.exists()
    headers_file = archived_item_path / 'headers.json'
    with open(headers_file, 'r', encoding='utf-8') as f:
        headers = pyjson.load(f)
    assert headers['Content-Language'] == 'en'
    assert headers['Content-Script-Type'] == 'text/javascript'
    assert headers['Content-Style-Type'] == 'text/css'

def test_headers_redirect_chain(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"SAVE_HEADERS": "true"})
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/redirect/headers/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    archived_item_path = list(tmp_path.glob("archive/**/*"))[0]
    output_file = archived_item_path / "headers.json" 
    with open(output_file, 'r', encoding='utf-8') as f:
        headers = pyjson.load(f)
    assert headers['Content-Language'] == 'en'
    assert headers['Content-Script-Type'] == 'text/javascript'
    assert headers['Content-Style-Type'] == 'text/css'

def test_headers_400_plus(tmp_path, process, disable_extractors_dict):
    disable_extractors_dict.update({"SAVE_HEADERS": "true"})
    add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/400/example.com.html'],
                                  capture_output=True, env=disable_extractors_dict)
    archived_item_path = list(tmp_path.glob("archive/**/*"))[0]
    output_file = archived_item_path / "headers.json" 
    with open(output_file, 'r', encoding='utf-8') as f:
        headers = pyjson.load(f)
    assert headers["Status-Code"] == "200"
fix: Add change to calculate wget folder when there is a port present 2020-07-17 21:55:56 +00:00			`from .fixtures import *`
Added more asserts 2020-09-14 21:35:45 +00:00			`import json as pyjson`
fix: Remove title from extractors for oneshot 2020-07-31 15:24:58 +00:00			`from archivebox.extractors import ignore_methods, get_default_archive_methods, should_save_title`
fix: Add change to calculate wget folder when there is a port present 2020-07-17 21:55:56 +00:00
tests: Add mechanism to avoid using extractors that we are not testing 2020-08-04 13:42:30 +00:00			`def test_wget_broken_pipe(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"USE_WGET": "true"})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
fix: Remove title from extractors for oneshot 2020-07-31 15:24:58 +00:00			`assert "TypeError chmod_file(..., path: str) got unexpected NoneType argument path=None" not in add_process.stdout.decode("utf-8")`

			`def test_ignore_methods():`
			`"""`
			`Takes the passed method out of the default methods list and returns that value`
			`"""`
			`ignored = ignore_methods(['title'])`
Add URL-specific method allow/deny lists Allows enabling only allow-listed extractors or disabling specific deny-listed extractors for a regular expression matched against an added site's URL. 2023-07-31 15:34:03 +00:00			`assert "title" not in ignored`

			`def test_save_allowdenylist_works(tmp_path, process, disable_extractors_dict):`
			`allow_list = {`
			`r'/static': ["headers", "singlefile"],`
			`r'example\.com\.html$': ["headers"],`
			`}`
			`deny_list = {`
			`"/static": ["singlefile"],`
			`}`
			`disable_extractors_dict.update({`
			`"SAVE_HEADERS": "true",`
			`"USE_SINGLEFILE": "true",`
			`"SAVE_ALLOWLIST": pyjson.dumps(allow_list),`
			`"SAVE_DENYLIST": pyjson.dumps(deny_list),`
			`})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob('archive/*/'))[0]`
			`singlefile_file = archived_item_path / "singlefile.html"`
			`assert not singlefile_file.exists()`
			`headers_file = archived_item_path / "headers.json"`
			`assert headers_file.exists()`

			`def test_save_denylist_works(tmp_path, process, disable_extractors_dict):`
			`deny_list = {`
			`"/static": ["singlefile"],`
			`}`
			`disable_extractors_dict.update({`
			`"SAVE_HEADERS": "true",`
			`"USE_SINGLEFILE": "true",`
			`"SAVE_DENYLIST": pyjson.dumps(deny_list),`
			`})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob('archive/*/'))[0]`
			`singlefile_file = archived_item_path / "singlefile.html"`
			`assert not singlefile_file.exists()`
			`headers_file = archived_item_path / "headers.json"`
			`assert headers_file.exists()`
tests: Add basic singlefile test 2020-07-31 19:49:54 +00:00
tests: Add mechanism to avoid using extractors that we are not testing 2020-08-04 13:42:30 +00:00			`def test_singlefile_works(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"USE_SINGLEFILE": "true"})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],`
Add URL-specific method allow/deny lists Allows enabling only allow-listed extractors or disabling specific deny-listed extractors for a regular expression matched against an added site's URL. 2023-07-31 15:34:03 +00:00			`capture_output=True, env=disable_extractors_dict)`
tests: Add basic singlefile test 2020-07-31 19:49:54 +00:00			`archived_item_path = list(tmp_path.glob('archive/*/'))[0]`
make filenames consistent with program name 2020-08-01 15:59:07 +00:00			`output_file = archived_item_path / "singlefile.html"`
tests: Add basic singlefile test 2020-07-31 19:49:54 +00:00			`assert output_file.exists()`
tests: Add readability tests 2020-08-11 16:15:15 +00:00
			`def test_readability_works(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"USE_READABILITY": "true"})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob("archive/*/"))[0]`
			`output_file = archived_item_path / "readability" / "content.html"`
			`assert output_file.exists()`

tests: add test for mercury-parser 2020-09-22 08:47:43 +00:00			`def test_mercury_works(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"USE_MERCURY": "true"})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob("archive/*/"))[0]`
			`output_file = archived_item_path / "mercury" / "content.html"`
			`assert output_file.exists()`

Add htmltotext extractor Saves HTML text nodes and selected element attributes in `htmltotext.txt` for each Snapshot. Primarily intended to be used for search indexing. 2023-10-24 01:42:25 +00:00			`def test_htmltotext_works(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"SAVE_HTMLTOTEXT": "true"})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob("archive/*/"))[0]`
			`output_file = archived_item_path / "htmltotext.txt"`
			`assert output_file.exists()`

tests: Add readability tests 2020-08-11 16:15:15 +00:00			`def test_readability_works_with_wget(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"USE_READABILITY": "true", "USE_WGET": "true"})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob("archive/*/"))[0]`
			`output_file = archived_item_path / "readability" / "content.html"`
			`assert output_file.exists()`

			`def test_readability_works_with_singlefile(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"USE_READABILITY": "true", "USE_SINGLEFILE": "true"})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob("archive/*/"))[0]`
			`output_file = archived_item_path / "readability" / "content.html"`
			`assert output_file.exists()`

			`def test_readability_works_with_dom(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"USE_READABILITY": "true", "SAVE_DOM": "true"})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob("archive/*/"))[0]`
			`output_file = archived_item_path / "readability" / "content.html"`
			`assert output_file.exists()`
feat: Add options to ease management of node related extractors 2020-08-18 15:34:28 +00:00
			`def test_use_node_false_disables_readability_and_singlefile(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"USE_READABILITY": "true", "SAVE_DOM": "true", "USE_SINGLEFILE": "true", "USE_NODE": "false"})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`output_str = add_process.stdout.decode("utf-8")`
			`assert "> singlefile" not in output_str`
Added test headers extractor 2020-09-11 19:18:36 +00:00			`assert "> readability" not in output_str`

fix: Improve headers handling 2020-09-24 13:37:27 +00:00			`def test_headers_ignored(tmp_path, process, disable_extractors_dict):`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/headers/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob("archive/*/"))[0]`
			`output_file = archived_item_path / "headers.json"`
			`assert not output_file.exists()`

			`def test_headers_retrieved(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"SAVE_HEADERS": "true"})`
Added more asserts 2020-09-14 21:35:45 +00:00			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/headers/example.com.html'],`
Added test headers extractor 2020-09-11 19:18:36 +00:00			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob("archive/*/"))[0]`
			`output_file = archived_item_path / "headers.json"`
Added more asserts 2020-09-14 21:35:45 +00:00			`assert output_file.exists()`
			`headers_file = archived_item_path / 'headers.json'`
enforce utf8 on literally all file operations because windows sucks 2021-03-27 05:01:29 +00:00			`with open(headers_file, 'r', encoding='utf-8') as f:`
Added more asserts 2020-09-14 21:35:45 +00:00			`headers = pyjson.load(f)`
			`assert headers['Content-Language'] == 'en'`
			`assert headers['Content-Script-Type'] == 'text/javascript'`
			`assert headers['Content-Style-Type'] == 'text/css'`
fix: Improve headers handling 2020-09-24 13:37:27 +00:00
			`def test_headers_redirect_chain(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"SAVE_HEADERS": "true"})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/redirect/headers/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob("archive/*/"))[0]`
			`output_file = archived_item_path / "headers.json"`
enforce utf8 on literally all file operations because windows sucks 2021-03-27 05:01:29 +00:00			`with open(output_file, 'r', encoding='utf-8') as f:`
fix: Improve headers handling 2020-09-24 13:37:27 +00:00			`headers = pyjson.load(f)`
			`assert headers['Content-Language'] == 'en'`
			`assert headers['Content-Script-Type'] == 'text/javascript'`
			`assert headers['Content-Style-Type'] == 'text/css'`

			`def test_headers_400_plus(tmp_path, process, disable_extractors_dict):`
			`disable_extractors_dict.update({"SAVE_HEADERS": "true"})`
			`add_process = subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/400/example.com.html'],`
			`capture_output=True, env=disable_extractors_dict)`
			`archived_item_path = list(tmp_path.glob("archive/*/"))[0]`
			`output_file = archived_item_path / "headers.json"`
enforce utf8 on literally all file operations because windows sucks 2021-03-27 05:01:29 +00:00			`with open(output_file, 'r', encoding='utf-8') as f:`
fix: Improve headers handling 2020-09-24 13:37:27 +00:00			`headers = pyjson.load(f)`
enforce utf8 on literally all file operations because windows sucks 2021-03-27 05:01:29 +00:00			`assert headers["Status-Code"] == "200"`