2020-08-21 18:32:31 +00:00
|
|
|
import os
|
|
|
|
import sqlite3
|
|
|
|
|
2020-07-23 20:07:00 +00:00
|
|
|
from .fixtures import *
|
|
|
|
|
2020-08-21 18:32:31 +00:00
|
|
|
def test_remove_single_page(tmp_path, process, disable_extractors_dict):
|
2020-07-23 20:07:00 +00:00
|
|
|
os.chdir(tmp_path)
|
2020-08-04 13:42:30 +00:00
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'], capture_output=True, env=disable_extractors_dict)
|
2020-08-21 18:32:31 +00:00
|
|
|
remove_process = subprocess.run(['archivebox', 'remove', 'http://127.0.0.1:8080/static/example.com.html', '--yes'], capture_output=True)
|
|
|
|
assert "Found 1 matching URLs to remove" in remove_process.stdout.decode("utf-8")
|
|
|
|
|
|
|
|
conn = sqlite3.connect("index.sqlite3")
|
|
|
|
c = conn.cursor()
|
|
|
|
count = c.execute("SELECT COUNT() from core_snapshot").fetchone()[0]
|
|
|
|
conn.commit()
|
|
|
|
conn.close()
|
|
|
|
|
|
|
|
assert count == 0
|
|
|
|
|
|
|
|
|
|
|
|
def test_remove_single_page_filesystem(tmp_path, process, disable_extractors_dict):
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
assert list((tmp_path / "archive").iterdir()) != []
|
|
|
|
|
|
|
|
subprocess.run(['archivebox', 'remove', 'http://127.0.0.1:8080/static/example.com.html', '--yes', '--delete'], capture_output=True)
|
|
|
|
|
|
|
|
assert list((tmp_path / "archive").iterdir()) == []
|
|
|
|
|
|
|
|
def test_remove_regex(tmp_path, process, disable_extractors_dict):
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/iana.org.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
assert list((tmp_path / "archive").iterdir()) != []
|
|
|
|
|
|
|
|
subprocess.run(['archivebox', 'remove', '--filter-type=regex', '.*', '--yes', '--delete'], capture_output=True)
|
|
|
|
|
|
|
|
assert list((tmp_path / "archive").iterdir()) == []
|
|
|
|
|
|
|
|
def test_remove_exact(tmp_path, process, disable_extractors_dict):
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/iana.org.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
assert list((tmp_path / "archive").iterdir()) != []
|
|
|
|
|
|
|
|
remove_process = subprocess.run(['archivebox', 'remove', '--filter-type=exact', 'http://127.0.0.1:8080/static/iana.org.html', '--yes', '--delete'], capture_output=True)
|
|
|
|
|
|
|
|
assert len(list((tmp_path / "archive").iterdir())) == 1
|
|
|
|
|
|
|
|
def test_remove_substr(tmp_path, process, disable_extractors_dict):
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/iana.org.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
assert list((tmp_path / "archive").iterdir()) != []
|
|
|
|
|
|
|
|
subprocess.run(['archivebox', 'remove', '--filter-type=substring', 'example.com', '--yes', '--delete'], capture_output=True)
|
|
|
|
|
|
|
|
assert len(list((tmp_path / "archive").iterdir())) == 1
|
|
|
|
|
|
|
|
def test_remove_domain(tmp_path, process, disable_extractors_dict):
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/iana.org.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
assert list((tmp_path / "archive").iterdir()) != []
|
|
|
|
|
|
|
|
remove_process = subprocess.run(['archivebox', 'remove', '--filter-type=domain', '127.0.0.1', '--yes', '--delete'], capture_output=True)
|
|
|
|
|
|
|
|
assert len(list((tmp_path / "archive").iterdir())) == 0
|
|
|
|
|
|
|
|
conn = sqlite3.connect("index.sqlite3")
|
|
|
|
c = conn.cursor()
|
|
|
|
count = c.execute("SELECT COUNT() from core_snapshot").fetchone()[0]
|
|
|
|
conn.commit()
|
|
|
|
conn.close()
|
|
|
|
|
2020-08-22 12:34:54 +00:00
|
|
|
assert count == 0
|
|
|
|
|
2020-11-13 19:17:12 +00:00
|
|
|
|
|
|
|
def test_remove_tag(tmp_path, process, disable_extractors_dict):
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/iana.org.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
assert list((tmp_path / "archive").iterdir()) != []
|
|
|
|
|
|
|
|
conn = sqlite3.connect("index.sqlite3")
|
|
|
|
c = conn.cursor()
|
|
|
|
c.execute("INSERT INTO core_tag (id, name, slug) VALUES (2, 'test-tag', 'test-tag')")
|
|
|
|
snapshot_ids = c.execute("SELECT id from core_snapshot")
|
|
|
|
c.executemany('INSERT INTO core_snapshot_tags (snapshot_id, tag_id) VALUES (?, 2)', list(snapshot_ids))
|
|
|
|
conn.commit()
|
|
|
|
|
|
|
|
remove_process = subprocess.run(['archivebox', 'remove', '--filter-type=tag', 'test-tag', '--yes', '--delete'], capture_output=True)
|
|
|
|
|
|
|
|
assert len(list((tmp_path / "archive").iterdir())) == 0
|
|
|
|
|
|
|
|
count = c.execute("SELECT COUNT() from core_snapshot").fetchone()[0]
|
|
|
|
conn.commit()
|
|
|
|
conn.close()
|
|
|
|
|
|
|
|
assert count == 0
|
|
|
|
|
2020-08-22 12:34:54 +00:00
|
|
|
def test_remove_before(tmp_path, process, disable_extractors_dict):
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/iana.org.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
assert list((tmp_path / "archive").iterdir()) != []
|
|
|
|
|
|
|
|
conn = sqlite3.connect("index.sqlite3")
|
|
|
|
c = conn.cursor()
|
2021-02-18 11:21:44 +00:00
|
|
|
higherts, lowerts = timestamp = c.execute("SELECT timestamp FROM core_snapshot ORDER BY timestamp DESC").fetchall()
|
2020-08-22 12:34:54 +00:00
|
|
|
conn.commit()
|
|
|
|
conn.close()
|
|
|
|
|
2021-03-31 03:39:11 +00:00
|
|
|
lowerts = lowerts[0]
|
|
|
|
higherts = higherts[0]
|
2020-08-22 12:34:54 +00:00
|
|
|
|
2021-02-18 11:21:44 +00:00
|
|
|
# before is less than, so only the lower snapshot gets deleted
|
|
|
|
subprocess.run(['archivebox', 'remove', '--filter-type=regex', '.*', '--yes', '--delete', '--before', higherts], capture_output=True)
|
2020-08-22 12:34:54 +00:00
|
|
|
|
2021-02-18 11:21:44 +00:00
|
|
|
assert not (tmp_path / "archive" / lowerts).exists()
|
|
|
|
assert (tmp_path / "archive" / higherts).exists()
|
2020-08-22 12:34:54 +00:00
|
|
|
|
|
|
|
def test_remove_after(tmp_path, process, disable_extractors_dict):
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/example.com.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
subprocess.run(['archivebox', 'add', 'http://127.0.0.1:8080/static/iana.org.html'], capture_output=True, env=disable_extractors_dict)
|
|
|
|
assert list((tmp_path / "archive").iterdir()) != []
|
|
|
|
|
|
|
|
conn = sqlite3.connect("index.sqlite3")
|
|
|
|
c = conn.cursor()
|
2021-02-18 11:21:44 +00:00
|
|
|
higherts, lowerts = c.execute("SELECT timestamp FROM core_snapshot ORDER BY timestamp DESC").fetchall()
|
2020-08-22 12:34:54 +00:00
|
|
|
conn.commit()
|
|
|
|
conn.close()
|
|
|
|
|
2021-02-18 11:21:44 +00:00
|
|
|
lowerts = lowerts[0].split(".")[0]
|
|
|
|
higherts = higherts[0].split(".")[0]
|
2020-08-22 12:34:54 +00:00
|
|
|
|
2021-02-18 11:21:44 +00:00
|
|
|
# after is greater than or equal to, so both snapshots get deleted
|
|
|
|
subprocess.run(['archivebox', 'remove', '--filter-type=regex', '.*', '--yes', '--delete', '--after', lowerts], capture_output=True)
|
2020-08-22 12:34:54 +00:00
|
|
|
|
2021-02-18 11:21:44 +00:00
|
|
|
assert not (tmp_path / "archive" / lowerts).exists()
|
|
|
|
assert not (tmp_path / "archive" / higherts).exists()
|