Cleanup formatting in tests

This commit is contained in:
AntiCompositeNumber
2020-07-21 21:05:20 -04:00
parent 4774530588
commit 2c4a7479ce

View File

@@ -1,18 +1,21 @@
# -*- coding: utf-8 -*- # -*- coding: utf-8 -*-
import sys import sys
sys.path.append("..")
import waybackpy
import pytest import pytest
import random import random
import time import time
sys.path.append("..")
import waybackpy # noqa: E402
if sys.version_info >= (3, 0): # If the python ver >= 3 if sys.version_info >= (3, 0): # If the python ver >= 3
from urllib.request import Request, urlopen from urllib.request import Request, urlopen
from urllib.error import URLError from urllib.error import URLError
else: # For python2.x else: # For python2.x
from urllib2 import Request, urlopen, URLError from urllib2 import Request, urlopen, URLError
user_agent = "Mozilla/5.0 (Windows NT 6.2; rv:20.0) Gecko/20121202 Firefox/20.0" user_agent = "Mozilla/5.0 (Windows NT 6.2; rv:20.0) Gecko/20121202 Firefox/20.0"
def test_clean_url(): def test_clean_url():
test_url = " https://en.wikipedia.org/wiki/Network security " test_url = " https://en.wikipedia.org/wiki/Network security "
answer = "https://en.wikipedia.org/wiki/Network_security" answer = "https://en.wikipedia.org/wiki/Network_security"
@@ -20,11 +23,13 @@ def test_clean_url():
test_result = target.clean_url() test_result = target.clean_url()
assert answer == test_result assert answer == test_result
def test_url_check(): def test_url_check():
broken_url = "http://wwwgooglecom/" broken_url = "http://wwwgooglecom/"
with pytest.raises(Exception) as e_info: with pytest.raises(Exception):
waybackpy.Url(broken_url, user_agent) waybackpy.Url(broken_url, user_agent)
def test_save(): def test_save():
# Test for urls that exist and can be archived. # Test for urls that exist and can be archived.
time.sleep(10) time.sleep(10)
@@ -35,89 +40,139 @@ def test_save():
"commons.wikimedia.org", "commons.wikimedia.org",
"www.wiktionary.org", "www.wiktionary.org",
"www.w3schools.com", "www.w3schools.com",
"www.youtube.com" "www.youtube.com",
] ]
x = random.randint(0, len(url_list)-1) x = random.randint(0, len(url_list) - 1)
url1 = url_list[x] url1 = url_list[x]
target = waybackpy.Url(url1, "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_9_2) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/36.0.1944.0 Safari/537.36") target = waybackpy.Url(
url1,
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_9_2) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/36.0.1944.0 Safari/537.36",
)
archived_url1 = target.save() archived_url1 = target.save()
assert url1 in archived_url1 assert url1 in archived_url1
if sys.version_info > (3, 6): if sys.version_info > (3, 6):
# Test for urls that are incorrect. # Test for urls that are incorrect.
with pytest.raises(Exception) as e_info: with pytest.raises(Exception):
url2 = "ha ha ha ha" url2 = "ha ha ha ha"
waybackpy.Url(url2, user_agent) waybackpy.Url(url2, user_agent)
time.sleep(5) time.sleep(5)
# Test for urls not allowed to archive by robot.txt. # Test for urls not allowed to archive by robot.txt.
with pytest.raises(Exception) as e_info: with pytest.raises(Exception):
url3 = "http://www.archive.is/faq.html" url3 = "http://www.archive.is/faq.html"
target = waybackpy.Url(url3, "Mozilla/5.0 (Macintosh; Intel Mac OS X 10.6; rv:25.0) Gecko/20100101 Firefox/25.0") target = waybackpy.Url(
url3,
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10.6; rv:25.0) "
"Gecko/20100101 Firefox/25.0",
)
target.save() target.save()
time.sleep(5) time.sleep(5)
# Non existent urls, test # Non existent urls, test
with pytest.raises(Exception) as e_info: with pytest.raises(Exception):
url4 = "https://githfgdhshajagjstgeths537agajaajgsagudadhuss8762346887adsiugujsdgahub.us" url4 = (
target = waybackpy.Url(url3, "Mozilla/5.0 (Windows; U; Windows NT 6.0; en-US) AppleWebKit/533.20.25 (KHTML, like Gecko) Version/5.0.4 Safari/533.20.27") "https://githfgdhshajagjstgeths537agajaajgsagudadhuss87623"
"46887adsiugujsdgahub.us"
)
target = waybackpy.Url(
url3,
"Mozilla/5.0 (Windows; U; Windows NT 6.0; en-US) "
"AppleWebKit/533.20.25 (KHTML, like Gecko) Version/5.0.4 "
"Safari/533.20.27",
)
target.save() target.save()
else: else:
pass pass
def test_near(): def test_near():
time.sleep(10) time.sleep(10)
url = "google.com" url = "google.com"
target = waybackpy.Url(url, "Mozilla/5.0 (Windows; U; Windows NT 6.0; de-DE) AppleWebKit/533.20.25 (KHTML, like Gecko) Version/5.0.3 Safari/533.19.4") target = waybackpy.Url(
url,
"Mozilla/5.0 (Windows; U; Windows NT 6.0; de-DE) AppleWebKit/533.20.25 "
"(KHTML, like Gecko) Version/5.0.3 Safari/533.19.4",
)
archive_near_year = target.near(year=2010) archive_near_year = target.near(year=2010)
assert "2010" in archive_near_year assert "2010" in archive_near_year
if sys.version_info > (3, 6): if sys.version_info > (3, 6):
time.sleep(5) time.sleep(5)
archive_near_month_year = target.near( year=2015, month=2) archive_near_month_year = target.near(year=2015, month=2)
assert ("201502" in archive_near_month_year) or ("201501" in archive_near_month_year) or ("201503" in archive_near_month_year) assert (
("201502" in archive_near_month_year)
or ("201501" in archive_near_month_year)
or ("201503" in archive_near_month_year)
)
target = waybackpy.Url("www.python.org", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/42.0.2311.135 Safari/537.36 Edge/12.246") target = waybackpy.Url(
archive_near_hour_day_month_year = target.near(year=2008, month=5, day=9, hour=15) "www.python.org",
assert ("2008050915" in archive_near_hour_day_month_year) or ("2008050914" in archive_near_hour_day_month_year) or ("2008050913" in archive_near_hour_day_month_year) "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/42.0.2311.135 Safari/537.36 Edge/12.246",
)
archive_near_hour_day_month_year = target.near(
year=2008, month=5, day=9, hour=15
)
assert (
("2008050915" in archive_near_hour_day_month_year)
or ("2008050914" in archive_near_hour_day_month_year)
or ("2008050913" in archive_near_hour_day_month_year)
)
with pytest.raises(Exception) as e_info: with pytest.raises(Exception):
NeverArchivedUrl = "https://ee_3n.wrihkeipef4edia.org/rwti5r_ki/Nertr6w_rork_rse7c_urity" NeverArchivedUrl = (
"https://ee_3n.wrihkeipef4edia.org/rwti5r_ki/Nertr6w_rork_rse7c_urity"
)
target = waybackpy.Url(NeverArchivedUrl, user_agent) target = waybackpy.Url(NeverArchivedUrl, user_agent)
target.near(year=2010) target.near(year=2010)
else: else:
pass pass
def test_oldest(): def test_oldest():
url = "github.com/akamhy/waybackpy" url = "github.com/akamhy/waybackpy"
target = waybackpy.Url(url, user_agent) target = waybackpy.Url(url, user_agent)
assert "20200504141153" in target.oldest() assert "20200504141153" in target.oldest()
def test_newest(): def test_newest():
url = "github.com/akamhy/waybackpy" url = "github.com/akamhy/waybackpy"
target = waybackpy.Url(url, user_agent) target = waybackpy.Url(url, user_agent)
assert url in target.newest() assert url in target.newest()
def test_get(): def test_get():
target = waybackpy.Url("google.com", user_agent) target = waybackpy.Url("google.com", user_agent)
assert "Welcome to Google" in target.get(target.oldest()) assert "Welcome to Google" in target.get(target.oldest())
def test_wayback_timestamp(): def test_wayback_timestamp():
ts = waybackpy.Url("https://www.google.com","UA").wayback_timestamp(year=2020,month=1,day=2,hour=3,minute=4) ts = waybackpy.Url("https://www.google.com", "UA").wayback_timestamp(
year=2020, month=1, day=2, hour=3, minute=4
)
assert "202001020304" in str(ts) assert "202001020304" in str(ts)
def test_get_response(): def test_get_response():
hdr = { 'User-Agent' : 'Mozilla/5.0 (X11; Ubuntu; Linux x86_64; rv:78.0) Gecko/20100101 Firefox/78.0'} hdr = {
req = Request("https://www.google.com", headers=hdr) # nosec "User-Agent": "Mozilla/5.0 (X11; Ubuntu; Linux x86_64; rv:78.0) "
response = waybackpy.Url("https://www.google.com","UA").get_response(req) "Gecko/20100101 Firefox/78.0"
}
req = Request("https://www.google.com", headers=hdr) # nosec
response = waybackpy.Url("https://www.google.com", "UA").get_response(req)
assert response.code == 200 assert response.code == 200
def test_total_archives(): def test_total_archives():
if sys.version_info > (3, 6): if sys.version_info > (3, 6):
target = waybackpy.Url(" https://google.com ", user_agent) target = waybackpy.Url(" https://google.com ", user_agent)
assert target.total_archives() > 500000 assert target.total_archives() > 500000
else: else:
pass pass
target = waybackpy.Url(" https://gaha.e4i3n.m5iai3kip6ied.cima/gahh2718gs/ahkst63t7gad8 ", user_agent) target = waybackpy.Url(
" https://gaha.e4i3n.m5iai3kip6ied.cima/gahh2718gs/ahkst63t7gad8 ", user_agent
)
assert target.total_archives() == 0 assert target.total_archives() == 0