32 lines
1.6 KiB
Python
32 lines
1.6 KiB
Python
"""Download only public assets referenced by the existing homepage."""
|
|
from pathlib import Path
|
|
from urllib.request import build_opener,ProxyHandler,HTTPSHandler,Request
|
|
from urllib.parse import urlsplit
|
|
import ssl,re,json
|
|
ROOT=Path(__file__).resolve().parent
|
|
BASE='https://servaki.online'
|
|
html=(ROOT/'reference/original.html').read_text(encoding='utf-8')
|
|
normal=build_opener(ProxyHandler({}))
|
|
try:
|
|
normal.open(BASE+'/',timeout=20).read(1)
|
|
print('TLS validation: OK')
|
|
except Exception as e:
|
|
print('TLS validation failed:',type(e).__name__)
|
|
# The original certificate was reported expired. This read-only fallback is
|
|
# restricted to public site resources; it never carries authentication data.
|
|
opener=build_opener(ProxyHandler({}),HTTPSHandler(context=ssl._create_unverified_context()))
|
|
assets=sorted(set(re.findall(r'(?:src|href)="(/[^"?#]+\.(?:png|jpg|jpeg|webp|svg|ico))"',html)))
|
|
manifest=[]
|
|
for rel in assets:
|
|
data=opener.open(BASE+rel,timeout=30).read()
|
|
dest=ROOT/'public'/rel.lstrip('/');dest.parent.mkdir(parents=True,exist_ok=True);dest.write_bytes(data)
|
|
manifest.append({'path':rel,'bytes':len(data)})
|
|
for name in ['robots.txt','sitemap.xml']:
|
|
try:
|
|
response=opener.open(BASE+'/'+name,timeout=20);data=response.read()
|
|
if b'<html' not in data.lower():(ROOT/'reference'/name).write_bytes(data)
|
|
print(name,response.status,'reference saved' if b'<html' not in data.lower() else 'HTML fallback; not used')
|
|
except Exception as e:print(name,type(e).__name__)
|
|
(ROOT/'reference/public-assets.json').write_text(json.dumps(manifest,indent=2),encoding='utf-8')
|
|
print('Public assets copied:',len(manifest))
|