Files
site-servaki-online/verify.py
T

107 lines
7.1 KiB
Python

"""Validate rendered site, links, structured data, images and served responses."""
from pathlib import Path
from html.parser import HTMLParser
from urllib.parse import urlsplit,unquote
from urllib.request import build_opener,ProxyHandler
from xml.etree import ElementTree as ET
import json,re,hashlib
ROOT=Path(__file__).resolve().parent;PUBLIC=ROOT/'public'
class Page(HTMLParser):
def __init__(self,text):
super().__init__(convert_charrefs=True)
self.h1=0;self.links=[];self.images=[];self.ids=[];self.ld=[];self.title='';self.description='';self.canonical='';self.script=None;self.in_title=False
self.feed(text)
def handle_starttag(self,tag,attrs):
a=dict(attrs)
if 'id' in a:self.ids.append(a['id'])
if tag=='h1':self.h1+=1
if tag=='title':self.in_title=True
if tag=='a' and a.get('href'):self.links.append(a['href'])
if tag=='link' and a.get('rel')=='stylesheet':self.links.append(a['href'])
if tag=='link' and a.get('rel')=='canonical':self.canonical=a['href']
if tag=='meta' and a.get('name')=='description':self.description=a.get('content','')
if tag=='img' and a.get('src'):
self.images.append(a)
self.links.append(a['src'])
self.links += [part.strip().split()[0] for part in a.get('srcset','').split(',') if part.strip()]
if tag=='meta' and a.get('property')=='og:image':self.links.append(a['content'])
if tag=='script' and a.get('type')=='application/ld+json':self.script=''
def handle_endtag(self,tag):
if tag=='title':self.in_title=False
if tag=='script' and self.script is not None:
self.ld.append(json.loads(self.script));self.script=None
def handle_data(self,d):
if self.in_title:self.title+=d
if self.script is not None:self.script+=d
def check(text,rel):
p=Page(text)
assert p.h1==1,(rel,'expected one H1')
assert p.title and p.description and p.canonical,(rel,'missing metadata')
assert len(set(p.ids))==len(p.ids),(rel,'duplicate IDs')
assert p.canonical=='https://servaki.online/'+('' if rel=='index.html' else rel.removesuffix('index.html')),(rel,'canonical')
for link in p.links:
u=urlsplit(link)
if u.scheme and u.scheme not in ('http','https'):continue
if u.netloc and u.netloc!='servaki.online':continue
target=PUBLIC/u.path.lstrip('/') if u.path.startswith('/') else PUBLIC/Path(rel).parent/u.path
if not u.path:target=PUBLIC/rel
if target.is_dir():target=target/'index.html'
assert target.is_file(),(rel,'missing link',link)
if u.fragment and target.suffix=='.html':
assert unquote(u.fragment) in Page(target.read_text(encoding='utf-8')).ids,(rel,'missing fragment',link)
for im in p.images:
assert 'alt' in im,(rel,'missing image alt')
if not im['src'].startswith('https://mc.yandex.ru/'):
assert im.get('width') and im.get('height'),(rel,'missing dimensions')
return p
def main():
files=sorted(PUBLIC.rglob('*.html'));titles=[];descriptions=[];checks=[]
scenes=json.loads((ROOT/'content/flow-scenes.json').read_text(encoding='utf-8'))
pages=json.loads((ROOT/'content/pages.json').read_text(encoding='utf-8'))
card_copy=json.loads((ROOT/'content/card-copy.json').read_text(encoding='utf-8'))
assert set(scenes)=={p['slug'] for p in pages},'missing or extra equipment scene'
assert set(card_copy)==set(scenes),'missing or extra homepage card description'
css=(PUBLIC/'assets/editorial/editorial.css').read_text(encoding='utf-8')
assert '@media(prefers-reduced-motion:reduce){.flow-link span{animation:none' in css,'missing static motion alternative'
opener=build_opener(ProxyHandler({}))
for file in files:
rel=file.relative_to(PUBLIC).as_posix();text=file.read_text(encoding='utf-8');p=check(text,rel)
if rel!='index.html' and rel.count('/')==2:
slug=Path(rel).parent.name
assert text.count('class="system-flow"')==1,(rel,'equipment diagram')
assert text.count('class="flow-device"')==3 and text.count('class="flow-link"')==2,(rel,'device and traffic path')
assert scenes[slug]['title'] in text,(rel,'wrong diagram')
photo=next((im for im in p.images if im['src'].startswith('/assets/editorial/photo-')),None)
assert photo and '520px' in photo.get('sizes',''),(rel,'responsive article photograph')
titles.append(p.title);descriptions.append(p.description)
route='/'+('' if rel=='index.html' else rel.removesuffix('index.html'))
served=opener.open('http://127.0.0.1:5186'+route,timeout=10)
assert served.status==200 and served.read()==file.read_bytes(),(rel,'HTTP mismatch')
checks.append({'path':route,'http':200,'h1':p.h1,'images':len(p.images),'schema_blocks':len(p.ld),'sha256':hashlib.sha256(file.read_bytes()).hexdigest()})
assert len(set(titles))==len(titles),'duplicate titles'
assert len(set(descriptions))==len(descriptions),'duplicate descriptions'
home=(PUBLIC/'index.html').read_text(encoding='utf-8')
for slug,description in card_copy.items():
assert description in home,(slug,'homepage card description not visible')
home_schema=Page(home).ld[0]['@graph']
plans=next(item for item in home_schema if item.get('@id')=='https://servaki.online/#plans')
expected_prices={'Старт':9000,'Бизнес':21000,'Премиум':45000}
actual_prices={offer['name']:int(offer['priceSpecification']['price']) for offer in plans['offers']}
assert actual_prices==expected_prices,('structured tariff prices',actual_prices)
for price in expected_prices.values():
visible=f'<span class="amount">{price:,}</span>'.replace(',', ' ')
assert home.count(visible)==1,('visible tariff price',price)
sample=PUBLIC/'services/networks-vpn/index.html'
broken=sample.read_text(encoding='utf-8').replace('/assets/editorial/photo-networks-vpn-960.webp','/assets/editorial/definitely-missing.webp',1)
try:check(broken,'services/networks-vpn/index.html')
except AssertionError:negative=True
else:raise AssertionError('Negative control did not detect broken image')
locations=[e.text for e in ET.parse(PUBLIC/'sitemap.xml').iter() if e.tag.endswith('}loc')]
assert len(locations)==16 and len(set(locations))==16,'sitemap pages'
small_photos=list((PUBLIC/'assets/editorial').glob('photo-*-480.webp'))
assert len(small_photos)==len(pages) and max(p.stat().st_size for p in small_photos)<50000,'oversized mobile photograph'
report={'html_pages':len(files),'article_pages':12,'animated_equipment_scenes':len(scenes),'max_mobile_photo_bytes':max(p.stat().st_size for p in small_photos),'animation_css_bytes':len(css.encode('utf-8')),'unique_titles':True,'unique_descriptions':True,'tariff_prices':expected_prices,'links_and_fragments':'pass','negative_control':negative,'sitemap_urls':len(locations),'checks':checks}
(ROOT/'verification.json').write_text(json.dumps(report,ensure_ascii=False,indent=2),encoding='utf-8')
print(json.dumps({k:v for k,v in report.items() if k!='checks'},ensure_ascii=False))
if __name__=='__main__':main()