107 lines
7.1 KiB
Python
107 lines
7.1 KiB
Python
"""Validate rendered site, links, structured data, images and served responses."""
|
|
from pathlib import Path
|
|
from html.parser import HTMLParser
|
|
from urllib.parse import urlsplit,unquote
|
|
from urllib.request import build_opener,ProxyHandler
|
|
from xml.etree import ElementTree as ET
|
|
import json,re,hashlib
|
|
ROOT=Path(__file__).resolve().parent;PUBLIC=ROOT/'public'
|
|
class Page(HTMLParser):
|
|
def __init__(self,text):
|
|
super().__init__(convert_charrefs=True)
|
|
self.h1=0;self.links=[];self.images=[];self.ids=[];self.ld=[];self.title='';self.description='';self.canonical='';self.script=None;self.in_title=False
|
|
self.feed(text)
|
|
def handle_starttag(self,tag,attrs):
|
|
a=dict(attrs)
|
|
if 'id' in a:self.ids.append(a['id'])
|
|
if tag=='h1':self.h1+=1
|
|
if tag=='title':self.in_title=True
|
|
if tag=='a' and a.get('href'):self.links.append(a['href'])
|
|
if tag=='link' and a.get('rel')=='stylesheet':self.links.append(a['href'])
|
|
if tag=='link' and a.get('rel')=='canonical':self.canonical=a['href']
|
|
if tag=='meta' and a.get('name')=='description':self.description=a.get('content','')
|
|
if tag=='img' and a.get('src'):
|
|
self.images.append(a)
|
|
self.links.append(a['src'])
|
|
self.links += [part.strip().split()[0] for part in a.get('srcset','').split(',') if part.strip()]
|
|
if tag=='meta' and a.get('property')=='og:image':self.links.append(a['content'])
|
|
if tag=='script' and a.get('type')=='application/ld+json':self.script=''
|
|
def handle_endtag(self,tag):
|
|
if tag=='title':self.in_title=False
|
|
if tag=='script' and self.script is not None:
|
|
self.ld.append(json.loads(self.script));self.script=None
|
|
def handle_data(self,d):
|
|
if self.in_title:self.title+=d
|
|
if self.script is not None:self.script+=d
|
|
def check(text,rel):
|
|
p=Page(text)
|
|
assert p.h1==1,(rel,'expected one H1')
|
|
assert p.title and p.description and p.canonical,(rel,'missing metadata')
|
|
assert len(set(p.ids))==len(p.ids),(rel,'duplicate IDs')
|
|
assert p.canonical=='https://servaki.online/'+('' if rel=='index.html' else rel.removesuffix('index.html')),(rel,'canonical')
|
|
for link in p.links:
|
|
u=urlsplit(link)
|
|
if u.scheme and u.scheme not in ('http','https'):continue
|
|
if u.netloc and u.netloc!='servaki.online':continue
|
|
target=PUBLIC/u.path.lstrip('/') if u.path.startswith('/') else PUBLIC/Path(rel).parent/u.path
|
|
if not u.path:target=PUBLIC/rel
|
|
if target.is_dir():target=target/'index.html'
|
|
assert target.is_file(),(rel,'missing link',link)
|
|
if u.fragment and target.suffix=='.html':
|
|
assert unquote(u.fragment) in Page(target.read_text(encoding='utf-8')).ids,(rel,'missing fragment',link)
|
|
for im in p.images:
|
|
assert 'alt' in im,(rel,'missing image alt')
|
|
if not im['src'].startswith('https://mc.yandex.ru/'):
|
|
assert im.get('width') and im.get('height'),(rel,'missing dimensions')
|
|
return p
|
|
def main():
|
|
files=sorted(PUBLIC.rglob('*.html'));titles=[];descriptions=[];checks=[]
|
|
scenes=json.loads((ROOT/'content/flow-scenes.json').read_text(encoding='utf-8'))
|
|
pages=json.loads((ROOT/'content/pages.json').read_text(encoding='utf-8'))
|
|
card_copy=json.loads((ROOT/'content/card-copy.json').read_text(encoding='utf-8'))
|
|
assert set(scenes)=={p['slug'] for p in pages},'missing or extra equipment scene'
|
|
assert set(card_copy)==set(scenes),'missing or extra homepage card description'
|
|
css=(PUBLIC/'assets/editorial/editorial.css').read_text(encoding='utf-8')
|
|
assert '@media(prefers-reduced-motion:reduce){.flow-link span{animation:none' in css,'missing static motion alternative'
|
|
opener=build_opener(ProxyHandler({}))
|
|
for file in files:
|
|
rel=file.relative_to(PUBLIC).as_posix();text=file.read_text(encoding='utf-8');p=check(text,rel)
|
|
if rel!='index.html' and rel.count('/')==2:
|
|
slug=Path(rel).parent.name
|
|
assert text.count('class="system-flow"')==1,(rel,'equipment diagram')
|
|
assert text.count('class="flow-device"')==3 and text.count('class="flow-link"')==2,(rel,'device and traffic path')
|
|
assert scenes[slug]['title'] in text,(rel,'wrong diagram')
|
|
photo=next((im for im in p.images if im['src'].startswith('/assets/editorial/photo-')),None)
|
|
assert photo and '520px' in photo.get('sizes',''),(rel,'responsive article photograph')
|
|
titles.append(p.title);descriptions.append(p.description)
|
|
route='/'+('' if rel=='index.html' else rel.removesuffix('index.html'))
|
|
served=opener.open('http://127.0.0.1:5186'+route,timeout=10)
|
|
assert served.status==200 and served.read()==file.read_bytes(),(rel,'HTTP mismatch')
|
|
checks.append({'path':route,'http':200,'h1':p.h1,'images':len(p.images),'schema_blocks':len(p.ld),'sha256':hashlib.sha256(file.read_bytes()).hexdigest()})
|
|
assert len(set(titles))==len(titles),'duplicate titles'
|
|
assert len(set(descriptions))==len(descriptions),'duplicate descriptions'
|
|
home=(PUBLIC/'index.html').read_text(encoding='utf-8')
|
|
for slug,description in card_copy.items():
|
|
assert description in home,(slug,'homepage card description not visible')
|
|
home_schema=Page(home).ld[0]['@graph']
|
|
plans=next(item for item in home_schema if item.get('@id')=='https://servaki.online/#plans')
|
|
expected_prices={'Старт':9000,'Бизнес':21000,'Премиум':45000}
|
|
actual_prices={offer['name']:int(offer['priceSpecification']['price']) for offer in plans['offers']}
|
|
assert actual_prices==expected_prices,('structured tariff prices',actual_prices)
|
|
for price in expected_prices.values():
|
|
visible=f'<span class="amount">{price:,}</span>'.replace(',', ' ')
|
|
assert home.count(visible)==1,('visible tariff price',price)
|
|
sample=PUBLIC/'services/networks-vpn/index.html'
|
|
broken=sample.read_text(encoding='utf-8').replace('/assets/editorial/photo-networks-vpn-960.webp','/assets/editorial/definitely-missing.webp',1)
|
|
try:check(broken,'services/networks-vpn/index.html')
|
|
except AssertionError:negative=True
|
|
else:raise AssertionError('Negative control did not detect broken image')
|
|
locations=[e.text for e in ET.parse(PUBLIC/'sitemap.xml').iter() if e.tag.endswith('}loc')]
|
|
assert len(locations)==16 and len(set(locations))==16,'sitemap pages'
|
|
small_photos=list((PUBLIC/'assets/editorial').glob('photo-*-480.webp'))
|
|
assert len(small_photos)==len(pages) and max(p.stat().st_size for p in small_photos)<50000,'oversized mobile photograph'
|
|
report={'html_pages':len(files),'article_pages':12,'animated_equipment_scenes':len(scenes),'max_mobile_photo_bytes':max(p.stat().st_size for p in small_photos),'animation_css_bytes':len(css.encode('utf-8')),'unique_titles':True,'unique_descriptions':True,'tariff_prices':expected_prices,'links_and_fragments':'pass','negative_control':negative,'sitemap_urls':len(locations),'checks':checks}
|
|
(ROOT/'verification.json').write_text(json.dumps(report,ensure_ascii=False,indent=2),encoding='utf-8')
|
|
print(json.dumps({k:v for k,v in report.items() if k!='checks'},ensure_ascii=False))
|
|
if __name__=='__main__':main()
|