#!/usr/bin/env python3
"""ARQS Flow 1.0.0 — local, bounded Chromium website QA.
No TinyFish, API key, hosted browser service, login, purchase or form submission.
Install: python -m pip install playwright; python -m playwright install chromium
Usage: python arqs_flow.py https://www.arqs.ca/imaging/product-creative
Exit: 0 reported checks pass; 1 review needed; 2 blocked/configuration error.
Screenshots and page metadata are saved locally. Protect evidence that may be sensitive.
Original ARQS implementation. Playwright and Chromium retain their respective licenses.
"""
from __future__ import annotations
import argparse, html, ipaddress, json, re, shutil, socket, sys
from datetime import datetime, timezone
from pathlib import Path
from urllib.parse import urlsplit, unquote
VERSION = "1.0.0"
DEFAULT_HOSTS = {"arqs.ca", "www.arqs.ca", "jasonsplace.ca", "www.jasonsplace.ca", "downtowntan.ca", "www.downtowntan.ca", "ackerman.ca", "www.ackerman.ca"}
DANGER = re.compile(r"(^|/)(api|admin|account|accounts|login|logout|signout|sign-out|checkout|cart|purchase|delete|remove|unsubscribe|webhook|webhooks)(/|$)", re.I)
def validate_url(value: str, hosts: set[str], allow_local: bool = False) -> str:
u = urlsplit(value)
path = unquote(u.path)
if u.scheme not in ("https", "http") or not u.hostname or u.username or u.password:
raise ValueError("Use an HTTP(S) URL without embedded credentials.")
if u.hostname.lower() not in hosts:
raise ValueError("Host is not authorized. Add --allow-host HOST only for a site you have permission to inspect.")
if u.query or DANGER.search(path) or "\\" in path or re.search(r"%[0-9a-f]{2}", path, re.I):
raise ValueError("Use a public, query-free URL. Accounts, checkout and action/API routes are excluded.")
if u.scheme != "https" and not (allow_local and u.hostname in {"localhost", "127.0.0.1", "::1"}):
raise ValueError("HTTPS is required except explicit loopback fixture tests.")
return value.split("#", 1)[0]
def public_host(host: str, allow_local: bool = False) -> bool:
if allow_local and host in {"localhost", "127.0.0.1", "::1"}: return True
try:
addresses = socket.getaddrinfo(host, None, type=socket.SOCK_STREAM)
return bool(addresses) and all(ipaddress.ip_address(a[4][0]).is_global for a in addresses)
except (OSError, ValueError): return False
AUDIT_JS = r'''(width)=>{
function inspectDocument(doc,width){
const win=doc.defaultView,issues=[],visible=el=>{const r=el.getBoundingClientRect(),s=win.getComputedStyle(el);return s.display!=='none'&&s.visibility!=='hidden'&&r.width>0&&r.height>0;};
const title=doc.title.trim(),heading=Array.from(doc.querySelectorAll('h1')).filter(visible);
if(!title)issues.push({kind:'missing-title',detail:'Document title is empty.'});
if(!heading.length)issues.push({kind:'missing-heading',detail:'No visible main heading.'});
if(!doc.querySelector('meta[name="viewport"]'))issues.push({kind:'missing-viewport',detail:'No viewport meta tag.'});
const overflow=Math.max(doc.documentElement.scrollWidth,doc.body?doc.body.scrollWidth:0)-win.innerWidth;
if(overflow>2)issues.push({kind:'horizontal-overflow',detail:overflow+'px wider than the '+width+'px viewport.'});
const images=Array.from(doc.images).filter(visible).map(i=>({src:i.currentSrc||i.src,alt:i.getAttribute('alt'),loaded:i.complete&&i.naturalWidth>0,pending:!i.complete}));
images.forEach(i=>{if(!i.loaded)issues.push({kind:i.pending?'image-unsettled':'broken-image',detail:i.src});if(i.alt===null)issues.push({kind:'missing-alt',detail:i.src});});
const anchors=Array.from(doc.querySelectorAll('a[href]')),badAnchors=[];
anchors.forEach(a=>{const href=a.getAttribute('href');if(href.startsWith('#')&&href.length>1){try{const id=decodeURIComponent(href.slice(1));if(!doc.getElementById(id)&&!Array.from(doc.getElementsByName(id)).length)badAnchors.push(href);}catch{badAnchors.push(href);}}});
for(const href of new Set(badAnchors))issues.push({kind:'broken-anchor',detail:href});
const unnamed=Array.from(doc.querySelectorAll('button,a[href],input[type="submit"]')).filter(visible).filter(e=>!(e.textContent.trim()||e.getAttribute('aria-label')||e.getAttribute('title')||e.getAttribute('value')||e.querySelector('img[alt]:not([alt=""])')||String(e.getAttribute('aria-labelledby')||'').split(/\s+/).some(id=>doc.getElementById(id)?.textContent.trim())));
if(unnamed.length)issues.push({kind:'unnamed-control',detail:unnamed.length+' visible controls need an accessible name.'});
const links=Array.from(new Set(anchors.map(a=>a.href))).slice(0,200);
return {width,title,url:win.location.href,headings:heading.map(h=>h.textContent.trim().slice(0,180)),imageCount:images.length,loadedImages:images.filter(i=>i.loaded).length,images,linkCount:anchors.length,links,overflowPixels:Math.max(0,overflow),issues};
}
return inspectDocument(document,width);}'''
def main(argv=None) -> int:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("url")
parser.add_argument("--output", type=Path, default=Path("flow-evidence-" + datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")))
parser.add_argument("--allow-host", action="append", default=[])
parser.add_argument("--allow-local", action="store_true", help="Only for your loopback development fixtures.")
parser.add_argument("--executable", help="Optional explicit Chromium executable; otherwise prefer Playwright's bundled browser.")
args = parser.parse_args(argv)
hosts = DEFAULT_HOSTS | {h.lower() for h in args.allow_host}
if args.allow_local: hosts |= {"localhost", "127.0.0.1", "::1"}
try: target = validate_url(args.url, hosts, args.allow_local)
except ValueError as exc: print(str(exc), file=sys.stderr); return 2
if args.output.exists() and any(args.output.iterdir()):
print("Output directory is not empty. Choose a new location; previous evidence will not be overwritten.", file=sys.stderr); return 2
args.output.mkdir(parents=True, exist_ok=True)
report = {"schema": "arqs-flow/1", "version": VERSION, "engine": "Playwright Chromium", "source": "live browser navigation", "target": target, "startedAt": datetime.now(timezone.utc).isoformat(), "pages": [], "limitations": ["CSS viewport sizes, not physical-device emulation.", "No form, purchase, authentication or fulfillment test.", "Automated checks are not a visual/accessibility certification.", "Bounded scroll (9000px), visible IMG elements; CSS backgrounds are not audited.", "Non-read-only requests, websockets and service workers are blocked; this may affect applications."]}
dns_cache = {}
def network_allowed(host):
if host not in dns_cache: dns_cache[host] = public_host(host, args.allow_local)
return dns_cache[host]
try:
if not network_allowed(urlsplit(target).hostname): raise RuntimeError("Public target DNS unavailable or resolves to a private address. No browser navigation attempted.")
from playwright.sync_api import sync_playwright
with sync_playwright() as p:
options = {"headless": True}
if args.executable: options["executable_path"] = args.executable
elif not Path(p.chromium.executable_path).exists():
system = shutil.which("chromium") or shutil.which("google-chrome")
if system: options["executable_path"] = system
browser = p.chromium.launch(**options)
try:
for width in (390, 768, 1440):
context = browser.new_context(viewport={"width": width, "height": 900}, service_workers="block", accept_downloads=False)
context.set_default_timeout(15000)
context.route_web_socket("**/*", lambda ws: ws.close())
page = context.new_page(); errors = []
page.on("pageerror", lambda error: errors.append(str(error)[:500]))
def route_request(route):
req = route.request; u = urlsplit(req.url)
if req.method not in ("GET", "HEAD", "OPTIONS"): route.abort(); return
if u.scheme not in ("https", "http"):
route.abort(); return
if DANGER.search(unquote(u.path)) or not u.hostname or not network_allowed(u.hostname): route.abort(); return
if req.is_navigation_request():
try: validate_url(req.url, hosts, args.allow_local)
except ValueError: route.abort(); return
route.continue_()
context.route("**/*", route_request)
try:
response = page.goto(target, wait_until="domcontentloaded", timeout=25000)
if not response or response.status >= 400: raise RuntimeError("Main document HTTP status: " + str(response.status if response else "unavailable"))
validate_url(page.url, hosts, args.allow_local)
page.wait_for_timeout(700)
height = min(page.evaluate("document.documentElement.scrollHeight"), 9000)
for y in range(0, height, 800): page.evaluate("y => scrollTo(0,y)", y); page.wait_for_timeout(100)
page.wait_for_timeout(1000);page.evaluate("scrollTo(0,0)")
result = page.evaluate(AUDIT_JS, width)
result["httpStatus"] = response.status; result["consoleErrors"] = errors
if errors: result["issues"].append({"kind": "javascript-error", "detail": str(len(errors)) + " uncaught errors; inspect consoleErrors."})
screenshot = f"viewport-{width}.png"
# Capture viewport only. Avoid unbounded full-page bitmap allocations.
page.screenshot(path=str(args.output / screenshot), full_page=False, timeout=15000)
result["screenshot"] = screenshot;report["pages"].append(result)
except Exception as exc: report["pages"].append({"width": width, "error": str(exc)[:1000]})
finally: context.close()
finally: browser.close()
except Exception as exc: report["error"] = str(exc)[:1000]
report["finishedAt"] = datetime.now(timezone.utc).isoformat()
blocked = sum(bool(x.get("error")) for x in report["pages"])
findings = sum(len(x.get("issues", [])) for x in report["pages"])
report["status"] = "BLOCKED" if report.get("error") or not report["pages"] or blocked == len(report["pages"]) else "REVIEW" if blocked or findings else "PASS"
(args.output / "report.json").write_text(json.dumps(report, indent=2), encoding="utf-8")
esc = lambda x: html.escape(str(x))
sections = []
for row in report["pages"]:
shot = '' if row.get("screenshot") else ''
sections.append('
' + esc(row["width"]) + 'px
' + esc(json.dumps(row, indent=2)) + '
' + shot + '
' + esc(report["status"]) + ' · ' + esc(target) + '
' + esc(report.get("error", "")) + '
' + ''.join(sections) + '