From eb31d56ad30011e41d51e1fd1c4418ac58697dde Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:47:14 +0700 Subject: [PATCH 01/15] Add lightweight website measurement client --- landing/analytics.js | 162 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 162 insertions(+) create mode 100644 landing/analytics.js diff --git a/landing/analytics.js b/landing/analytics.js new file mode 100644 index 000000000..03cce0a59 --- /dev/null +++ b/landing/analytics.js @@ -0,0 +1,162 @@ +(() => { + const config = document.getElementById('arsas-analytics'); + if (!(config instanceof HTMLScriptElement)) return; + + const measurementId = config.dataset.measurementId || ''; + if (!/^G-[A-Z0-9]+$/.test(measurementId)) return; + if (navigator.doNotTrack === '1' || window.doNotTrack === '1') return; + + const stableVersion = config.dataset.stableVersion || 'unknown'; + const siteLanguage = document.documentElement.lang || 'en'; + const contentGroup = document.body.dataset.page || 'unknown'; + const pagePath = `${window.location.pathname}${window.location.search}`; + + window.dataLayer = window.dataLayer || []; + window.gtag = window.gtag || function gtag() { + window.dataLayer.push(arguments); + }; + + window.gtag('js', new Date()); + window.gtag('config', measurementId, { + send_page_view: false, + allow_google_signals: false, + allow_ad_personalization_signals: false, + transport_type: 'beacon' + }); + + const tag = document.createElement('script'); + tag.async = true; + tag.src = `https://www.googletagmanager.com/gtag/js?id=${encodeURIComponent(measurementId)}`; + document.head.appendChild(tag); + + const common = { + page_path: pagePath, + page_title: document.title, + site_language: siteLanguage, + content_group: contentGroup, + stable_version: stableVersion, + transport_type: 'beacon' + }; + + window.gtag('event', 'page_view', { + ...common, + page_location: window.location.href + }); + + if (contentGroup === 'none') { + window.gtag('event', 'page_not_found', { + ...common, + referrer: document.referrer || '(direct)' + }); + } + + const classifyDownload = href => { + if (href.endsWith('/ARSAS-Windows-x64-Setup.exe')) return ['download_installer', 'ARSAS-Windows-x64-Setup.exe']; + if (href.endsWith('/ARSAS-Windows-x64-Portable.zip')) return ['download_portable', 'ARSAS-Windows-x64-Portable.zip']; + if (href.endsWith('/ARSAS-Windows-x64-SHA256SUMS.txt')) return ['download_checksums', 'ARSAS-Windows-x64-SHA256SUMS.txt']; + return null; + }; + + document.addEventListener('click', event => { + const target = event.target instanceof Element ? event.target.closest('a[href]') : null; + if (!(target instanceof HTMLAnchorElement)) return; + + let href; + try { + href = new URL(target.href, window.location.href); + } catch { + return; + } + + const download = classifyDownload(href.href); + if (download) { + const [eventName, fileName] = download; + window.gtag('event', eventName, { + ...common, + file_name: fileName, + link_url: href.href, + link_text: (target.textContent || '').trim().slice(0, 100) + }); + } + + const alternateLanguage = target.getAttribute('hreflang'); + if (alternateLanguage && alternateLanguage !== siteLanguage) { + window.gtag('event', 'language_switch', { + ...common, + destination_language: alternateLanguage, + link_url: href.href + }); + } + }, { capture: true }); + + const rating = (name, value) => { + const thresholds = { + LCP: [2500, 4000], + CLS: [0.1, 0.25], + INP: [200, 500] + }[name]; + if (!thresholds) return 'diagnostic'; + if (value <= thresholds[0]) return 'good'; + if (value <= thresholds[1]) return 'needs-improvement'; + return 'poor'; + }; + + const reportVital = (name, value, source = 'browser-rum') => { + if (!Number.isFinite(value) || value < 0) return; + const rounded = name === 'CLS' ? Math.round(value * 1000) / 1000 : Math.round(value); + window.gtag('event', `web_vital_${name.toLowerCase()}`, { + ...common, + value: rounded, + metric_name: name, + metric_value: rounded, + metric_rating: rating(name, rounded), + metric_source: source, + non_interaction: true + }); + }; + + let lcp = 0; + let cls = 0; + let inp = 0; + let reported = false; + + try { + const navigation = performance.getEntriesByType('navigation')[0]; + if (navigation && Number.isFinite(navigation.responseStart)) { + reportVital('TTFB', navigation.responseStart, 'navigation-timing'); + } + + new PerformanceObserver(list => { + const entries = list.getEntries(); + const latest = entries[entries.length - 1]; + if (latest) lcp = latest.startTime; + }).observe({ type: 'largest-contentful-paint', buffered: true }); + + new PerformanceObserver(list => { + list.getEntries().forEach(entry => { + if (!entry.hadRecentInput) cls += entry.value; + }); + }).observe({ type: 'layout-shift', buffered: true }); + + new PerformanceObserver(list => { + list.getEntries().forEach(entry => { + if (entry.interactionId && entry.duration > inp) inp = entry.duration; + }); + }).observe({ type: 'event', buffered: true, durationThreshold: 40 }); + } catch { + // Older browsers still provide page, download, language and 404 measurement. + } + + const flushVitals = () => { + if (reported) return; + reported = true; + reportVital('LCP', lcp); + reportVital('CLS', cls); + if (inp > 0) reportVital('INP', inp, 'event-timing'); + }; + + document.addEventListener('visibilitychange', () => { + if (document.visibilityState === 'hidden') flushVitals(); + }); + window.addEventListener('pagehide', flushVitals, { once: true }); +})(); From 23fbc750edf29fdab5d21ca740f534441251e21b Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:47:35 +0700 Subject: [PATCH 02/15] Wire analytics client into shared footer --- landing/partials/footer.html | 1 + 1 file changed, 1 insertion(+) diff --git a/landing/partials/footer.html b/landing/partials/footer.html index ced79ccbe..5986ac8c3 100644 --- a/landing/partials/footer.html +++ b/landing/partials/footer.html @@ -25,4 +25,5 @@ + From 0fc2a919d448da1e81bd19a029f37ec1a9101c70 Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:48:33 +0700 Subject: [PATCH 03/15] Use deploy-time analytics placeholder --- landing/partials/footer.html | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/landing/partials/footer.html b/landing/partials/footer.html index 5986ac8c3..5204177be 100644 --- a/landing/partials/footer.html +++ b/landing/partials/footer.html @@ -25,5 +25,5 @@ - + From cdc204c54d1f0fcc470d9f761d0c5ddc1dc20518 Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:50:09 +0700 Subject: [PATCH 04/15] Add site measurement configuration injector --- scripts/inject-site-measurement.py | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) create mode 100644 scripts/inject-site-measurement.py diff --git a/scripts/inject-site-measurement.py b/scripts/inject-site-measurement.py new file mode 100644 index 000000000..8be5a9c29 --- /dev/null +++ b/scripts/inject-site-measurement.py @@ -0,0 +1,23 @@ +#!/usr/bin/env python3 +"""Configure optional site measurement after the deterministic website build.""" + +from __future__ import annotations + +import argparse +from pathlib import Path + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("site", nargs="?", default="_site") + parser.add_argument("--measurement-id", default="") + args = parser.parse_args() + site = Path(args.site).resolve() + if not site.is_dir(): + raise SystemExit(f"Site directory does not exist: {site}") + print("Site measurement configuration ready.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From 76e5e39d87dd53a8d8fbdbc37a30e326578bcdd8 Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:50:33 +0700 Subject: [PATCH 05/15] Implement deploy-time measurement configuration --- scripts/inject-site-measurement.py | 60 +++++++++++++++++++++++++++--- 1 file changed, 55 insertions(+), 5 deletions(-) diff --git a/scripts/inject-site-measurement.py b/scripts/inject-site-measurement.py index 8be5a9c29..1f8432fe3 100644 --- a/scripts/inject-site-measurement.py +++ b/scripts/inject-site-measurement.py @@ -4,18 +4,68 @@ from __future__ import annotations import argparse +import json +import os +import re from pathlib import Path +PLACEHOLDER = "__ARSAS_GA4_MEASUREMENT_ID__" +MEASUREMENT_PATTERN = re.compile(r"G-[A-Z0-9]+") + + +def configure(site: Path, measurement_id: str) -> None: + measurement_id = measurement_id.strip().upper() + if measurement_id and not MEASUREMENT_PATTERN.fullmatch(measurement_id): + raise SystemExit("Measurement ID must use the G-XXXXXXXX format") + if not site.is_dir(): + raise SystemExit(f"Site directory does not exist: {site}") + if not (site / "analytics.js").is_file(): + raise SystemExit("Built site is missing analytics.js") + + pages = sorted(site.rglob("*.html")) + if not pages: + raise SystemExit("Built site contains no HTML pages") + + replacements = 0 + for page in pages: + text = page.read_text(encoding="utf-8") + count = text.count(PLACEHOLDER) + if count != 1: + raise SystemExit( + f"{page.relative_to(site)} must contain exactly one measurement placeholder, found {count}" + ) + page.write_text(text.replace(PLACEHOLDER, measurement_id), encoding="utf-8") + replacements += count + + build_info_path = site / "build-info.json" + if build_info_path.exists(): + build_info = json.loads(build_info_path.read_text(encoding="utf-8")) + build_info["measurement"] = { + "provider": "google-analytics-4", + "enabled": bool(measurement_id), + "client": "analytics.js", + "doNotTrackRespected": True, + "advertisingSignals": False, + } + build_info_path.write_text( + json.dumps(build_info, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8", + ) + + state = "enabled" if measurement_id else "disabled" + print(f"ARSAS site measurement {state}: {replacements} pages configured.") + def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("site", nargs="?", default="_site") - parser.add_argument("--measurement-id", default="") + parser.add_argument( + "--measurement-id", + default=os.environ.get("GA4_MEASUREMENT_ID", ""), + help="Public measurement ID. Empty keeps client measurement disabled.", + ) args = parser.parse_args() - site = Path(args.site).resolve() - if not site.is_dir(): - raise SystemExit(f"Site directory does not exist: {site}") - print("Site measurement configuration ready.") + configure(Path(args.site).resolve(), args.measurement_id) return 0 From 90c90927073bcbf260c3b1874614f2644a259071 Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:51:18 +0700 Subject: [PATCH 06/15] Add internal link and deployed 404 health checks --- scripts/check-site-health.py | 207 +++++++++++++++++++++++++++++++++++ 1 file changed, 207 insertions(+) create mode 100644 scripts/check-site-health.py diff --git a/scripts/check-site-health.py b/scripts/check-site-health.py new file mode 100644 index 000000000..6eea261df --- /dev/null +++ b/scripts/check-site-health.py @@ -0,0 +1,207 @@ +#!/usr/bin/env python3 +"""Check generated ARSAS links locally and optionally verify the deployed site.""" + +from __future__ import annotations + +import argparse +import json +import sys +import time +import urllib.error +import urllib.parse +import urllib.request +from dataclasses import asdict, dataclass +from html.parser import HTMLParser +from pathlib import Path + +CANONICAL_ROOT = "https://masarray.github.io/arsas/" +USER_AGENT = "ARSAS-Site-Health/1.0" +SKIP_SCHEMES = {"mailto", "tel", "javascript", "data"} + + +class PageParser(HTMLParser): + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self.refs: list[tuple[str, str]] = [] + self.ids: set[str] = set() + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + values = dict(attrs) + element_id = values.get("id") + if element_id: + self.ids.add(element_id) + for key in ("href", "src"): + value = values.get(key) + if value: + self.refs.append((key, value)) + + +@dataclass +class Finding: + severity: str + source: str + target: str + message: str + + +def parse_pages(site: Path) -> dict[Path, PageParser]: + parsed: dict[Path, PageParser] = {} + for page in sorted(site.rglob("*.html")): + parser = PageParser() + parser.feed(page.read_text(encoding="utf-8")) + parsed[page.resolve()] = parser + return parsed + + +def local_target(site: Path, page: Path, reference: str) -> tuple[Path | None, str]: + split = urllib.parse.urlsplit(reference) + if split.scheme in SKIP_SCHEMES: + return None, "" + if split.scheme in {"http", "https"}: + if reference.startswith(CANONICAL_ROOT): + relative = urllib.parse.urlsplit(reference[len(CANONICAL_ROOT):]).path + target = site / (relative or "index.html") + return target.resolve(), split.fragment + return None, "" + if split.netloc: + return None, "" + clean = urllib.parse.unquote(split.path) + if not clean: + target = page + elif clean.endswith("/"): + target = page.parent / clean / "index.html" + else: + target = page.parent / clean + return target.resolve(), split.fragment + + +def check_local(site: Path) -> list[Finding]: + findings: list[Finding] = [] + pages = parse_pages(site) + for page, parser in pages.items(): + source = str(page.relative_to(site)) + for _, reference in parser.refs: + target, fragment = local_target(site, page, reference) + if target is None: + continue + try: + target.relative_to(site) + except ValueError: + findings.append(Finding("error", source, reference, "local reference escapes the site root")) + continue + if not target.exists(): + findings.append(Finding("error", source, reference, "target does not exist")) + continue + if fragment and target.suffix.lower() == ".html": + target_parser = pages.get(target) + if target_parser is None: + target_parser = PageParser() + target_parser.feed(target.read_text(encoding="utf-8")) + pages[target] = target_parser + if fragment not in target_parser.ids: + findings.append(Finding("error", source, reference, f"fragment #{fragment} does not exist")) + return findings + + +def request_status(url: str, timeout: float = 20.0) -> tuple[int, str]: + request = urllib.request.Request(url, headers={"User-Agent": USER_AGENT, "Accept": "text/html,*/*"}) + try: + with urllib.request.urlopen(request, timeout=timeout) as response: + return int(response.status), response.geturl() + except urllib.error.HTTPError as exc: + return int(exc.code), exc.geturl() + except (urllib.error.URLError, TimeoutError) as exc: + return 0, str(exc) + + +def sitemap_urls(site: Path) -> list[str]: + import xml.etree.ElementTree as ET + + tree = ET.parse(site / "sitemap.xml") + namespace = {"sm": "http://www.sitemaps.org/schemas/sitemap/0.9"} + return [ + (node.text or "").strip() + for node in tree.findall("sm:url/sm:loc", namespace) + if (node.text or "").strip() + ] + + +def check_remote(site: Path, base_url: str) -> list[Finding]: + findings: list[Finding] = [] + for url in sitemap_urls(site): + status, final_url = request_status(url) + if status != 200: + findings.append(Finding("error", "sitemap.xml", url, f"deployed page returned HTTP {status}: {final_url}")) + time.sleep(0.05) + + missing_url = urllib.parse.urljoin(base_url.rstrip("/") + "/", "__arsas_measurement_missing_page__.html") + status, final_url = request_status(missing_url) + if status != 404: + findings.append(Finding("error", "404-probe", missing_url, f"expected HTTP 404, received {status}: {final_url}")) + + latest = json.loads((site / "latest.json").read_text(encoding="utf-8")) + for key in ("installer", "portable", "checksums"): + url = str(latest[key]["url"]) + status, final_url = request_status(url, timeout=45.0) + if status != 200: + findings.append(Finding("error", "latest.json", url, f"release asset returned HTTP {status}: {final_url}")) + return findings + + +def write_reports(output: Path, findings: list[Finding], remote: bool) -> None: + output.mkdir(parents=True, exist_ok=True) + errors = [item for item in findings if item.severity == "error"] + warnings = [item for item in findings if item.severity == "warning"] + payload = { + "schemaVersion": 1, + "remoteChecked": remote, + "errors": len(errors), + "warnings": len(warnings), + "findings": [asdict(item) for item in findings], + } + (output / "site-health.json").write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8") + + lines = [ + "# ARSAS site health", + "", + f"- Internal/deployed errors: **{len(errors)}**", + f"- Warnings: **{len(warnings)}**", + f"- Remote deployment checked: **{'yes' if remote else 'no'}**", + "", + ] + if findings: + lines.extend(["## Findings", ""]) + for item in findings: + lines.append(f"- **{item.severity.upper()}** `{item.source}` → `{item.target}` — {item.message}") + else: + lines.append("No broken local links, missing fragments, failed deployed pages or invalid 404 response were found.") + (output / "site-health.md").write_text("\n".join(lines) + "\n", encoding="utf-8") + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--site", default="_site") + parser.add_argument("--output", default="_measurement") + parser.add_argument("--remote", action="store_true") + parser.add_argument("--base-url", default=CANONICAL_ROOT) + args = parser.parse_args() + + site = Path(args.site).resolve() + if not site.is_dir(): + raise SystemExit(f"Site directory does not exist: {site}") + + findings = check_local(site) + if args.remote: + findings.extend(check_remote(site, args.base_url)) + write_reports(Path(args.output).resolve(), findings, args.remote) + + errors = [item for item in findings if item.severity == "error"] + if errors: + print(f"ARSAS site health failed with {len(errors)} error(s).", file=sys.stderr) + return 1 + print("ARSAS site health passed: links, fragments, deployed pages and 404 behavior are valid.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From 5aa4ae2804bb6888622bb4a529403dd00db8adc1 Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:51:48 +0700 Subject: [PATCH 07/15] Add measurement instrumentation validation --- scripts/validate-site-measurement.py | 118 +++++++++++++++++++++++++++ 1 file changed, 118 insertions(+) create mode 100644 scripts/validate-site-measurement.py diff --git a/scripts/validate-site-measurement.py b/scripts/validate-site-measurement.py new file mode 100644 index 000000000..2317aa359 --- /dev/null +++ b/scripts/validate-site-measurement.py @@ -0,0 +1,118 @@ +#!/usr/bin/env python3 +"""Validate ARSAS client measurement instrumentation in a rendered site.""" + +from __future__ import annotations + +import argparse +import json +import re +import sys +from html.parser import HTMLParser +from pathlib import Path + +MEASUREMENT_PATTERN = re.compile(r"G-[A-Z0-9]+") +PLACEHOLDER = "__ARSAS_GA4_MEASUREMENT_ID__" + + +class Parser(HTMLParser): + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self.analytics: list[dict[str, str | None]] = [] + self.body_page: str | None = None + self.language: str | None = None + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + values = dict(attrs) + if tag == "html": + self.language = values.get("lang") + elif tag == "body": + self.body_page = values.get("data-page") + elif tag == "script" and values.get("id") == "arsas-analytics": + self.analytics.append(values) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("site", nargs="?", default="_site") + parser.add_argument("--measurement-id", default="") + args = parser.parse_args() + + site = Path(args.site).resolve() + expected_id = args.measurement_id.strip().upper() + errors: list[str] = [] + if expected_id and not MEASUREMENT_PATTERN.fullmatch(expected_id): + errors.append("expected measurement ID is invalid") + + client = site / "analytics.js" + if not client.exists(): + errors.append("analytics.js is missing") + client_text = "" + else: + client_text = client.read_text(encoding="utf-8") + for required in ( + "download_installer", "download_portable", "download_checksums", + "page_not_found", "language_switch", "web_vital_lcp", "web_vital_cls", + "web_vital_inp", "navigator.doNotTrack", "allow_google_signals: false", + "allow_ad_personalization_signals: false", + ): + if required not in client_text: + errors.append(f"analytics.js missing measurement contract: {required}") + + pages = sorted(site.rglob("*.html")) + if not pages: + errors.append("rendered site has no HTML pages") + for page in pages: + text = page.read_text(encoding="utf-8") + label = page.relative_to(site) + if PLACEHOLDER in text: + errors.append(f"{label}: unresolved measurement placeholder") + parsed = Parser() + parsed.feed(text) + if len(parsed.analytics) != 1: + errors.append(f"{label}: expected one shared analytics client") + continue + script = parsed.analytics[0] + if script.get("src") != "analytics.js" or script.get("defer") is None: + errors.append(f"{label}: analytics client must be local and deferred") + actual_id = script.get("data-measurement-id") or "" + if actual_id != expected_id: + errors.append(f"{label}: measurement ID does not match configured deployment value") + stable_version = script.get("data-stable-version") or "" + if not re.fullmatch(r"\d+\.\d+\.\d+", stable_version): + errors.append(f"{label}: stable release version is missing from measurement context") + if parsed.language not in {"en", "id"}: + errors.append(f"{label}: language is unavailable for traffic segmentation") + if page.name == "404.html" and parsed.body_page != "none": + errors.append("404.html must use data-page=none for page_not_found measurement") + + build_info_path = site / "build-info.json" + if not build_info_path.exists(): + errors.append("build-info.json is missing") + else: + build_info = json.loads(build_info_path.read_text(encoding="utf-8")) + measurement = build_info.get("measurement") + if not isinstance(measurement, dict): + errors.append("build-info.json is missing measurement status") + else: + if measurement.get("provider") != "google-analytics-4": + errors.append("build-info.json has invalid measurement provider") + if measurement.get("enabled") is not bool(expected_id): + errors.append("build-info.json measurement enabled state is incorrect") + if measurement.get("doNotTrackRespected") is not True: + errors.append("build-info.json must declare Do Not Track handling") + if measurement.get("advertisingSignals") is not False: + errors.append("build-info.json must declare advertising signals disabled") + + errors = list(dict.fromkeys(errors)) + if errors: + print("ARSAS site-measurement validation failed:", file=sys.stderr) + for error in errors: + print(f"- {error}", file=sys.stderr) + return 1 + state = "enabled" if expected_id else "disabled/no-op" + print(f"ARSAS site-measurement validation passed: {len(pages)} pages, client {state}, downloads, language, 404 and Core Web Vitals contracts present.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From ec248d4e414a4668535af3350207a2727056824a Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:53:42 +0700 Subject: [PATCH 08/15] Add aggregated analytics and search insight report --- scripts/build-site-measurement-report.py | 496 +++++++++++++++++++++++ 1 file changed, 496 insertions(+) create mode 100644 scripts/build-site-measurement-report.py diff --git a/scripts/build-site-measurement-report.py b/scripts/build-site-measurement-report.py new file mode 100644 index 000000000..03386e019 --- /dev/null +++ b/scripts/build-site-measurement-report.py @@ -0,0 +1,496 @@ +#!/usr/bin/env python3 +"""Build a private, aggregated ARSAS website measurement report for GitHub Actions.""" + +from __future__ import annotations + +import argparse +import base64 +import json +import os +import sys +import urllib.parse +from collections import defaultdict +from datetime import date, datetime, timedelta, timezone +from pathlib import Path +from typing import Any + +CANONICAL_ROOT = "https://masarray.github.io/arsas/" +DEFAULT_PSI_URLS = ( + CANONICAL_ROOT, + CANONICAL_ROOT + "download.html", + CANONICAL_ROOT + "smart-reporting.html", + CANONICAL_ROOT + "guides.html", + CANONICAL_ROOT + "id.html", + CANONICAL_ROOT + "unduh.html", +) + + +def load_service_account() -> dict[str, Any] | None: + raw = os.environ.get("GOOGLE_SERVICE_ACCOUNT_JSON", "").strip() + if not raw: + return None + if raw.startswith("base64:"): + raw = base64.b64decode(raw.removeprefix("base64:")).decode("utf-8") + value = json.loads(raw) + if not isinstance(value, dict) or value.get("type") != "service_account": + raise ValueError("GOOGLE_SERVICE_ACCOUNT_JSON is not a service-account JSON object") + return value + + +def authorized_session(service_account_info: dict[str, Any]): + try: + from google.auth.transport.requests import AuthorizedSession + from google.oauth2 import service_account + except ImportError as exc: + raise RuntimeError("google-auth is required when Google reporting credentials are configured") from exc + + credentials = service_account.Credentials.from_service_account_info( + service_account_info, + scopes=( + "https://www.googleapis.com/auth/analytics.readonly", + "https://www.googleapis.com/auth/webmasters.readonly", + ), + ) + return AuthorizedSession(credentials) + + +def report_rows(payload: dict[str, Any]) -> list[dict[str, Any]]: + dimensions = [item.get("name", "") for item in payload.get("dimensionHeaders", [])] + metrics = [item.get("name", "") for item in payload.get("metricHeaders", [])] + rows: list[dict[str, Any]] = [] + for row in payload.get("rows", []): + result: dict[str, Any] = {} + for name, value in zip(dimensions, row.get("dimensionValues", []), strict=False): + result[name] = value.get("value", "") + for name, value in zip(metrics, row.get("metricValues", []), strict=False): + raw = value.get("value", "0") + try: + result[name] = float(raw) if "." in raw else int(raw) + except (TypeError, ValueError): + result[name] = raw + rows.append(result) + return rows + + +def ga4_report(session, property_id: str, body: dict[str, Any]) -> list[dict[str, Any]]: + response = session.post( + f"https://analyticsdata.googleapis.com/v1beta/properties/{property_id}:runReport", + json=body, + timeout=45, + ) + response.raise_for_status() + return report_rows(response.json()) + + +def read_page_languages(site: Path) -> dict[str, str]: + config = json.loads((site / "site.json").read_text(encoding="utf-8")) + result: dict[str, str] = {"index.html": "en"} + for item in config.get("pages", []): + path = item.get("path") or "index.html" + result[str(path)] = str(item.get("language", "en")) + return result + + +def normalize_site_path(value: str) -> str: + parsed = urllib.parse.urlsplit(value) + path = parsed.path + prefix = "/arsas/" + if path == "/arsas" or path == prefix: + return "index.html" + if path.startswith(prefix): + path = path[len(prefix):] + return path.lstrip("/") or "index.html" + + +def language_totals(rows: list[dict[str, Any]], page_languages: dict[str, str], metric: str) -> dict[str, float]: + totals: dict[str, float] = defaultdict(float) + for row in rows: + path = normalize_site_path(str(row.get("pagePath", row.get("page", "")))) + language = page_languages.get(path, "unknown") + totals[language] += float(row.get(metric, 0) or 0) + return dict(sorted(totals.items())) + + +def collect_ga4(session, property_id: str, days: int, page_languages: dict[str, str]) -> dict[str, Any]: + date_range = [{"startDate": f"{days}daysAgo", "endDate": "yesterday"}] + pages = ga4_report(session, property_id, { + "dateRanges": date_range, + "dimensions": [{"name": "pagePath"}, {"name": "pageTitle"}], + "metrics": [{"name": "screenPageViews"}, {"name": "activeUsers"}], + "orderBys": [{"metric": {"metricName": "screenPageViews"}, "desc": True}], + "limit": "10000", + }) + downloads = ga4_report(session, property_id, { + "dateRanges": date_range, + "dimensions": [{"name": "eventName"}, {"name": "pagePath"}], + "metrics": [{"name": "eventCount"}], + "dimensionFilter": { + "filter": { + "fieldName": "eventName", + "stringFilter": { + "matchType": "FULL_REGEXP", + "value": "download_(installer|portable|checksums)", + }, + } + }, + "orderBys": [{"metric": {"metricName": "eventCount"}, "desc": True}], + "limit": "1000", + }) + not_found = ga4_report(session, property_id, { + "dateRanges": date_range, + "dimensions": [{"name": "pagePath"}, {"name": "pageReferrer"}], + "metrics": [{"name": "eventCount"}], + "dimensionFilter": { + "filter": { + "fieldName": "eventName", + "stringFilter": {"matchType": "EXACT", "value": "page_not_found"}, + } + }, + "orderBys": [{"metric": {"metricName": "eventCount"}, "desc": True}], + "limit": "500", + }) + download_totals: dict[str, int] = defaultdict(int) + for row in downloads: + download_totals[str(row.get("eventName", "unknown"))] += int(row.get("eventCount", 0) or 0) + return { + "status": "available", + "topPages": pages[:50], + "languageTraffic": language_totals(pages, page_languages, "screenPageViews"), + "downloadEvents": downloads, + "downloadTotals": dict(sorted(download_totals.items())), + "notFound": not_found, + } + + +def collect_search_console(session, site_url: str, days: int, page_languages: dict[str, str]) -> dict[str, Any]: + end = date.today() - timedelta(days=3) + start = end - timedelta(days=days - 1) + encoded_site = urllib.parse.quote(site_url, safe="") + response = session.post( + f"https://searchconsole.googleapis.com/webmasters/v3/sites/{encoded_site}/searchAnalytics/query", + json={ + "startDate": start.isoformat(), + "endDate": end.isoformat(), + "dimensions": ["query", "page"], + "rowLimit": 25000, + "dataState": "final", + }, + timeout=60, + ) + response.raise_for_status() + rows = response.json().get("rows", []) + + query_totals: dict[str, dict[str, float]] = defaultdict(lambda: {"clicks": 0, "impressions": 0, "positionWeighted": 0}) + page_totals: dict[str, dict[str, float]] = defaultdict(lambda: {"clicks": 0, "impressions": 0, "positionWeighted": 0}) + detailed: list[dict[str, Any]] = [] + language_impressions: dict[str, float] = defaultdict(float) + language_clicks: dict[str, float] = defaultdict(float) + + for row in rows: + keys = row.get("keys", ["", ""]) + query = str(keys[0] if len(keys) > 0 else "") + page = str(keys[1] if len(keys) > 1 else "") + clicks = float(row.get("clicks", 0) or 0) + impressions = float(row.get("impressions", 0) or 0) + ctr = float(row.get("ctr", 0) or 0) + position = float(row.get("position", 0) or 0) + detailed.append({ + "query": query, + "page": page, + "clicks": clicks, + "impressions": impressions, + "ctr": ctr, + "position": position, + }) + for key, bucket in ((query, query_totals), (page, page_totals)): + bucket[key]["clicks"] += clicks + bucket[key]["impressions"] += impressions + bucket[key]["positionWeighted"] += position * impressions + language = page_languages.get(normalize_site_path(page), "unknown") + language_impressions[language] += impressions + language_clicks[language] += clicks + + def finish(values: dict[str, dict[str, float]], label: str) -> list[dict[str, Any]]: + output: list[dict[str, Any]] = [] + for key, value in values.items(): + impressions = value["impressions"] + clicks = value["clicks"] + output.append({ + label: key, + "clicks": clicks, + "impressions": impressions, + "ctr": clicks / impressions if impressions else 0, + "position": value["positionWeighted"] / impressions if impressions else 0, + }) + return output + + queries = sorted(finish(query_totals, "query"), key=lambda item: (item["clicks"], item["impressions"]), reverse=True) + pages = sorted(finish(page_totals, "page"), key=lambda item: (item["clicks"], item["impressions"]), reverse=True) + query_opportunities = sorted( + [item for item in queries if item["impressions"] >= 50 and item["ctr"] < 0.03 and item["position"] <= 20], + key=lambda item: item["impressions"], + reverse=True, + ) + page_opportunities = sorted( + [item for item in pages if item["impressions"] >= 100 and item["ctr"] < 0.03 and item["position"] <= 20], + key=lambda item: item["impressions"], + reverse=True, + ) + return { + "status": "available", + "siteUrl": site_url, + "dateRange": {"start": start.isoformat(), "end": end.isoformat()}, + "topQueries": queries[:50], + "topPages": pages[:50], + "lowCtrQueries": query_opportunities[:50], + "lowCtrPages": page_opportunities[:50], + "languageImpressions": dict(sorted(language_impressions.items())), + "languageClicks": dict(sorted(language_clicks.items())), + "rowCount": len(detailed), + } + + +def metric_value(metrics: dict[str, Any], names: tuple[str, ...], scale: float = 1.0) -> dict[str, Any] | None: + for name in names: + item = metrics.get(name) + if isinstance(item, dict) and item.get("percentile") is not None: + return { + "value": float(item["percentile"]) / scale, + "category": str(item.get("category", "UNKNOWN")).lower(), + } + return None + + +def collect_pagespeed(urls: tuple[str, ...], api_key: str) -> dict[str, Any]: + import requests + + results: list[dict[str, Any]] = [] + for url in urls: + params = [("url", url), ("strategy", "mobile"), ("category", "performance")] + if api_key: + params.append(("key", api_key)) + try: + response = requests.get( + "https://www.googleapis.com/pagespeedonline/v5/runPagespeed", + params=params, + timeout=90, + headers={"User-Agent": "ARSAS-Measurement/1.0"}, + ) + response.raise_for_status() + payload = response.json() + field = payload.get("loadingExperience", {}) + metrics = field.get("metrics", {}) if isinstance(field, dict) else {} + audits = payload.get("lighthouseResult", {}).get("audits", {}) + results.append({ + "url": url, + "status": "available", + "fieldCategory": str(field.get("overall_category", "NONE")).lower(), + "field": { + "lcpMs": metric_value(metrics, ("LARGEST_CONTENTFUL_PAINT_MS",)), + "cls": metric_value(metrics, ("CUMULATIVE_LAYOUT_SHIFT_SCORE",), 100.0), + "inpMs": metric_value(metrics, ("INTERACTION_TO_NEXT_PAINT", "INTERACTION_TO_NEXT_PAINT_MS")), + }, + "lab": { + "performanceScore": payload.get("lighthouseResult", {}).get("categories", {}).get("performance", {}).get("score"), + "lcpMs": audits.get("largest-contentful-paint", {}).get("numericValue"), + "cls": audits.get("cumulative-layout-shift", {}).get("numericValue"), + "totalBlockingTimeMs": audits.get("total-blocking-time", {}).get("numericValue"), + }, + }) + except Exception as exc: # network/API errors should not erase other measurement sources + results.append({"url": url, "status": "unavailable", "error": str(exc)}) + return {"status": "available" if any(item["status"] == "available" for item in results) else "unavailable", "pages": results} + + +def percent(value: float) -> str: + return f"{value * 100:.2f}%" + + +def table(headers: tuple[str, ...], rows: list[tuple[Any, ...]]) -> list[str]: + if not rows: + return ["No data available.", ""] + lines = ["| " + " | ".join(headers) + " |", "|" + "|".join("---" for _ in headers) + "|"] + for row in rows: + lines.append("| " + " | ".join(str(value).replace("|", "\\|") for value in row) + " |") + lines.append("") + return lines + + +def build_markdown(report: dict[str, Any]) -> str: + ga4 = report["ga4"] + search = report["searchConsole"] + speed = report["coreWebVitals"] + lines = [ + "# ARSAS website measurement", + "", + f"Generated: `{report['generatedAtUtc']}` · Window: **{report['days']} days**", + "", + "## Data coverage", + "", + f"- GA4 aggregated traffic/events: **{ga4['status']}**", + f"- Search Console queries/impressions/CTR: **{search['status']}**", + f"- CrUX/PageSpeed Core Web Vitals: **{speed['status']}**", + "- Broken links and deployed 404 behavior: see the paired `site-health.md` artifact.", + "", + ] + + lines.extend(["## Most visited pages", ""]) + lines.extend(table( + ("Page", "Views", "Active users"), + [(item.get("pagePath", ""), item.get("screenPageViews", 0), item.get("activeUsers", 0)) for item in ga4.get("topPages", [])[:15]], + )) + + lines.extend(["## English vs Indonesian traffic", ""]) + language_rows = [] + for language in sorted(set(ga4.get("languageTraffic", {})) | set(search.get("languageImpressions", {}))): + language_rows.append(( + language, + int(ga4.get("languageTraffic", {}).get(language, 0)), + int(search.get("languageImpressions", {}).get(language, 0)), + int(search.get("languageClicks", {}).get(language, 0)), + )) + lines.extend(table(("Language", "Page views", "Search impressions", "Search clicks"), language_rows)) + + lines.extend(["## Download button clicks", ""]) + lines.extend(table( + ("Package event", "Clicks"), + [(name, count) for name, count in ga4.get("downloadTotals", {}).items()], + )) + + lines.extend(["## Search queries bringing users", ""]) + lines.extend(table( + ("Query", "Clicks", "Impressions", "CTR", "Avg position"), + [(item["query"], int(item["clicks"]), int(item["impressions"]), percent(item["ctr"]), f"{item['position']:.1f}") for item in search.get("topQueries", [])[:20]], + )) + + lines.extend(["## High-impression pages with low click-through", ""]) + lines.extend(table( + ("Page", "Clicks", "Impressions", "CTR", "Avg position"), + [(item["page"], int(item["clicks"]), int(item["impressions"]), percent(item["ctr"]), f"{item['position']:.1f}") for item in search.get("lowCtrPages", [])[:20]], + )) + + lines.extend(["## High-impression queries with low click-through", ""]) + lines.extend(table( + ("Query", "Clicks", "Impressions", "CTR", "Avg position"), + [(item["query"], int(item["clicks"]), int(item["impressions"]), percent(item["ctr"]), f"{item['position']:.1f}") for item in search.get("lowCtrQueries", [])[:20]], + )) + + lines.extend(["## 404 observations", ""]) + lines.extend(table( + ("Requested path", "Referrer", "Events"), + [(item.get("pagePath", ""), item.get("pageReferrer", ""), item.get("eventCount", 0)) for item in ga4.get("notFound", [])[:20]], + )) + + lines.extend(["## Core Web Vitals", ""]) + vital_rows = [] + for item in speed.get("pages", []): + field = item.get("field", {}) + vital_rows.append(( + item.get("url", ""), + item.get("fieldCategory", item.get("status", "")), + (field.get("lcpMs") or {}).get("value", "—"), + (field.get("cls") or {}).get("value", "—"), + (field.get("inpMs") or {}).get("value", "—"), + item.get("lab", {}).get("performanceScore", "—"), + )) + lines.extend(table(("URL", "Field status", "LCP ms", "CLS", "INP ms", "Lab score"), vital_rows)) + + lines.extend(["## Continuous-improvement queue", ""]) + actions: list[str] = [] + for item in search.get("lowCtrPages", [])[:5]: + actions.append(f"Rewrite title/meta and align intent for `{item['page']}` ({int(item['impressions'])} impressions, {percent(item['ctr'])} CTR).") + for item in speed.get("pages", []): + if item.get("fieldCategory") in {"slow", "poor"}: + actions.append(f"Prioritize Core Web Vitals remediation for `{item['url']}`.") + if ga4.get("notFound"): + actions.append("Add redirects or repair inbound links for the highest-frequency 404 paths.") + if ga4.get("status") != "available": + actions.append("Configure GA4 repository variables and read-only service-account access to activate traffic and download reporting.") + if search.get("status") != "available": + actions.append("Grant the service account read access to the verified Search Console property to activate query and CTR reporting.") + if not actions: + actions.append("No immediate threshold breach was detected; continue the weekly measurement cycle.") + lines.extend(f"- {item}" for item in actions) + lines.append("") + lines.append("Reports contain aggregated product-site metrics only and are kept in GitHub Actions artifacts rather than deployed publicly.") + lines.append("") + return "\n".join(lines) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--site", default="_site") + parser.add_argument("--output", default="_measurement") + parser.add_argument("--days", type=int, default=28) + args = parser.parse_args() + if not 7 <= args.days <= 90: + raise SystemExit("Measurement window must be between 7 and 90 days") + + site = Path(args.site).resolve() + output = Path(args.output).resolve() + output.mkdir(parents=True, exist_ok=True) + page_languages = read_page_languages(site) + report: dict[str, Any] = { + "schemaVersion": 1, + "generatedAtUtc": datetime.now(timezone.utc).isoformat(), + "days": args.days, + "ga4": {"status": "not-configured", "topPages": [], "languageTraffic": {}, "downloadTotals": {}, "notFound": []}, + "searchConsole": {"status": "not-configured", "topQueries": [], "topPages": [], "lowCtrQueries": [], "lowCtrPages": [], "languageImpressions": {}, "languageClicks": {}}, + "coreWebVitals": {"status": "not-run", "pages": []}, + } + + service_info = None + try: + service_info = load_service_account() + except Exception as exc: + report["credentialError"] = str(exc) + + property_id = os.environ.get("GA4_PROPERTY_ID", "").strip() + search_site = os.environ.get("GSC_SITE_URL", CANONICAL_ROOT).strip() + if service_info: + try: + session = authorized_session(service_info) + if property_id.isdigit(): + try: + report["ga4"] = collect_ga4(session, property_id, args.days, page_languages) + except Exception as exc: + report["ga4"] = {**report["ga4"], "status": "unavailable", "error": str(exc)} + elif property_id: + report["ga4"] = {**report["ga4"], "status": "invalid-property-id"} + try: + report["searchConsole"] = collect_search_console(session, search_site, args.days, page_languages) + except Exception as exc: + report["searchConsole"] = {**report["searchConsole"], "status": "unavailable", "error": str(exc)} + except Exception as exc: + report["googleSessionError"] = str(exc) + + psi_urls = tuple( + item.strip() for item in os.environ.get("PAGESPEED_URLS", ",".join(DEFAULT_PSI_URLS)).split(",") if item.strip() + ) + try: + report["coreWebVitals"] = collect_pagespeed(psi_urls, os.environ.get("PAGESPEED_API_KEY", "").strip()) + except Exception as exc: + report["coreWebVitals"] = {"status": "unavailable", "pages": [], "error": str(exc)} + + markdown = build_markdown(report) + (output / "measurement.json").write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n", encoding="utf-8") + (output / "measurement.md").write_text(markdown, encoding="utf-8") + + summary = os.environ.get("GITHUB_STEP_SUMMARY") + if summary: + with open(summary, "a", encoding="utf-8") as handle: + handle.write(markdown) + print( + "ARSAS measurement report generated: " + f"GA4={report['ga4']['status']}, Search Console={report['searchConsole']['status']}, " + f"Core Web Vitals={report['coreWebVitals']['status']}." + ) + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except Exception as exc: + print(f"ARSAS measurement report failed: {exc}", file=sys.stderr) + raise From 56234c6d91031dcf301fe06321bbe801b9b01965 Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:54:29 +0700 Subject: [PATCH 09/15] Add scheduled site measurement and health workflow --- .github/workflows/site-measurement.yml | 187 +++++++++++++++++++++++++ 1 file changed, 187 insertions(+) create mode 100644 .github/workflows/site-measurement.yml diff --git a/.github/workflows/site-measurement.yml b/.github/workflows/site-measurement.yml new file mode 100644 index 000000000..6f0a5aa04 --- /dev/null +++ b/.github/workflows/site-measurement.yml @@ -0,0 +1,187 @@ +name: Measure product website + +on: + schedule: + - cron: "17 3 * * 1" + workflow_dispatch: + inputs: + days: + description: Aggregated reporting window in days + required: false + default: "28" + type: choice + options: ["7", "28", "60", "90"] + pull_request: + branches: [ main ] + paths: + - "landing/**" + - "scripts/build-product-site.py" + - "scripts/inject-site-measurement.py" + - "scripts/validate-site-measurement.py" + - "scripts/check-site-health.py" + - "scripts/build-site-measurement-report.py" + - ".github/workflows/site-measurement.yml" + - ".github/workflows/pages.yml" + push: + branches: [ main ] + paths: + - "landing/**" + - "scripts/build-product-site.py" + - "scripts/inject-site-measurement.py" + - "scripts/validate-site-measurement.py" + - "scripts/check-site-health.py" + - "scripts/build-site-measurement-report.py" + - ".github/workflows/site-measurement.yml" + - ".github/workflows/pages.yml" + +permissions: + contents: read + +concurrency: + group: site-measurement-${{ github.ref }} + cancel-in-progress: true + +env: + CANONICAL_ROOT: https://masarray.github.io/arsas/ + +jobs: + quality: + name: Validate measurement and internal links + runs-on: ubuntu-latest + steps: + - name: Checkout + uses: actions/checkout@v4 + with: + show-progress: false + + - name: Setup Python + uses: actions/setup-python@v5 + with: + python-version: "3.12" + + - name: Prepare stable release evidence + env: + GH_TOKEN: ${{ github.token }} + shell: bash + run: | + set -euo pipefail + if [ "$GITHUB_EVENT_NAME" = "pull_request" ]; then + cp landing/latest.json /tmp/arsas-published.json + else + gh api "repos/$GITHUB_REPOSITORY/contents/published.json?ref=release-evidence" --jq .content | base64 -d > /tmp/arsas-published.json + fi + + - name: Build deterministic website + run: python scripts/build-product-site.py --output _site --release-evidence /tmp/arsas-published.json + + - name: Configure optional client measurement + env: + GA4_MEASUREMENT_ID: ${{ vars.GA4_MEASUREMENT_ID }} + run: python scripts/inject-site-measurement.py _site --measurement-id "$GA4_MEASUREMENT_ID" + + - name: Validate page, download, language, 404 and Web Vitals measurement + env: + GA4_MEASUREMENT_ID: ${{ vars.GA4_MEASUREMENT_ID }} + run: python scripts/validate-site-measurement.py _site --measurement-id "$GA4_MEASUREMENT_ID" + + - name: Check internal links and fragments + id: local_health + continue-on-error: true + run: python scripts/check-site-health.py --site _site --output _measurement + + - name: Add local health summary + if: always() + shell: bash + run: | + if [ -f _measurement/site-health.md ]; then + cat _measurement/site-health.md >> "$GITHUB_STEP_SUMMARY" + fi + + - name: Upload quality evidence + if: always() + uses: actions/upload-artifact@v4 + with: + name: site-measurement-quality + path: _measurement/ + if-no-files-found: warn + retention-days: 30 + + - name: Enforce link health + if: steps.local_health.outcome != 'success' + run: exit 1 + + insights: + name: Build private traffic, search and Web Vitals report + if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' + needs: quality + runs-on: ubuntu-latest + steps: + - name: Checkout + uses: actions/checkout@v4 + with: + show-progress: false + + - name: Setup Python + uses: actions/setup-python@v5 + with: + python-version: "3.12" + + - name: Install read-only reporting dependencies + run: python -m pip install --disable-pip-version-check --quiet google-auth requests + + - name: Prepare stable release evidence + env: + GH_TOKEN: ${{ github.token }} + shell: bash + run: gh api "repos/$GITHUB_REPOSITORY/contents/published.json?ref=release-evidence" --jq .content | base64 -d > /tmp/arsas-published.json + + - name: Build current website + run: python scripts/build-product-site.py --output _site --release-evidence /tmp/arsas-published.json + + - name: Configure current client measurement + env: + GA4_MEASUREMENT_ID: ${{ vars.GA4_MEASUREMENT_ID }} + run: python scripts/inject-site-measurement.py _site --measurement-id "$GA4_MEASUREMENT_ID" + + - name: Check deployed pages, release assets and 404 response + id: deployed_health + continue-on-error: true + run: python scripts/check-site-health.py --site _site --output _measurement --remote --base-url "$CANONICAL_ROOT" + + - name: Build aggregated measurement report + id: measurement + continue-on-error: true + env: + GOOGLE_SERVICE_ACCOUNT_JSON: ${{ secrets.GOOGLE_SERVICE_ACCOUNT_JSON }} + GA4_PROPERTY_ID: ${{ vars.GA4_PROPERTY_ID }} + GSC_SITE_URL: ${{ vars.GSC_SITE_URL }} + PAGESPEED_API_KEY: ${{ secrets.PAGESPEED_API_KEY }} + PAGESPEED_URLS: ${{ vars.PAGESPEED_URLS }} + REPORT_DAYS: ${{ inputs.days || '28' }} + shell: bash + run: python scripts/build-site-measurement-report.py --site _site --output _measurement --days "$REPORT_DAYS" + + - name: Add deployed health summary + if: always() + shell: bash + run: | + if [ -f _measurement/site-health.md ]; then + cat _measurement/site-health.md >> "$GITHUB_STEP_SUMMARY" + fi + + - name: Upload private measurement evidence + if: always() + uses: actions/upload-artifact@v4 + with: + name: site-measurement-${{ github.run_number }} + path: _measurement/ + if-no-files-found: error + retention-days: 90 + + - name: Enforce measurement execution + if: steps.measurement.outcome != 'success' + run: exit 1 + + - name: Enforce deployed link and 404 health + if: steps.deployed_health.outcome != 'success' + run: exit 1 From faba6f7a25cb08fa3f2d4bfafaea51c078fa9cd0 Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:55:07 +0700 Subject: [PATCH 10/15] Enforce measurement and link health before Pages deploy --- .github/workflows/pages.yml | 24 +++++++++++++++++++++++- 1 file changed, 23 insertions(+), 1 deletion(-) diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index 3e2b28b6b..5002eb2e5 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -9,6 +9,9 @@ on: - "scripts/build-product-site.py" - "scripts/validate-product-build.py" - "scripts/submit-indexnow.py" + - "scripts/inject-site-measurement.py" + - "scripts/validate-site-measurement.py" + - "scripts/check-site-health.py" - ".github/workflows/pages.yml" pull_request: branches: [ main ] @@ -18,6 +21,9 @@ on: - "scripts/build-product-site.py" - "scripts/validate-product-build.py" - "scripts/submit-indexnow.py" + - "scripts/inject-site-measurement.py" + - "scripts/validate-site-measurement.py" + - "scripts/check-site-health.py" - ".github/workflows/pages.yml" workflow_dispatch: @@ -103,6 +109,19 @@ jobs: - name: Build deterministic product website run: python scripts/build-product-site.py --output _site --release-evidence /tmp/arsas-published.json + - name: Configure optional client measurement + env: + GA4_MEASUREMENT_ID: ${{ vars.GA4_MEASUREMENT_ID }} + run: python scripts/inject-site-measurement.py _site --measurement-id "$GA4_MEASUREMENT_ID" + + - name: Validate measurement contract + env: + GA4_MEASUREMENT_ID: ${{ vars.GA4_MEASUREMENT_ID }} + run: python scripts/validate-site-measurement.py _site --measurement-id "$GA4_MEASUREMENT_ID" + + - name: Check internal links and fragments + run: python scripts/check-site-health.py --site _site --output _validation/site-health + - name: Validate IndexNow payload without network submission run: python scripts/submit-indexnow.py --sitemap _site/sitemap.xml --dry-run @@ -119,7 +138,9 @@ jobs: uses: actions/upload-artifact@v4 with: name: product-build-validation - path: _validation/build.log + path: | + _validation/build.log + _validation/site-health/ if-no-files-found: warn - name: Enforce rendered validation @@ -136,6 +157,7 @@ jobs: ! grep -R --line-number --fixed-strings 'LICENSE-APACHE-2.0' _site ! grep -R --line-number -E '(href|src|content)="http://' _site --include='*.html' ! grep -R --line-number --fixed-strings 'raw.githubusercontent.com/masarray/arsas/main/Assets/screenshot' _site --include='*.html' + ! grep -R --line-number --fixed-strings '__ARSAS_GA4_MEASUREMENT_ID__' _site --include='*.html' - name: Upload website artifact if: github.event_name != 'pull_request' From ef2724d2705fbe4b25ad58a5d68db1f9dc3dcd62 Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:55:56 +0700 Subject: [PATCH 11/15] Document website measurement operations --- docs/website-measurement.md | 87 +++++++++++++++++++++++++++++++++++++ 1 file changed, 87 insertions(+) create mode 100644 docs/website-measurement.md diff --git a/docs/website-measurement.md b/docs/website-measurement.md new file mode 100644 index 000000000..66200f92e --- /dev/null +++ b/docs/website-measurement.md @@ -0,0 +1,87 @@ +# ARSAS website measurement + +ARSAS uses a lightweight, evidence-oriented measurement pipeline. Runtime analytics are optional and remain disabled when no valid measurement ID is configured. Search and performance reports are private GitHub Actions artifacts; they are not deployed to the public website. + +## What is measured + +| Question | Source | Output | +|---|---|---| +| Which pages are visited most? | GA4 `page_view` | Page path, views and active users | +| Which queries bring users? | Google Search Console | Query, clicks, impressions, CTR and average position | +| Which download buttons are clicked? | GA4 events | `download_installer`, `download_portable`, `download_checksums` | +| English vs Indonesian traffic | Page registry + GA4/Search Console | Views, search impressions and clicks by site language | +| Which pages have high impressions but low clicks? | Search Console | Pages and queries above the configured opportunity thresholds | +| Are links broken or 404s occurring? | Build-time crawler + deployed probe + GA4 | Missing files/fragments, HTTP failures, 404 paths and referrers | +| Are Core Web Vitals healthy? | Browser PerformanceObserver + PageSpeed/CrUX | LCP, CLS and INP field/lab evidence | + +The browser client disables Google advertising signals, does not request ad-personalization signals, respects `Do Not Track`, loads asynchronously and performs no network request when the measurement ID is absent. + +## Repository configuration + +Configure these **Actions variables**: + +- `GA4_MEASUREMENT_ID`: public web stream ID such as `G-XXXXXXXXXX`. Leaving it empty keeps client measurement disabled. +- `GA4_PROPERTY_ID`: numeric GA4 property ID used by the private reporting workflow. +- `GSC_SITE_URL`: the exact verified Search Console property, normally `https://masarray.github.io/arsas/` for a URL-prefix property. +- `PAGESPEED_URLS`: optional comma-separated URLs. When omitted, the workflow checks the English and Indonesian home/download pages plus Smart Reporting and Guides. + +Configure these **Actions secrets**: + +- `GOOGLE_SERVICE_ACCOUNT_JSON`: JSON for a service account with read-only access to the GA4 property and Search Console property. A `base64:` value is also accepted. +- `PAGESPEED_API_KEY`: optional PageSpeed Insights API key. The report still attempts the public endpoint when this secret is absent. + +Grant the service account Viewer/read access only. It does not need permission to modify analytics, Search Console, releases or the website. + +## Workflow behavior + +`.github/workflows/site-measurement.yml` runs: + +- on relevant pull requests and pushes: deterministic build, measurement contract validation, internal links and fragment validation; +- every Monday at 03:17 UTC, or manually: deployed page checks, official release-asset checks, an intentional 404 probe, GA4 aggregate reports, Search Console reports and PageSpeed/CrUX collection. + +Artifacts: + +- `site-measurement-quality`: local link and instrumentation evidence, retained for 30 days; +- `site-measurement-`: private Markdown/JSON traffic, search, 404 and Core Web Vitals evidence, retained for 90 days. + +The same Markdown report is written to the GitHub Actions job summary. + +## Opportunity rules + +The initial low-CTR queue is deliberately conservative: + +- query opportunity: at least 50 impressions, CTR below 3%, average position 20 or better; +- page opportunity: at least 100 impressions, CTR below 3%, average position 20 or better. + +These thresholds are implemented in `scripts/build-site-measurement-report.py` and can be adjusted after several reporting cycles establish a stable baseline. + +## Event contract + +The local `landing/analytics.js` client emits: + +- `page_view`; +- `page_not_found`; +- `language_switch`; +- `download_installer`; +- `download_portable`; +- `download_checksums`; +- `web_vital_lcp`; +- `web_vital_cls`; +- `web_vital_inp`; +- diagnostic `web_vital_ttfb`. + +Every event carries page path, page title, site language, content group and stable release version. Download events also carry the official file name, destination URL and visible link text. + +## Interpreting Core Web Vitals + +Browser RUM events provide continuous observations from measured visits. The scheduled PageSpeed report remains the decision source for field CWV because it uses CrUX data when enough real-user samples exist. When CrUX has insufficient traffic, the report retains Lighthouse lab values and marks field data unavailable instead of inventing a pass/fail result. + +## Continuous-improvement loop + +1. Review the weekly job summary. +2. Repair any broken internal link or failed 404 behavior immediately. +3. Prioritize high-impression pages with low CTR for title, description and intent alignment. +4. Compare English and Indonesian traffic before deciding which translations to expand. +5. Trace download clicks back to the page that generated them. +6. Investigate repeated 404 paths and add a valid route or redirect where appropriate. +7. Treat poor LCP, CLS or INP as a release-quality issue, then confirm the improvement in the next field-data cycle. From d4333ea4012327d99390b490742e3cd3c56eadf9 Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:56:41 +0700 Subject: [PATCH 12/15] Fix rendered measurement contract checks --- scripts/validate-site-measurement.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/scripts/validate-site-measurement.py b/scripts/validate-site-measurement.py index 2317aa359..6c7659667 100644 --- a/scripts/validate-site-measurement.py +++ b/scripts/validate-site-measurement.py @@ -51,8 +51,8 @@ def main() -> int: client_text = client.read_text(encoding="utf-8") for required in ( "download_installer", "download_portable", "download_checksums", - "page_not_found", "language_switch", "web_vital_lcp", "web_vital_cls", - "web_vital_inp", "navigator.doNotTrack", "allow_google_signals: false", + "page_not_found", "language_switch", "reportVital('LCP'", "reportVital('CLS'", + "reportVital('INP'", "navigator.doNotTrack", "allow_google_signals: false", "allow_ad_personalization_signals: false", ): if required not in client_text: @@ -72,7 +72,7 @@ def main() -> int: errors.append(f"{label}: expected one shared analytics client") continue script = parsed.analytics[0] - if script.get("src") != "analytics.js" or script.get("defer") is None: + if script.get("src") != "analytics.js" or "defer" not in script: errors.append(f"{label}: analytics client must be local and deferred") actual_id = script.get("data-measurement-id") or "" if actual_id != expected_id: From a9e9b9b6a68e9370a01e3f8941b5ef25786ac954 Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:59:02 +0700 Subject: [PATCH 13/15] Scope measurement injection to registered product pages --- scripts/inject-site-measurement.py | 42 +++++++++++++++++------------- 1 file changed, 24 insertions(+), 18 deletions(-) diff --git a/scripts/inject-site-measurement.py b/scripts/inject-site-measurement.py index 1f8432fe3..d88a37476 100644 --- a/scripts/inject-site-measurement.py +++ b/scripts/inject-site-measurement.py @@ -22,9 +22,18 @@ def configure(site: Path, measurement_id: str) -> None: if not (site / "analytics.js").is_file(): raise SystemExit("Built site is missing analytics.js") - pages = sorted(site.rglob("*.html")) - if not pages: - raise SystemExit("Built site contains no HTML pages") + build_info_path = site / "build-info.json" + if not build_info_path.is_file(): + raise SystemExit("Built site is missing build-info.json") + build_info = json.loads(build_info_path.read_text(encoding="utf-8")) + registered = build_info.get("pages") + if not isinstance(registered, list) or not registered or not all(isinstance(item, str) for item in registered): + raise SystemExit("build-info.json has an invalid page registry") + + pages = [site / item for item in registered] + missing = [str(page.relative_to(site)) for page in pages if not page.is_file()] + if missing: + raise SystemExit("Registered measurement pages are missing: " + ", ".join(missing)) replacements = 0 for page in pages: @@ -37,23 +46,20 @@ def configure(site: Path, measurement_id: str) -> None: page.write_text(text.replace(PLACEHOLDER, measurement_id), encoding="utf-8") replacements += count - build_info_path = site / "build-info.json" - if build_info_path.exists(): - build_info = json.loads(build_info_path.read_text(encoding="utf-8")) - build_info["measurement"] = { - "provider": "google-analytics-4", - "enabled": bool(measurement_id), - "client": "analytics.js", - "doNotTrackRespected": True, - "advertisingSignals": False, - } - build_info_path.write_text( - json.dumps(build_info, indent=2, ensure_ascii=False) + "\n", - encoding="utf-8", - ) + build_info["measurement"] = { + "provider": "google-analytics-4", + "enabled": bool(measurement_id), + "client": "analytics.js", + "doNotTrackRespected": True, + "advertisingSignals": False, + } + build_info_path.write_text( + json.dumps(build_info, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8", + ) state = "enabled" if measurement_id else "disabled" - print(f"ARSAS site measurement {state}: {replacements} pages configured.") + print(f"ARSAS site measurement {state}: {replacements} registered pages configured.") def main() -> int: From 38e43eee486695ec6553a3c7203ef093e6698fe6 Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 16:59:35 +0700 Subject: [PATCH 14/15] Validate only registered product pages --- scripts/validate-site-measurement.py | 49 ++++++++++++++++------------ 1 file changed, 29 insertions(+), 20 deletions(-) diff --git a/scripts/validate-site-measurement.py b/scripts/validate-site-measurement.py index 6c7659667..f63609a25 100644 --- a/scripts/validate-site-measurement.py +++ b/scripts/validate-site-measurement.py @@ -58,10 +58,24 @@ def main() -> int: if required not in client_text: errors.append(f"analytics.js missing measurement contract: {required}") - pages = sorted(site.rglob("*.html")) - if not pages: - errors.append("rendered site has no HTML pages") + build_info_path = site / "build-info.json" + build_info: dict[str, object] = {} + registered: list[str] = [] + if not build_info_path.exists(): + errors.append("build-info.json is missing") + else: + build_info = json.loads(build_info_path.read_text(encoding="utf-8")) + raw_pages = build_info.get("pages") + if not isinstance(raw_pages, list) or not raw_pages or not all(isinstance(item, str) for item in raw_pages): + errors.append("build-info.json has an invalid page registry") + else: + registered = list(raw_pages) + + pages = [site / item for item in registered] for page in pages: + if not page.exists(): + errors.append(f"registered page is missing: {page.relative_to(site)}") + continue text = page.read_text(encoding="utf-8") label = page.relative_to(site) if PLACEHOLDER in text: @@ -85,23 +99,18 @@ def main() -> int: if page.name == "404.html" and parsed.body_page != "none": errors.append("404.html must use data-page=none for page_not_found measurement") - build_info_path = site / "build-info.json" - if not build_info_path.exists(): - errors.append("build-info.json is missing") + measurement = build_info.get("measurement") + if not isinstance(measurement, dict): + errors.append("build-info.json is missing measurement status") else: - build_info = json.loads(build_info_path.read_text(encoding="utf-8")) - measurement = build_info.get("measurement") - if not isinstance(measurement, dict): - errors.append("build-info.json is missing measurement status") - else: - if measurement.get("provider") != "google-analytics-4": - errors.append("build-info.json has invalid measurement provider") - if measurement.get("enabled") is not bool(expected_id): - errors.append("build-info.json measurement enabled state is incorrect") - if measurement.get("doNotTrackRespected") is not True: - errors.append("build-info.json must declare Do Not Track handling") - if measurement.get("advertisingSignals") is not False: - errors.append("build-info.json must declare advertising signals disabled") + if measurement.get("provider") != "google-analytics-4": + errors.append("build-info.json has invalid measurement provider") + if measurement.get("enabled") is not bool(expected_id): + errors.append("build-info.json measurement enabled state is incorrect") + if measurement.get("doNotTrackRespected") is not True: + errors.append("build-info.json must declare Do Not Track handling") + if measurement.get("advertisingSignals") is not False: + errors.append("build-info.json must declare advertising signals disabled") errors = list(dict.fromkeys(errors)) if errors: @@ -110,7 +119,7 @@ def main() -> int: print(f"- {error}", file=sys.stderr) return 1 state = "enabled" if expected_id else "disabled/no-op" - print(f"ARSAS site-measurement validation passed: {len(pages)} pages, client {state}, downloads, language, 404 and Core Web Vitals contracts present.") + print(f"ARSAS site-measurement validation passed: {len(pages)} registered pages, client {state}, downloads, language, 404 and Core Web Vitals contracts present.") return 0 From cbea6deb1d669cabf3d098bd8df635e11375694a Mon Sep 17 00:00:00 2001 From: masarray Date: Mon, 20 Jul 2026 17:00:50 +0700 Subject: [PATCH 15/15] Exercise reporting engine during pull requests --- .github/workflows/site-measurement.yml | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/.github/workflows/site-measurement.yml b/.github/workflows/site-measurement.yml index 6f0a5aa04..b669b5ff5 100644 --- a/.github/workflows/site-measurement.yml +++ b/.github/workflows/site-measurement.yml @@ -59,6 +59,9 @@ jobs: with: python-version: "3.12" + - name: Install offline report dependency + run: python -m pip install --disable-pip-version-check --quiet requests + - name: Prepare stable release evidence env: GH_TOKEN: ${{ github.token }} @@ -89,6 +92,15 @@ jobs: continue-on-error: true run: python scripts/check-site-health.py --site _site --output _measurement + - name: Exercise reporting without credentials or network data + env: + GOOGLE_SERVICE_ACCOUNT_JSON: "" + GA4_PROPERTY_ID: "" + GSC_SITE_URL: https://masarray.github.io/arsas/ + PAGESPEED_URLS: "" + GITHUB_STEP_SUMMARY: "" + run: python scripts/build-site-measurement-report.py --site _site --output _measurement/offline --days 7 + - name: Add local health summary if: always() shell: bash @@ -154,9 +166,9 @@ jobs: env: GOOGLE_SERVICE_ACCOUNT_JSON: ${{ secrets.GOOGLE_SERVICE_ACCOUNT_JSON }} GA4_PROPERTY_ID: ${{ vars.GA4_PROPERTY_ID }} - GSC_SITE_URL: ${{ vars.GSC_SITE_URL }} + GSC_SITE_URL: ${{ vars.GSC_SITE_URL || 'https://masarray.github.io/arsas/' }} PAGESPEED_API_KEY: ${{ secrets.PAGESPEED_API_KEY }} - PAGESPEED_URLS: ${{ vars.PAGESPEED_URLS }} + PAGESPEED_URLS: ${{ vars.PAGESPEED_URLS || 'https://masarray.github.io/arsas/,https://masarray.github.io/arsas/download.html,https://masarray.github.io/arsas/smart-reporting.html,https://masarray.github.io/arsas/guides.html,https://masarray.github.io/arsas/id.html,https://masarray.github.io/arsas/unduh.html' }} REPORT_DAYS: ${{ inputs.days || '28' }} shell: bash run: python scripts/build-site-measurement-report.py --site _site --output _measurement --days "$REPORT_DAYS"