"""Owner-operated browser capture. Never exports cookies, storage or request headers. pip install -r examples/browser/requirements.txt python -m playwright install chromium python examples/browser/capture.py --url https://your-approved-site.example/page --output snapshot.json """ from __future__ import annotations import argparse import json import os import re from datetime import datetime, timezone from pathlib import Path from urllib.parse import parse_qsl, urlsplit, urlunsplit def capture_snapshot(page): url = page.url parsed = urlsplit(url) if parsed.scheme not in {"http", "https"} or not parsed.hostname or parsed.username or parsed.password: raise ValueError("Capture requires a credential-free HTTP(S) source URL") if any(re.fullmatch(r"(?i)(?:token|access_token|refresh_token|id_token|secret|password|session|signature|auth|api_key|key|code)", key) for key, _ in parse_qsl(parsed.query)): raise ValueError("Remove credential-like query parameters before capture") text = page.locator("body").inner_text(timeout=10000) if not text.strip() or len(text.encode()) > 12 * 1024 * 1024: raise ValueError("Visible text must contain 1 byte to 12 MiB") return {"format": "BROWSER", "title": page.title(), "source_url": urlunsplit(parsed._replace(fragment="")), "source_date": datetime.now(timezone.utc).isoformat(), "text": text, "origin": "OWNER_BROWSER_CAPTURE", "warning": "Review text for sensitive information before importing. This is not a server-fetched source."} def main(): from playwright.sync_api import sync_playwright parser = argparse.ArgumentParser(description="Capture visible text locally after owner sign-in; nothing is uploaded automatically") parser.add_argument("--url", required=True) parser.add_argument("--output", type=Path, required=True) parser.add_argument("--profile", type=Path, default=Path(".citedoor-browser")) args = parser.parse_args() parsed = urlsplit(args.url) if parsed.scheme not in {"http", "https"} or not parsed.hostname or parsed.username or parsed.password: parser.error("Use a credential-free HTTP(S) URL") args.profile.mkdir(parents=True, exist_ok=True, mode=0o700) with sync_playwright() as playwright: context = playwright.chromium.launch_persistent_context(str(args.profile.resolve()), headless=False) try: page = context.pages[0] if context.pages else context.new_page() page.goto(args.url, wait_until="domcontentloaded") print("Sign in and navigate locally. Check that you may share this page with your research campaign.") input("Press Enter when the visible page is ready to capture: ") snapshot = capture_snapshot(page) fd = os.open(args.output, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) with os.fdopen(fd, "w") as output: json.dump(snapshot, output, ensure_ascii=False, indent=2) print("Snapshot saved locally. Review it, then import it in CiteDoor → Evidence → Browser snapshot.") finally: context.close() if __name__ == "__main__": main()