feat(download-js-sourcemaps): reconstruct sources from source maps
This commit is contained in:
Executable
+194
@@ -0,0 +1,194 @@
|
|||||||
|
#!/usr/bin/env -S uv run --script
|
||||||
|
# /// script
|
||||||
|
# dependencies = ["playwright", "httpx"]
|
||||||
|
# ///
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Optional
|
||||||
|
from urllib.parse import urljoin, urlparse
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
from playwright.sync_api import sync_playwright
|
||||||
|
|
||||||
|
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s: %(message)s")
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def str_to_bool(value: str) -> bool:
|
||||||
|
"""Convert string to boolean."""
|
||||||
|
if isinstance(value, bool):
|
||||||
|
return value
|
||||||
|
return value.lower() in ("true", "1", "yes", "on")
|
||||||
|
|
||||||
|
|
||||||
|
def parse_args() -> argparse.Namespace:
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Download JavaScript files and reconstruct sources from source maps",
|
||||||
|
formatter_class=argparse.ArgumentDefaultsHelpFormatter,
|
||||||
|
)
|
||||||
|
parser.add_argument("url", help="URL of the page to scrape")
|
||||||
|
parser.add_argument(
|
||||||
|
"--output-dir",
|
||||||
|
"-o",
|
||||||
|
type=Path,
|
||||||
|
default=Path("output"),
|
||||||
|
help="Output directory for reconstructed sources",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--headless",
|
||||||
|
type=str_to_bool,
|
||||||
|
default=True,
|
||||||
|
help="Run browser in headless mode",
|
||||||
|
)
|
||||||
|
return parser.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def get_script_urls(page_url: str, headless: bool = True) -> list[str]:
|
||||||
|
"""Extract all script URLs from a page using Playwright."""
|
||||||
|
script_urls = []
|
||||||
|
|
||||||
|
with sync_playwright() as playwright:
|
||||||
|
browser = playwright.chromium.launch(headless=headless)
|
||||||
|
page = browser.new_page()
|
||||||
|
|
||||||
|
def handle_response(response):
|
||||||
|
if response.request.resource_type == "script":
|
||||||
|
script_urls.append(response.url)
|
||||||
|
logger.debug(f"Detected script: {response.url}")
|
||||||
|
|
||||||
|
page.on("response", handle_response)
|
||||||
|
logger.info("Navigating to page...")
|
||||||
|
page.goto(url=page_url)
|
||||||
|
logger.info("Waiting for DOM ready...")
|
||||||
|
page.wait_for_load_state(state="domcontentloaded")
|
||||||
|
logger.info("Waiting for network idle (30s timeout)...")
|
||||||
|
try:
|
||||||
|
page.wait_for_load_state(state="networkidle", timeout=30000)
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(f"Network idle timeout: {e}")
|
||||||
|
|
||||||
|
browser.close()
|
||||||
|
|
||||||
|
return script_urls
|
||||||
|
|
||||||
|
|
||||||
|
def download_file(url: str, timeout: int = 30) -> bytes:
|
||||||
|
"""Download a file from a URL."""
|
||||||
|
try:
|
||||||
|
with httpx.Client(timeout=timeout, follow_redirects=True) as client:
|
||||||
|
response = client.get(url=url)
|
||||||
|
response.raise_for_status()
|
||||||
|
return response.content
|
||||||
|
except httpx.HTTPError as e:
|
||||||
|
logger.error(f"Failed to download {url}: {e}")
|
||||||
|
raise
|
||||||
|
|
||||||
|
|
||||||
|
def get_sourcemap_url(script_content: str, script_url: str) -> Optional[str]:
|
||||||
|
"""Extract source map URL from script content."""
|
||||||
|
match = re.search(r"//# sourceMappingURL=(.+?)(?:\n|$)", script_content)
|
||||||
|
if not match:
|
||||||
|
return None
|
||||||
|
|
||||||
|
sourcemap_ref = match.group(1).strip()
|
||||||
|
|
||||||
|
if sourcemap_ref.startswith("http"):
|
||||||
|
return sourcemap_ref
|
||||||
|
|
||||||
|
return urljoin(base=script_url, url=sourcemap_ref)
|
||||||
|
|
||||||
|
|
||||||
|
def decode_sourcemap(sourcemap_data: dict) -> dict:
|
||||||
|
"""Extract sources and content from source map."""
|
||||||
|
sources = sourcemap_data.get("sources", [])
|
||||||
|
sources_content = sourcemap_data.get("sourcesContent", [])
|
||||||
|
|
||||||
|
return {
|
||||||
|
"sources": sources,
|
||||||
|
"sources_content": sources_content,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def reconstruct_sources(
|
||||||
|
script_content: str,
|
||||||
|
sourcemap_data: dict,
|
||||||
|
output_dir: Path,
|
||||||
|
script_name: str,
|
||||||
|
) -> None:
|
||||||
|
"""Reconstruct original sources from source map."""
|
||||||
|
sources = sourcemap_data.get("sources", [])
|
||||||
|
sources_content = sourcemap_data.get("sourcesContent", [])
|
||||||
|
|
||||||
|
logger.info(f"Source map has {len(sources)} source files")
|
||||||
|
if sources:
|
||||||
|
logger.debug(f"First few sources: {sources[:3]}")
|
||||||
|
|
||||||
|
if not sources_content or all(c is None for c in sources_content):
|
||||||
|
logger.error(f"No source content found in source map for {script_name}")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
for i, source_file in enumerate(sources):
|
||||||
|
if i < len(sources_content) and sources_content[i]:
|
||||||
|
source_path = output_dir / source_file
|
||||||
|
source_path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
with open(source_path, mode="w", encoding="utf-8") as f:
|
||||||
|
f.write(sources_content[i])
|
||||||
|
|
||||||
|
logger.info(f"Reconstructed {source_file}")
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
args = parse_args()
|
||||||
|
args.output_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
logger.info(f"Fetching page: {args.url}")
|
||||||
|
script_urls = get_script_urls(page_url=args.url, headless=args.headless)
|
||||||
|
logger.info(f"Found {len(script_urls)} script(s)")
|
||||||
|
|
||||||
|
if not script_urls:
|
||||||
|
logger.error("No scripts found on page")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
for script_url in script_urls:
|
||||||
|
logger.info(f"Processing {script_url}")
|
||||||
|
|
||||||
|
try:
|
||||||
|
script_content = download_file(url=script_url).decode(encoding="utf-8")
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"Failed to get script {script_url}: {e}")
|
||||||
|
continue
|
||||||
|
|
||||||
|
sourcemap_url = get_sourcemap_url(script_content=script_content, script_url=script_url)
|
||||||
|
|
||||||
|
if not sourcemap_url:
|
||||||
|
logger.error(f"No source map found for {script_url}")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
logger.info(f"Downloading source map: {sourcemap_url}")
|
||||||
|
|
||||||
|
try:
|
||||||
|
sourcemap_content = download_file(url=sourcemap_url).decode(encoding="utf-8")
|
||||||
|
sourcemap_data = json.loads(sourcemap_content)
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"Failed to download/parse source map: {e}")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
script_name = urlparse(script_url).path.split("/")[-1]
|
||||||
|
reconstruct_sources(
|
||||||
|
script_content=script_content,
|
||||||
|
sourcemap_data=sourcemap_data,
|
||||||
|
output_dir=args.output_dir,
|
||||||
|
script_name=script_name,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info(f"All sources reconstructed to {args.output_dir}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user