diff --git a/.github/workflows/canary.yml b/.github/workflows/canary.yml index d7ceb32..7d4c290 100644 --- a/.github/workflows/canary.yml +++ b/.github/workflows/canary.yml @@ -119,6 +119,18 @@ jobs: f"the {floor:,} floor") if len(r.get("sku") or "") != 36: fail.append(f"@{name}: sku {r.get('sku')!r} is not a UUID") + # Snapchat's page state is not a published contract. A served page + # that has lost a key the parser reads gives silent nulls, so the + # parser names such keys and this fails on them by name. + for label, m in (("profiles", meta),): + if m.get("payload_keys_missing"): + fail.append(f"{label}: Snapchat moved key(s) the parser " + f"reads: {m['payload_keys_missing']}") + # The default transport is plain HTTP; a browser here means the + # site refused it, which is the README's central claim failing. + if meta.get("transport") != "http": + fail.append(f"transport {meta.get('transport')!r}: the default " + f"run did not stay on plain HTTP") espn = by.get("espn") if espn is None or espn.get("public_profile") is not False: fail.append(f"@espn: expected an ordinary account row, got " @@ -140,6 +152,9 @@ jobs: if bad: fail.append(f"spotlight: {len(bad)} row(s) without a positive " f"view_count — an empty slot read as a video?") + if smeta.get("payload_keys_missing"): + fail.append(f"spotlight: Snapchat moved key(s) the parser " + f"reads: {smeta['payload_keys_missing']}") pairs = [(r["page"], r["position"]) for r in spot] if len(set(pairs)) != len(pairs): fail.append("spotlight: page+position is not unique") diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index a2e0b65..2bac1a8 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -17,7 +17,7 @@ jobs: matrix: # 3.9 is the floor the README claims. Claiming it without testing it is # how a walrus operator or a `X | None` annotation ships and breaks it. - python-version: ["3.9", "3.12"] + python-version: ["3.9", "3.14"] steps: - uses: actions/checkout@v4 diff --git a/CHANGELOG.md b/CHANGELOG.md index 9d73c85..bbc5b87 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,45 @@ closely as a CLI toolkit can. A patch release means **fixes** — it does not promise that every flag's default is frozen, and where a default does change in one, the note leads with it. +## [0.2.0] — 2026-10-05 + +Changes from a third-party audit, each checked against a live run before it +was accepted. Two columns are added to the profile row, which is why this +is a minor release; no column was removed or renamed. + +### Added + +- **`has_more_spotlight` and `has_more_highlights` on the profile row**, + from the page's own cursors. The page lists about 25 Spotlight videos + and a slice of highlights; these say when the account has more, on the + row itself, so a row read without its sidecar still says its counts are + a slice. +- **`payload_keys_missing` in the sidecar.** The parser now checks that + every `__NEXT_DATA__` key it reads is present, and names the ones that + are not, per account. A page that renders but has lost a field used to + produce silent nulls. Zero false positives on 20 raw captures and a live + page; the daily canary fails on a non-empty value. +- **`transport` records what actually fetched the pages** (`http`, + `browser`, `cdp`), with `transport_requested` beside it. It used to + record the flag, so the default run said `auto` and never which one. +- **CSV formula neutralisation**, lifted from rakuten-scraper: a string cell + beginning `=`, `+`, `-`, `@` or a control character is prefixed with `'` + in CSV only, and counted in `csv_cells_escaped`. It does not fire on + today's data (0 of 10,554 string cells, measured); it is here because + the text is the account owner's. +- **A check that the three engines' shared code is identical**, so the + three copies cannot drift apart unnoticed. +- CI tests the newest end of the supported range on Python 3.14 (was 3.12). + +### Fixed + +- **Output files were created owner-only (0600).** The atomic writer's + temporary file is 0600 and the rename kept it: nine of nine files on a + live run under umask 022. New files now get the umask's mode (0644 + there). A file that already exists keeps its mode, so outputs written by + 0.1.0 stay 0600 until you delete them once — the writer cannot tell a + mode someone chose from one the bug left. + ## [0.1.0] — 2026-09-24 First release. diff --git a/README.md b/README.md index 7a920d4..4c2ffe1 100644 --- a/README.md +++ b/README.md @@ -3,7 +3,7 @@ [![release](https://img.shields.io/github/v/release/2scraper/snapchat-scraper?sort=semver)](https://github.com/2scraper/snapchat-scraper/releases) [![tests](https://github.com/2scraper/snapchat-scraper/actions/workflows/tests.yml/badge.svg)](https://github.com/2scraper/snapchat-scraper/actions/workflows/tests.yml) [![canary](https://github.com/2scraper/snapchat-scraper/actions/workflows/canary.yml/badge.svg)](https://github.com/2scraper/snapchat-scraper/actions/workflows/canary.yml) -[![python](https://img.shields.io/badge/python-3.9%20%7C%203.12-blue)](pyproject.toml) +[![python](https://img.shields.io/badge/python-3.9%20%7C%203.14-blue)](pyproject.toml) [![licence](https://img.shields.io/badge/licence-MIT-green)](LICENSE) [![engines](https://img.shields.io/badge/engines-Playwright%20%7C%20Selenium%20%7C%20pyppeteer%20%7C%20CDP-informational)](#engines-and-what-each-one-costs-you) [![runs without an account](https://img.shields.io/badge/runs%20without-an%20account-brightgreen)](#you-do-not-need-a-key-a-proxy-or-an-account) @@ -130,9 +130,10 @@ Measured on 2026-09-24: @nasa — 757,800 subscribers, 7 Spotlight videos, * **At most ~25 Spotlight videos per account.** That is what a profile page lists. The page carries a cursor for the rest, consumed by Snapchat's own protobuf API, which **this repo does not implement**. The - sidecar names every account whose page said there was more, in - `handles_with_more_than_page`, so "complete" is never misread as "the - account's whole history". + profile row says so itself, in `has_more_spotlight` and + `has_more_highlights`, and the sidecar names every such account in + `handles_with_more_than_page`. `status: complete` means every account + asked for was answered — never "the account's whole history". * **An account with one row and nothing in it.** An ordinary account — not a Public Profile — is served as a username and a Snapcode and nothing else (@espn, measured). `--mode profile` emits a row with @@ -230,7 +231,23 @@ reason, **which** accounts failed by number, `handles_unavailable`, `handles_without_public_profile`, `handles_with_more_than_page`, `spotlight_empty_slots`, and `viewer_countries` — the country Snapchat says the request came from, which is how you check that a proxy's or a -Scraping Browser's `country-` segment did what you asked. +Scraping Browser's `country-` segment did what you asked. Also: + +* `transport` — what actually fetched the pages (`http`, `browser` or + `cdp`), beside `transport_requested`. The default `auto` is HTTP until + the site refuses, so this is the field that says whether a browser was + ever involved; `engine` names the script that ran. +* `payload_keys_missing` — keys this parser reads that a served page no + longer carried, with the accounts each was missing on. The source is + Snapchat's own page state, not a published contract, so a page can + render and still have moved a field; this is how that shows up instead + of as silent nulls. Empty on every page this repo was tested against, + and the daily canary fails if it is not. +* `csv_cells_escaped` — CSV cells that began with `=`, `+`, `-`, `@` or a + control character and were prefixed with `'` so a spreadsheet reads + them as text (bios and captions are written by account owners). JSON + keeps the bytes as served. 0 of 10,554 string cells on a live run of + 2026-10-05. --- diff --git a/output_writer.py b/output_writer.py index 02e69de..a48c1b6 100644 --- a/output_writer.py +++ b/output_writer.py @@ -137,6 +137,13 @@ class Profile: has_story: Optional[bool] = None has_curated_highlights: Optional[bool] = None has_spotlight_highlights: Optional[bool] = None + # True when the page carried a cursor for MORE Spotlight videos / + # highlights than it listed. The counts above are then the page's + # slice, not the account's total. On the ROW rather than only in the + # sidecar, because a row that reaches a consumer without its sidecar + # must still say so (third-party audit, 2026-10-05). + has_more_spotlight: Optional[bool] = None + has_more_highlights: Optional[bool] = None # ---- what the account says about itself -------------------------------- bio: Optional[str] = None @@ -314,6 +321,28 @@ def _csv_value(v: Any) -> Any: return v +def _target_mode(path: str) -> int: + """The permission bits the finished file should carry. + + `NamedTemporaryFile` creates its file 0600 and `os.replace` keeps the + mode, so every output this module wrote was readable by its owner alone + — measured 2026-10-05 on a live run under umask 022: nine of nine + `.json`/`.csv`/`.meta.json` files came out 0600, where the `open(path, + "w")` this writer replaced would have given 0644. A file a cron job + writes and another user's pipeline reads then fails on permission + rather than on content. + + An EXISTING target keeps its mode, because someone may have tightened + it on purpose; a new one gets what `open()` would have given it. + """ + try: + return os.stat(path).st_mode & 0o777 + except OSError: + umask = os.umask(0) + os.umask(umask) + return 0o666 & ~umask + + @contextlib.contextmanager def _atomic(path: str, newline: Optional[str] = None): """Write to a temporary file beside `path`, then rename over it. @@ -345,6 +374,7 @@ def _atomic(path: str, newline: Optional[str] = None): yield handle handle.flush() os.fsync(handle.fileno()) + os.chmod(handle.name, _target_mode(path)) os.replace(handle.name, path) except BaseException: # Leave the destination untouched. A failed write must not be @@ -361,7 +391,42 @@ def write_json(rows: Sequence[Any], path: str) -> None: json.dump([asdict(r) for r in rows], f, ensure_ascii=False, indent=2) -def write_csv(rows: Sequence[Any], path: str, row_cls: Type = Profile) -> None: +# A spreadsheet treats a cell beginning with one of these as a FORMULA, not +# as text, and the bio, title, address and caption columns here are written +# by whoever owns the account. `=HYPERLINK(...)`, `+cmd|...` and `@SUM(...)` +# are the classic shapes; the tab and the newline are here because a leading +# one is stripped by some readers, which exposes whatever follows it. +# +# Measured on this site on 2026-10-05 before shipping the guard: 0 of 10,554 +# string cells across a live run of five accounts in all three modes begin +# with any of them. So this does not fire on today's data and is not claimed +# to; it is here because the text is not ours. Lifted verbatim from +# rakuten-scraper, where it was written first. +CSV_FORMULA_LEADS = ("=", "+", "-", "@", "\t", "\r", "\n") + +# Prefixed to a cell that would otherwise be read as a formula. Excel, Google +# Sheets and LibreOffice all treat the apostrophe as "the rest of this cell +# is text" and do not show it; a consumer parsing the CSV with `csv` sees it, +# which is why the count goes in the sidecar rather than staying silent. +CSV_FORMULA_ESCAPE = "'" + + +def _csv_escape(v: Any) -> Any: + """Neutralise a formula-shaped cell. Returns (value, was_escaped). + + Only STRINGS are touched. Escaping a number would turn `-5` into the text + `'-5` and break every sum a consumer writes over the column. + + CSV only. The JSON output keeps the site's bytes exactly as served, so + the two files deliberately differ; `csv_cells_escaped` in the sidecar is + what declares that divergence instead of leaving it to be discovered. + """ + if isinstance(v, str) and v.startswith(CSV_FORMULA_LEADS): + return CSV_FORMULA_ESCAPE + v, True + return v, False + + +def write_csv(rows: Sequence[Any], path: str, row_cls: Type = Profile) -> int: # An empty result still gets the header row. A zero-byte file makes a # consumer fail on read (no columns to parse) instead of reading a valid # table with zero rows — and "an empty result is still a well-formed @@ -370,11 +435,21 @@ def write_csv(rows: Sequence[Any], path: str, row_cls: Type = Profile) -> None: # The header comes from `row_cls`, not from the first row, so an empty # run still writes the columns of the mode that produced it. fieldnames = [f.name for f in fields(row_cls)] + escaped = 0 with _atomic(path, newline="") as f: writer = csv.DictWriter(f, fieldnames=fieldnames) writer.writeheader() for r in rows: - writer.writerow({k: _csv_value(v) for k, v in asdict(r).items()}) + row = {} + for k, v in asdict(r).items(): + # Escape AFTER `_csv_value`: a list joined into one cell is + # text too, and its first element can be formula-shaped + # while the list itself is not a string. + value, was_escaped = _csv_escape(_csv_value(v)) + row[k] = value + escaped += was_escaped + writer.writerow(row) + return escaped # Exit code used when a run completes but produced nothing. Distinct from 1 @@ -515,7 +590,8 @@ def run_meta(status: str, stop_reason: str, pages_requested: int, def save(rows: Sequence[Any], out_prefix: str, fmt: str, - allow_empty: bool = False, row_cls: Type = Profile) -> int: + allow_empty: bool = False, row_cls: Type = Profile, + stats: Optional[dict] = None) -> int: """Write JSON/CSV and return a process exit code. Returns 0 when rows were written, EXIT_NO_PRODUCTS when there were none. @@ -542,8 +618,17 @@ def save(rows: Sequence[Any], out_prefix: str, fmt: str, write_json(rows, f"{out_prefix}.json") print(f"[+] Saved {len(rows)} rows -> {out_prefix}.json") if fmt in ("csv", "both"): - write_csv(rows, f"{out_prefix}.csv", row_cls=row_cls) + escaped = write_csv(rows, f"{out_prefix}.csv", row_cls=row_cls) print(f"[+] Saved {len(rows)} rows -> {out_prefix}.csv") + # Reported through `stats` rather than as a return value because + # this function's return IS the exit code. + if stats is not None: + stats["csv_cells_escaped"] = escaped + if escaped: + print(f"[!] {escaped} CSV cell(s) began with a formula character " + f"and were prefixed with {CSV_FORMULA_ESCAPE!r} so a " + f"spreadsheet reads them as text. The JSON output is " + f"unchanged — see csv_cells_escaped in the sidecar.") return 0 if rows else EXIT_NO_PRODUCTS @@ -604,8 +689,14 @@ def finish_run(rows: Sequence[Any], out_prefix: str, fmt: str, # failed, the run is not complete, whatever it stopped for. complete = stop_reason in COMPLETE_STOP_REASONS and not pages_failed row_cls = ROW_CLASS_BY_MODE.get(mode, Profile) - rc = save(rows, out_prefix, fmt, allow_empty=allow_empty, row_cls=row_cls) + stats: dict = {} + rc = save(rows, out_prefix, fmt, allow_empty=allow_empty, row_cls=row_cls, + stats=stats) wrote_output = bool(rows) or allow_empty + # Housekeeping goes UNDER the caller's extra, never over it: a name + # collision must not let a count of ours silently replace a fact the + # engine recorded about the site. + extra = {**stats, **(extra or {})} if wrote_output: status = "complete" if (rows and complete) else ( diff --git a/playwright_scraper.py b/playwright_scraper.py index 0b6f821..37d04a0 100644 --- a/playwright_scraper.py +++ b/playwright_scraper.py @@ -1029,6 +1029,7 @@ def _run_profile(session_box, pw, args, pool) -> Tuple[List[Any], Dict[str, Any] more_beyond_page: List[str] = [] empty_slots = 0 countries = set() + drift: Dict[str, List[str]] = {} blocked = False for i, outcome in enumerate(outcomes): blocked = blocked or outcome.blocked @@ -1046,6 +1047,8 @@ def _run_profile(session_box, pw, args, pool) -> Tuple[List[Any], Dict[str, Any] empty_slots += int(diag.get("spotlight_empty_slots") or 0) if diag.get("viewer_country"): countries.add(diag["viewer_country"]) + for key in diag.get("payload_keys_missing") or (): + drift.setdefault(key, []).append(handles[i]) # `page` is WHICH ACCOUNT of the run a row came from, so that # `page`+`position` is unique across a multi-account run — CLAUDE.md # §18 records a sibling where 60 of 119 rows claimed a position @@ -1088,6 +1091,12 @@ def _run_profile(session_box, pw, args, pool) -> Tuple[List[Any], Dict[str, Any] # how a reader checks that a proxy or a Scraping Browser # `country-` segment did what it was asked. "viewer_countries": sorted(countries) or None, + # Keys this parser reads that a served page no longer carried, with + # the accounts each was missing on. Empty on every page this repo + # was built and tested against; non-empty means Snapchat moved a + # field and some columns of those accounts' rows are nulls of the + # parser's making, not the account's. The canary fails on it. + "payload_keys_missing": drift or None, } return rows, meta @@ -1111,7 +1120,23 @@ def __enter__(self): def __exit__(self, *exc): return self._pw.__exit__(*exc) +def _transport_used(args) -> str: + """What actually fetched the pages: "http", "browser" or "cdp". + + `--transport` says what was ASKED for, and the default `auto` is not a + transport at all — the sidecar used to record "auto" for a run that + never started a browser, so a reader could not tell from it whether + Chromium had been involved (third-party audit, 2026-10-05). `auto` + switches `args.transport` to "browser" the moment it falls back, so + reading it after the run gives the answer. + """ + if args.cdp_endpoint: + return "cdp" + return "browser" if getattr(args, "transport", "auto") == "browser" else "http" + + def scrape(args) -> int: + transport_requested = getattr(args, "transport", "auto") pool = proxy_pool_from_args(args) handles = _handles(args) if pool and args.concurrency > 1: @@ -1141,7 +1166,12 @@ def scrape(args) -> int: "blocked")} extra["engine"] = "playwright" extra["category"] = args.category - extra["transport"] = getattr(args, "transport", "auto") + # `engine` names the CLI that ran (which driver the browser path would + # use); `transport` names what actually fetched the pages. The audit + # asked for `engine=http` instead — but the engine is still this + # script, and folding the two into one field would lose which one. + extra["transport"] = _transport_used(args) + extra["transport_requested"] = transport_requested unavailable = meta.get("handles_unavailable") if unavailable: @@ -1153,6 +1183,14 @@ def scrape(args) -> int: logger.info("%d account(s) have no public profile, so Snapchat shows " "them no Spotlight and no story: %s", len(not_public), ", ".join("@" + h for h in not_public)) + drift = meta.get("payload_keys_missing") + if drift: + logger.warning("Snapchat's page no longer carries %d key(s) this " + "parser reads: %s. The page was served and parsed, " + "but columns read from those keys are null for the " + "parser's reasons, not the account's. See " + "payload_keys_missing in the sidecar.", len(drift), + ", ".join(sorted(drift))) more = meta.get("handles_with_more_than_page") if more and args.mode != "profile": logger.info("%d account(s) have more Spotlight/highlights than their " diff --git a/product_parser.py b/product_parser.py index 216e4eb..d34a611 100644 --- a/product_parser.py +++ b/product_parser.py @@ -330,6 +330,73 @@ def _story_snaps(props: Dict[str, Any]) -> List[Dict[str, Any]]: return [s for s in (story.get("snapList") or []) if isinstance(s, dict)] +# --------------------------------------------------------------------------- +# Schema drift +# --------------------------------------------------------------------------- +# +# The source is Snapchat's own page state, not a published contract, so it can +# change shape without a version bump. A page that still renders but has lost +# a key this parser reads produces rows whose columns are silently null, and +# nothing about the run looks wrong — the failure a third-party audit named +# first (2026-10-05). +# +# So the parser checks for the KEYS it reads, never their values. A null value +# is a legitimate answer here (a hidden subscriber count, an account with no +# bio, a slot that is empty); a missing key is the site having moved. Each +# list below was read off the real captures this repo's fixtures come from, +# and every key in it is present on every capture that should carry it. +DRIFT_PUBLIC_PROFILE = ("username", "title", "subscriberCount", + "businessProfileId", "bio", "websiteUrl", + "categoryStringId", "badge", "profilePictureUrl", + "hasStory", "hasCuratedHighlights", + "hasSpotlightHighlights") +DRIFT_USER_INFO = ("username",) +DRIFT_PAGE_PROPS = ("story", "curatedHighlights", "spotlightHighlights", + "spotlightStoryMetadata", "lenses") +DRIFT_SPOTLIGHT_META = ("videoMetadata", "engagementStats") +DRIFT_VIDEO_METADATA = ("name", "description", "uploadDateMs", "durationMs", + "contentUrl") +DRIFT_ENGAGEMENT = ("viewCount", "shareCount", "commentCount", "boostCount") +DRIFT_SNAP = ("snapIndex", "snapId", "snapMediaType", "snapUrls", + "timestampInSec") + + +def payload_keys_missing(props: Dict[str, Any]) -> List[str]: + """Keys this parser reads that the page no longer carries, as paths. + + Empty on every capture this repo was built from. Non-empty means the + site moved a field, and the rows from this page may carry nulls that + are the parser's, not the account's. + """ + missing = set() + + def need(obj, keys, path): + if isinstance(obj, dict): + for k in keys: + if k not in obj: + missing.add(f"{path}.{k}") + + kind, info = account(props) + if kind == PROFILE_PUBLIC: + need(info, DRIFT_PUBLIC_PROFILE, "publicProfileInfo") + need(props, DRIFT_PAGE_PROPS, "pageProps") + for _, meta in _filled_spotlights(props)[0]: + need(meta, DRIFT_SPOTLIGHT_META, "spotlightStoryMetadata[]") + need(meta.get("videoMetadata"), DRIFT_VIDEO_METADATA, + "spotlightStoryMetadata[].videoMetadata") + need(meta.get("engagementStats"), DRIFT_ENGAGEMENT, + "spotlightStoryMetadata[].engagementStats") + snaps = list(_story_snaps(props)) + for coll in _highlight_collections(props): + snaps.extend(s for s in coll.get("snapList") or [] + if isinstance(s, dict)) + for snap in snaps: + need(snap, DRIFT_SNAP, "snapList[]") + elif kind == PROFILE_USER: + need(info, DRIFT_USER_INFO, "userInfo") + return sorted(missing) + + def _profile_row(props, html, url, scraped_at, row_cls, handle): kind, info = account(props) username = (_text(info.get("username")) or handle or "").lower() or None @@ -378,6 +445,8 @@ def _profile_row(props, html, url, scraped_at, row_cls, handle): has_story=_bool(info.get("hasStory")), has_curated_highlights=_bool(info.get("hasCuratedHighlights")), has_spotlight_highlights=_bool(info.get("hasSpotlightHighlights")), + has_more_spotlight=bool(_text(props.get("spotlightHighlightsCursor"))), + has_more_highlights=bool(_text(props.get("curatedHighlightsCursor"))), bio=_text(info.get("bio")), website_url=_text(info.get("websiteUrl")), address=_text(info.get("address")), @@ -571,6 +640,7 @@ def parse_page(html: Any, url: str, scraped_at: str, row_cls: Any, "more_spotlight": bool(_text(props.get("spotlightHighlightsCursor"))), "more_highlights": bool(_text(props.get("curatedHighlightsCursor"))), "viewer_country": _text((props.get("viewerInfo") or {}).get("country")), + "payload_keys_missing": payload_keys_missing(props), } if mode == "profile": diff --git a/puppeteer_scraper.py b/puppeteer_scraper.py index 6dc6684..83d1732 100644 --- a/puppeteer_scraper.py +++ b/puppeteer_scraper.py @@ -1198,6 +1198,7 @@ def _run_profile(session_box, pw, args, pool) -> Tuple[List[Any], Dict[str, Any] more_beyond_page: List[str] = [] empty_slots = 0 countries = set() + drift: Dict[str, List[str]] = {} blocked = False for i, outcome in enumerate(outcomes): blocked = blocked or outcome.blocked @@ -1215,6 +1216,8 @@ def _run_profile(session_box, pw, args, pool) -> Tuple[List[Any], Dict[str, Any] empty_slots += int(diag.get("spotlight_empty_slots") or 0) if diag.get("viewer_country"): countries.add(diag["viewer_country"]) + for key in diag.get("payload_keys_missing") or (): + drift.setdefault(key, []).append(handles[i]) # `page` is WHICH ACCOUNT of the run a row came from, so that # `page`+`position` is unique across a multi-account run — CLAUDE.md # §18 records a sibling where 60 of 119 rows claimed a position @@ -1257,6 +1260,12 @@ def _run_profile(session_box, pw, args, pool) -> Tuple[List[Any], Dict[str, Any] # how a reader checks that a proxy or a Scraping Browser # `country-` segment did what it was asked. "viewer_countries": sorted(countries) or None, + # Keys this parser reads that a served page no longer carried, with + # the accounts each was missing on. Empty on every page this repo + # was built and tested against; non-empty means Snapchat moved a + # field and some columns of those accounts' rows are nulls of the + # parser's making, not the account's. The canary fails on it. + "payload_keys_missing": drift or None, } return rows, meta @@ -1280,7 +1289,23 @@ def __exit__(self, *exc): return False +def _transport_used(args) -> str: + """What actually fetched the pages: "http", "browser" or "cdp". + + `--transport` says what was ASKED for, and the default `auto` is not a + transport at all — the sidecar used to record "auto" for a run that + never started a browser, so a reader could not tell from it whether + Chromium had been involved (third-party audit, 2026-10-05). `auto` + switches `args.transport` to "browser" the moment it falls back, so + reading it after the run gives the answer. + """ + if args.cdp_endpoint: + return "cdp" + return "browser" if getattr(args, "transport", "auto") == "browser" else "http" + + def scrape(args) -> int: + transport_requested = getattr(args, "transport", "auto") pool = proxy_pool_from_args(args) handles = _handles(args) if pool and args.concurrency > 1: @@ -1310,7 +1335,12 @@ def scrape(args) -> int: "blocked")} extra["engine"] = "puppeteer" extra["category"] = args.category - extra["transport"] = getattr(args, "transport", "auto") + # `engine` names the CLI that ran (which driver the browser path would + # use); `transport` names what actually fetched the pages. The audit + # asked for `engine=http` instead — but the engine is still this + # script, and folding the two into one field would lose which one. + extra["transport"] = _transport_used(args) + extra["transport_requested"] = transport_requested unavailable = meta.get("handles_unavailable") if unavailable: @@ -1322,6 +1352,14 @@ def scrape(args) -> int: logger.info("%d account(s) have no public profile, so Snapchat shows " "them no Spotlight and no story: %s", len(not_public), ", ".join("@" + h for h in not_public)) + drift = meta.get("payload_keys_missing") + if drift: + logger.warning("Snapchat's page no longer carries %d key(s) this " + "parser reads: %s. The page was served and parsed, " + "but columns read from those keys are null for the " + "parser's reasons, not the account's. See " + "payload_keys_missing in the sidecar.", len(drift), + ", ".join(sorted(drift))) more = meta.get("handles_with_more_than_page") if more and args.mode != "profile": logger.info("%d account(s) have more Spotlight/highlights than their " diff --git a/pyproject.toml b/pyproject.toml index 5397774..0ce335b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "snapchat-scraper" -version = "0.1.0" +version = "0.2.0" description = "Snapchat scraper: public profiles (subscribers, bio, category, website), Spotlight videos with view/share/comment/boost counts, and story and highlight snaps, read from the page Snapchat server-renders to anyone." readme = "README.md" license = "MIT" diff --git a/sample_output.csv b/sample_output.csv index 11a09fa..41a8a94 100644 --- a/sample_output.csv +++ b/sample_output.csv @@ -1,5 +1,5 @@ -source,scraped_at,url,sku,title,username,public_profile,business_profile_id,host_user_id,entity_type,subscriber_count,spotlight_count,highlight_count,story_snap_count,lens_count,has_story,has_curated_highlights,has_spotlight_highlights,bio,website_url,address,category,subcategory,badge,profile_picture_url,hero_image_url,snapcode_url,created_at,modified_at,page,position -snapchat.com,2026-09-24T08:12:42Z,https://www.snapchat.com/@nasa,cb983b26-c537-474b-9699-065309205ea2,NASA,nasa,True,cb983b26-c537-474b-9699-065309205ea2,92ea5bae-4f78-4611-ba8c-a60235ac2b49,Organization,757800,7,10,0,3,False,True,True,Explore the universe and discover our home planet with official NASA snaps.,https://www.nasa.gov,,public-profile-category-v3-business-group,public-profile-subcategory-v3-government-org,1,"https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvUVg2VHZkaDBHNUMzcXVidWhWaTJhP2JvPUVnMGFBQm9BTWdFRVNBSlFHV0FCJnVjPTI1._RS0,90_FMjpeg","https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvalRSaVBkS2NYWmJlV0w4dXNPeWNtP2JvPUVna3lBUVJJQWxBWllBRSUzRCZ1Yz0yNQ._RS0,1080_FMjpeg",https://app.snapchat.com/web/deeplink/snapcode?username=nasa&type=SVG&bitmoji=enable,2019-05-22T17:38:27Z,2026-09-18T20:46:24Z,1,1 -snapchat.com,2026-09-24T08:12:42Z,https://www.snapchat.com/@mrbeast,fe63dec1-4fa4-476c-9c73-19b2bc2e8589,MrBeast,mrbeast,True,fe63dec1-4fa4-476c-9c73-19b2bc2e8589,cb1adae6-ecc9-4a5f-b979-3f34d4cf1681,Person,1466200,4,6,0,0,False,True,True,,,,,,1,"https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvR2M2MEJNODVEdjRHWU5yVk9EbFZwP2JvPUVnMGFBQm9BTWdFRVNBSlFHV0FCJnVjPTI1._RS0,90_FMpng","https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvQ3F4eklZZHVGZUI3Tm55SGRBM2FEP2JvPUVnMGFBQm9BTWdFRVNBSlFHV0FCJnVjPTI1._RS0,1080_FMjpeg",https://app.snapchat.com/web/deeplink/snapcode?username=mrbeast&type=SVG&bitmoji=enable,2019-05-16T14:46:37Z,2026-09-11T20:48:24Z,2,1 -snapchat.com,2026-09-24T08:12:42Z,https://www.snapchat.com/@khaby00,f18693b6-e5db-4c51-bfc2-6688a18c85ab,Khaby Lame,khaby00,True,f18693b6-e5db-4c51-bfc2-6688a18c85ab,274c98a9-c7f7-4233-94c6-9212e963087f,Person,,7,0,0,0,False,False,True,Official Khaby Lame Snapchat 70 MLN on Tik Tok,,Subscribe for a Cookie,public-profile-category-v3-people,,0,"https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvNndyY252ODVFSTZpWUJOdXY2cjM1P2JvPUVnMGFBQm9BTWdFRVNBSlFHV0FCJnVjPTI1._RS0,90_FMjpeg",,https://app.snapchat.com/web/deeplink/snapcode?username=khaby00&type=SVG&bitmoji=enable,2021-05-16T02:06:57Z,2025-08-19T07:21:14Z,3,1 -snapchat.com,2026-09-24T08:12:42Z,https://www.snapchat.com/@espn,espn,espn,espn,False,,,,,,,,,,,,,,,,,,,,https://app.snapchat.com/web/deeplink/snapcode?username=espn&type=SVG&bitmoji=enable,,,4,1 +source,scraped_at,url,sku,title,username,public_profile,business_profile_id,host_user_id,entity_type,subscriber_count,spotlight_count,highlight_count,story_snap_count,lens_count,has_story,has_curated_highlights,has_spotlight_highlights,has_more_spotlight,has_more_highlights,bio,website_url,address,category,subcategory,badge,profile_picture_url,hero_image_url,snapcode_url,created_at,modified_at,page,position +snapchat.com,2026-10-05T20:38:37Z,https://www.snapchat.com/@nasa,cb983b26-c537-474b-9699-065309205ea2,NASA,nasa,True,cb983b26-c537-474b-9699-065309205ea2,92ea5bae-4f78-4611-ba8c-a60235ac2b49,Organization,755800,7,10,0,3,False,True,True,False,True,Explore the universe and discover our home planet with official NASA snaps.,https://www.nasa.gov,,public-profile-category-v3-business-group,public-profile-subcategory-v3-government-org,1,"https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvUVg2VHZkaDBHNUMzcXVidWhWaTJhP2JvPUVnMGFBQm9BTWdFRVNBSlFHV0FCJnVjPTI1._RS0,90_FMjpeg","https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvalRSaVBkS2NYWmJlV0w4dXNPeWNtP2JvPUVna3lBUVJJQWxBWllBRSUzRCZ1Yz0yNQ._RS0,1080_FMjpeg",https://app.snapchat.com/web/deeplink/snapcode?username=nasa&type=SVG&bitmoji=enable,2019-05-22T17:38:27Z,2026-10-05T16:34:06Z,1,1 +snapchat.com,2026-10-05T20:38:37Z,https://www.snapchat.com/@mrbeast,fe63dec1-4fa4-476c-9c73-19b2bc2e8589,MrBeast,mrbeast,True,fe63dec1-4fa4-476c-9c73-19b2bc2e8589,cb1adae6-ecc9-4a5f-b979-3f34d4cf1681,,1468400,4,6,0,0,False,True,True,False,True,,,,,,1,"https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvR2M2MEJNODVEdjRHWU5yVk9EbFZwP2JvPUVnMGFBQm9BTWdFRVNBSlFHV0FCJnVjPTI1._RS0,90_FMpng","https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvQ3F4eklZZHVGZUI3Tm55SGRBM2FEP2JvPUVnMGFBQm9BTWdFRVNBSlFHV0FCJnVjPTI1._RS0,1080_FMjpeg",https://app.snapchat.com/web/deeplink/snapcode?username=mrbeast&type=SVG&bitmoji=enable,,,2,1 +snapchat.com,2026-10-05T20:38:37Z,https://www.snapchat.com/@khaby00,f18693b6-e5db-4c51-bfc2-6688a18c85ab,Khaby Lame,khaby00,True,f18693b6-e5db-4c51-bfc2-6688a18c85ab,274c98a9-c7f7-4233-94c6-9212e963087f,Person,,7,0,0,0,False,False,True,False,False,Official Khaby Lame Snapchat 70 MLN on Tik Tok,,Subscribe for a Cookie,public-profile-category-v3-people,,0,"https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvNndyY252ODVFSTZpWUJOdXY2cjM1P2JvPUVnMGFBQm9BTWdFRVNBSlFHV0FCJnVjPTI1._RS0,90_FMjpeg",,https://app.snapchat.com/web/deeplink/snapcode?username=khaby00&type=SVG&bitmoji=enable,2021-05-16T02:06:57Z,2025-08-19T07:21:14Z,3,1 +snapchat.com,2026-10-05T20:38:37Z,https://www.snapchat.com/@espn,espn,espn,espn,False,,,,,,,,,,,,,,,,,,,,,,https://app.snapchat.com/web/deeplink/snapcode?username=espn&type=SVG&bitmoji=enable,,,4,1 diff --git a/sample_output.json b/sample_output.json index 0145121..abc5c4e 100644 --- a/sample_output.json +++ b/sample_output.json @@ -1,7 +1,7 @@ [ { "source": "snapchat.com", - "scraped_at": "2026-09-24T08:12:42Z", + "scraped_at": "2026-10-05T20:38:37Z", "url": "https://www.snapchat.com/@nasa", "sku": "cb983b26-c537-474b-9699-065309205ea2", "title": "NASA", @@ -10,7 +10,7 @@ "business_profile_id": "cb983b26-c537-474b-9699-065309205ea2", "host_user_id": "92ea5bae-4f78-4611-ba8c-a60235ac2b49", "entity_type": "Organization", - "subscriber_count": 757800, + "subscriber_count": 755800, "spotlight_count": 7, "highlight_count": 10, "story_snap_count": 0, @@ -18,6 +18,8 @@ "has_story": false, "has_curated_highlights": true, "has_spotlight_highlights": true, + "has_more_spotlight": false, + "has_more_highlights": true, "bio": "Explore the universe and discover our home planet with official NASA snaps.", "website_url": "https://www.nasa.gov", "address": null, @@ -28,13 +30,13 @@ "hero_image_url": "https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvalRSaVBkS2NYWmJlV0w4dXNPeWNtP2JvPUVna3lBUVJJQWxBWllBRSUzRCZ1Yz0yNQ._RS0,1080_FMjpeg", "snapcode_url": "https://app.snapchat.com/web/deeplink/snapcode?username=nasa&type=SVG&bitmoji=enable", "created_at": "2019-05-22T17:38:27Z", - "modified_at": "2026-09-18T20:46:24Z", + "modified_at": "2026-10-05T16:34:06Z", "page": 1, "position": 1 }, { "source": "snapchat.com", - "scraped_at": "2026-09-24T08:12:42Z", + "scraped_at": "2026-10-05T20:38:37Z", "url": "https://www.snapchat.com/@mrbeast", "sku": "fe63dec1-4fa4-476c-9c73-19b2bc2e8589", "title": "MrBeast", @@ -42,8 +44,8 @@ "public_profile": true, "business_profile_id": "fe63dec1-4fa4-476c-9c73-19b2bc2e8589", "host_user_id": "cb1adae6-ecc9-4a5f-b979-3f34d4cf1681", - "entity_type": "Person", - "subscriber_count": 1466200, + "entity_type": null, + "subscriber_count": 1468400, "spotlight_count": 4, "highlight_count": 6, "story_snap_count": 0, @@ -51,6 +53,8 @@ "has_story": false, "has_curated_highlights": true, "has_spotlight_highlights": true, + "has_more_spotlight": false, + "has_more_highlights": true, "bio": null, "website_url": null, "address": null, @@ -60,14 +64,14 @@ "profile_picture_url": "https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvR2M2MEJNODVEdjRHWU5yVk9EbFZwP2JvPUVnMGFBQm9BTWdFRVNBSlFHV0FCJnVjPTI1._RS0,90_FMpng", "hero_image_url": "https://cf-st.sc-cdn.net/aps/bolt/aHR0cHM6Ly9jZi1zdC5zYy1jZG4ubmV0L2QvQ3F4eklZZHVGZUI3Tm55SGRBM2FEP2JvPUVnMGFBQm9BTWdFRVNBSlFHV0FCJnVjPTI1._RS0,1080_FMjpeg", "snapcode_url": "https://app.snapchat.com/web/deeplink/snapcode?username=mrbeast&type=SVG&bitmoji=enable", - "created_at": "2019-05-16T14:46:37Z", - "modified_at": "2026-09-11T20:48:24Z", + "created_at": null, + "modified_at": null, "page": 2, "position": 1 }, { "source": "snapchat.com", - "scraped_at": "2026-09-24T08:12:42Z", + "scraped_at": "2026-10-05T20:38:37Z", "url": "https://www.snapchat.com/@khaby00", "sku": "f18693b6-e5db-4c51-bfc2-6688a18c85ab", "title": "Khaby Lame", @@ -84,6 +88,8 @@ "has_story": false, "has_curated_highlights": false, "has_spotlight_highlights": true, + "has_more_spotlight": false, + "has_more_highlights": false, "bio": "Official Khaby Lame Snapchat 70 MLN on Tik Tok", "website_url": null, "address": "Subscribe for a Cookie", @@ -100,7 +106,7 @@ }, { "source": "snapchat.com", - "scraped_at": "2026-09-24T08:12:42Z", + "scraped_at": "2026-10-05T20:38:37Z", "url": "https://www.snapchat.com/@espn", "sku": "espn", "title": "espn", @@ -117,6 +123,8 @@ "has_story": null, "has_curated_highlights": null, "has_spotlight_highlights": null, + "has_more_spotlight": null, + "has_more_highlights": null, "bio": null, "website_url": null, "address": null, diff --git a/selenium_scraper.py b/selenium_scraper.py index 12e9e30..0379762 100644 --- a/selenium_scraper.py +++ b/selenium_scraper.py @@ -998,6 +998,7 @@ def _run_profile(session_box, pw, args, pool) -> Tuple[List[Any], Dict[str, Any] more_beyond_page: List[str] = [] empty_slots = 0 countries = set() + drift: Dict[str, List[str]] = {} blocked = False for i, outcome in enumerate(outcomes): blocked = blocked or outcome.blocked @@ -1015,6 +1016,8 @@ def _run_profile(session_box, pw, args, pool) -> Tuple[List[Any], Dict[str, Any] empty_slots += int(diag.get("spotlight_empty_slots") or 0) if diag.get("viewer_country"): countries.add(diag["viewer_country"]) + for key in diag.get("payload_keys_missing") or (): + drift.setdefault(key, []).append(handles[i]) # `page` is WHICH ACCOUNT of the run a row came from, so that # `page`+`position` is unique across a multi-account run — CLAUDE.md # §18 records a sibling where 60 of 119 rows claimed a position @@ -1057,6 +1060,12 @@ def _run_profile(session_box, pw, args, pool) -> Tuple[List[Any], Dict[str, Any] # how a reader checks that a proxy or a Scraping Browser # `country-` segment did what it was asked. "viewer_countries": sorted(countries) or None, + # Keys this parser reads that a served page no longer carried, with + # the accounts each was missing on. Empty on every page this repo + # was built and tested against; non-empty means Snapchat moved a + # field and some columns of those accounts' rows are nulls of the + # parser's making, not the account's. The canary fails on it. + "payload_keys_missing": drift or None, } return rows, meta @@ -1080,7 +1089,23 @@ def __exit__(self, *exc): return False +def _transport_used(args) -> str: + """What actually fetched the pages: "http", "browser" or "cdp". + + `--transport` says what was ASKED for, and the default `auto` is not a + transport at all — the sidecar used to record "auto" for a run that + never started a browser, so a reader could not tell from it whether + Chromium had been involved (third-party audit, 2026-10-05). `auto` + switches `args.transport` to "browser" the moment it falls back, so + reading it after the run gives the answer. + """ + if args.cdp_endpoint: + return "cdp" + return "browser" if getattr(args, "transport", "auto") == "browser" else "http" + + def scrape(args) -> int: + transport_requested = getattr(args, "transport", "auto") pool = proxy_pool_from_args(args) handles = _handles(args) if pool and args.concurrency > 1: @@ -1110,7 +1135,12 @@ def scrape(args) -> int: "blocked")} extra["engine"] = "selenium" extra["category"] = args.category - extra["transport"] = getattr(args, "transport", "auto") + # `engine` names the CLI that ran (which driver the browser path would + # use); `transport` names what actually fetched the pages. The audit + # asked for `engine=http` instead — but the engine is still this + # script, and folding the two into one field would lose which one. + extra["transport"] = _transport_used(args) + extra["transport_requested"] = transport_requested unavailable = meta.get("handles_unavailable") if unavailable: @@ -1122,6 +1152,14 @@ def scrape(args) -> int: logger.info("%d account(s) have no public profile, so Snapchat shows " "them no Spotlight and no story: %s", len(not_public), ", ".join("@" + h for h in not_public)) + drift = meta.get("payload_keys_missing") + if drift: + logger.warning("Snapchat's page no longer carries %d key(s) this " + "parser reads: %s. The page was served and parsed, " + "but columns read from those keys are null for the " + "parser's reasons, not the account's. See " + "payload_keys_missing in the sidecar.", len(drift), + ", ".join(sorted(drift))) more = meta.get("handles_with_more_than_page") if more and args.mode != "profile": logger.info("%d account(s) have more Spotlight/highlights than their " diff --git a/smoke_test.py b/smoke_test.py index 71be873..afcc151 100644 --- a/smoke_test.py +++ b/smoke_test.py @@ -884,6 +884,168 @@ def json(self): "HCAPTCHA_MAX_WAIT" in inspect.getsource(fn)) +def check_outputs_get_the_umask_mode_not_0600(): + """`NamedTemporaryFile` creates 0600 and a rename keeps it, so every + output came out owner-only — nine of nine files on a live run under + umask 022 (2026-10-05). A NEW file gets what `open()` would give; an + EXISTING file keeps the mode someone chose for it.""" + old = os.umask(0o022) + try: + with tempfile.TemporaryDirectory() as tmp: + prefix = os.path.join(tmp, "out") + output_writer.save([row("nasa")], prefix, "both") + for ext in (".json", ".csv"): + equal("a new %s is 0644 under umask 022" % ext, + oct(os.stat(prefix + ext).st_mode & 0o777), oct(0o644)) + os.chmod(prefix + ".json", 0o640) + output_writer.save([row("nasa")], prefix, "json") + equal("an existing 0640 file stays 0640", + oct(os.stat(prefix + ".json").st_mode & 0o777), oct(0o640)) + finally: + os.umask(old) + + +def check_csv_neutralises_formulas_and_json_keeps_the_bytes(): + """A bio is written by whoever owns the account. A cell opening with + = + - @ or a control character is executed by a spreadsheet, so CSV + prefixes an apostrophe; JSON keeps the site's bytes; numbers are never + touched; a joined list is escaped too; the count reaches the sidecar.""" + r = row("nasa") + r.bio = "=HYPERLINK(\"http://x\")" + r.subscriber_count = -5 + sp = rows("mrbeast", "spotlight")[0] + sp.hashtags = ["+cmd", "#ok"] + with tempfile.TemporaryDirectory() as tmp: + prefix = os.path.join(tmp, "out") + buf = io.StringIO() + with redirect_stdout(buf): + output_writer.finish_run([r], prefix, "both", False, blocked=False, + stop_reason="completed", pages_requested=1, + pages_completed=1, start_url="u", + final_url="u", mode="profile", + extra={"engine": "x"}) + got = list(csv.DictReader(open(prefix + ".csv", encoding="utf-8")))[0] + equal("the formula-shaped bio is text in CSV", got["bio"], + "'=HYPERLINK(\"http://x\")") + equal("a negative NUMBER is left alone", got["subscriber_count"], "-5") + equal("JSON keeps the bytes as served", + json.load(open(prefix + ".json", encoding="utf-8"))[0]["bio"], + "=HYPERLINK(\"http://x\")") + meta = json.load(open(prefix + ".meta.json", encoding="utf-8")) + equal("the sidecar declares the divergence", meta["csv_cells_escaped"], 1) + equal("and the caller's own fields survive the merge", meta["engine"], "x") + output_writer.save([sp], prefix + "s", "csv", row_cls=Spotlight) + got = list(csv.DictReader(open(prefix + "s.csv", encoding="utf-8")))[0] + check("a joined list whose first item is formula-shaped is escaped", + got["hashtags"].startswith("'+cmd"), "got %r" % got["hashtags"]) + + +def check_a_moved_key_is_reported_not_silently_nulled(): + """The source is Snapchat's page state, not a contract. A page that + still renders but has lost a key the parser reads would give rows with + silent nulls; the parser lists such keys, the engines put them in the + sidecar, and the canary fails on them (third-party audit, 2026-10-05). + + Empty on every fixture — counted, along with 20 raw captures and a + live page, before the check was written — and non-empty once a key is + removed from a real fixture.""" + for name in PROFILES: + if name == "not_found": + continue + equal("no drift on %s" % name, + diag(name).get("payload_keys_missing"), []) + payload = json.loads(json.dumps(PROFILES["mrbeast"]["payload"])) + pp = payload["next_data"]["props"]["pageProps"] + del pp["userProfile"]["publicProfileInfo"]["subscriberCount"] + del pp["spotlightStoryMetadata"][0]["engagementStats"]["viewCount"] + got = product_parser.parse_page(make_fixtures.as_page(payload), FIXTURE_URL, + SCRAPED_AT, Profile, "profile")[1] + equal("both moved keys are named", + got["payload_keys_missing"], + ["publicProfileInfo.subscriberCount", + "spotlightStoryMetadata[].engagementStats.viewCount"]) + pp2 = json.loads(json.dumps(PROFILES["mrbeast"]["payload"])) + pp2["next_data"]["props"]["pageProps"]["userProfile"]["publicProfileInfo"][ + "subscriberCount"] = "0" + equal("a hidden (zero) count is a VALUE, not drift", + product_parser.parse_page(make_fixtures.as_page(pp2), FIXTURE_URL, + SCRAPED_AT, Profile)[1]["payload_keys_missing"], + []) + for module in ENGINES: + src = open(os.path.join(HERE, module + ".py"), encoding="utf-8").read() + check("%s puts it in the sidecar" % module, + '"payload_keys_missing": drift or None' in src) + canary = open(os.path.join(HERE, ".github", "workflows", "canary.yml"), + encoding="utf-8").read() if os.path.isdir( + os.path.join(HERE, ".github")) else None + if canary is not None: + check("and the canary fails on it", "payload_keys_missing" in canary) + + +def check_the_profile_row_says_when_the_page_is_not_the_account(): + """`has_more_spotlight` / `has_more_highlights` on the ROW, from the + page's own cursors, so a row read without its sidecar still says the + counts are a slice.""" + equal("@kyliejenner: more of both", + (row("kyliejenner").has_more_spotlight, + row("kyliejenner").has_more_highlights), (True, True)) + equal("@nasa: more highlights only", + (row("nasa").has_more_spotlight, row("nasa").has_more_highlights), + (False, True)) + equal("an ordinary account: unknown, not False", + (row("espn").has_more_spotlight, row("espn").has_more_highlights), + (None, None)) + + +def check_the_sidecar_records_the_transport_that_ran(): + """The default `--transport auto` was written to the sidecar as "auto", + which names no transport at all (live run, 2026-10-05). It now records + what fetched the pages, and what was asked for beside it.""" + engine = _import_engine("playwright_scraper") + if engine is None: + return + ns = types.SimpleNamespace + equal("auto that never fell back is http", + engine._transport_used(ns(transport="auto", cdp_endpoint=None)), "http") + equal("auto that fell back is browser (the engine rewrites args.transport)", + engine._transport_used(ns(transport="browser", cdp_endpoint=None)), + "browser") + equal("a CDP endpoint is cdp", + engine._transport_used(ns(transport="auto", cdp_endpoint="ws://h:1")), + "cdp") + for module in ENGINES: + src = open(os.path.join(HERE, module + ".py"), encoding="utf-8").read() + check("%s records the transport used, not the one asked for" % module, + 'extra["transport"] = _transport_used(args)' in src + and 'extra["transport_requested"]' in src) + + +def check_the_three_engines_share_one_copy_of_the_loop(): + """Everything from `_dump` to the driver context, and `scrape()` to + the end, is the SAME code in all three engines — only the driver layer + above it differs. Held by this check rather than by discipline; a + third-party audit (2026-10-05) named the three copies as the risk.""" + def block(src, start, end=None): + i = src.index(start) + j = src.index(end, i) if end else len(src) + return src[i:j] + srcs = {m: open(os.path.join(HERE, m + ".py"), encoding="utf-8").read() + for m in ENGINES} + mids = {m: block(s, "def _dump(", "class _driver_context") for m, s in srcs.items()} + equal("the fetch/run block is identical in all three", + len(set(mids.values())), 1) + exc = {"playwright_scraper": "PWError", "puppeteer_scraper": "PyppeteerError", + "selenium_scraper": "WebDriverException"} + tails = set() + for m, s in srcs.items(): + t = block(s, "def scrape(args) -> int:") + t = t.replace('extra["engine"] = "%s"' % m.split("_")[0], 'extra["engine"] = E') + t = t.replace("except %s as exc:" % exc[m], "except DRIVER as exc:") + tails.add(t) + equal("scrape() and the CLI are identical but for the engine's own name", + len(tails), 1) + + def check_the_modes_cannot_disagree_about_an_account(): """All three modes read one page, so their counts must agree.""" for name in ("nasa", "mrbeast", "kyliejenner", "khaby00"):