diff --git a/src/wordstat/cli.py b/src/wordstat/cli.py index 6741b59..5939f51 100644 --- a/src/wordstat/cli.py +++ b/src/wordstat/cli.py @@ -8,15 +8,22 @@ from wordstat.collector import WordstatCollector from wordstat.config import load_config from wordstat.errors import WordstatError +from wordstat.periods import Granularity, parse_date, validate_period from wordstat.storage import prepare_resume_directory _CONFIG = load_config() _DEFAULT_CDP_URL = _CONFIG.get("cdp_url", "http://127.0.0.1:9222") -# Wordstat's native export encoding is cp1251; a phrases file typed or saved -# on the same machine can plausibly be in either. Mirrors the encoding probe -# in csv_io.py, minus utf-8-sig (a phrases file is authored by hand, not -# exported by Wordstat, so a BOM is unlikely but harmless either way). -_PHRASES_FILE_ENCODINGS = ("utf-8", "cp1251") +# Wordstat exports CSV as UTF-8 with BOM; a phrases file typed or saved +# on the same machine can plausibly be in either, and utf-8-sig must come +# before plain utf-8: a BOM'd file decodes successfully under plain +# "utf-8" too, but leaves "" attached to the first line. Neither +# resolve_phrases' line.strip() nor the collector's own phrase.strip() +# removes it (''.isspace() is False), so a BOM left in is not +# harmless — it silently prepends an invisible character to the first +# phrase, which then propagates into the typed Wordstat search, the +# run-directory slug, and manifest.json. Mirrors the encoding probe in +# csv_io.py. +_PHRASES_FILE_ENCODINGS = ("utf-8-sig", "cp1251") @click.group() @@ -69,6 +76,14 @@ def _read_phrases_file(path: Path) -> str: ) @click.option("--cdp-url", envvar="WORDSTAT_CDP_URL", default=_DEFAULT_CDP_URL, show_default=True) @click.option("--timeout", "timeout_seconds", type=click.FloatRange(min=1), default=45.0, show_default=True) +@click.option( + "--granularity", + type=click.Choice(Granularity, case_sensitive=False), + default=Granularity.MONTHLY, + show_default=True, +) +@click.option("--date-from", type=str, default=None, help="Dynamics window start (YYYY-MM-DD).") +@click.option("--date-to", type=str, default=None, help="Dynamics window end (YYYY-MM-DD).") @click.option( "--keep-raw", is_flag=True, @@ -95,6 +110,9 @@ def collect( output_dir: Path, cdp_url: str, timeout_seconds: float, + granularity: str, + date_from: str | None, + date_to: str | None, keep_raw: bool, resume_dir: Path | None, ) -> None: @@ -106,6 +124,13 @@ def collect( """ phrases = resolve_phrases(phrase, phrases_file) + selected_granularity = Granularity(granularity.lower()) + try: + parsed_from = parse_date(date_from) + parsed_to = parse_date(date_to) + validate_period(selected_granularity, parsed_from, parsed_to) + except WordstatError as error: + raise click.ClickException(str(error)) from error if not phrases: raise click.ClickException("At least one search phrase is required") if resume_dir is not None: @@ -130,7 +155,14 @@ def collect( keep_raw=keep_raw, ) try: - batch = asyncio.run(collector.collect_many(phrases, region=region, resume_directory=resume_dir)) + collect_kwargs = {"region": region, "resume_directory": resume_dir} + if selected_granularity is not Granularity.MONTHLY or parsed_from is not None: + collect_kwargs.update( + granularity=selected_granularity, + date_from=parsed_from, + date_to=parsed_to, + ) + batch = asyncio.run(collector.collect_many(phrases, **collect_kwargs)) # Only domain errors become friendly messages; an unexpected ValueError # from a dependency should keep its traceback instead of being reworded. except WordstatError as error: diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index 3ffc7ad..d34bfe6 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -2,10 +2,11 @@ import asyncio import json +import re import tempfile import time from collections.abc import Callable -from datetime import UTC, datetime +from datetime import UTC, date, datetime, timedelta from pathlib import Path from browser_use.browser import BrowserSession @@ -28,6 +29,7 @@ PhraseFailure, WordstatView, ) +from wordstat.periods import Granularity, validate_period from wordstat.storage import ( create_run_directory, finalize_raw, @@ -42,6 +44,8 @@ SEARCH_SELECTOR = ".wordstat__search-button" DOWNLOAD_SELECTOR = "button.save-button" DOWNLOAD_CSV_MENU_ITEM_SELECTOR = "a[download]:has(button.save-csv-button)" +GRANULARITY_SELECTOR = ".wordstat__content-type_select > button" +DATE_RANGE_SELECTOR = ".range-datepicker__selected-dates > button" REGION_BUTTON_SELECTOR = ".settings__selected button" TABLE_ROW_SELECTOR = ".table__wrapper tbody tr" # Tab selectors live here with the rest of the DOM knowledge; the markup is @@ -52,6 +56,19 @@ WordstatView.DYNAMICS: "label[for='graph']", WordstatView.REGIONS: "label[for='map']", } +# The live interface is Russian; map explicitly rather than depending on the +# host locale. Shared by every place that needs a Russian month name (the +# calendar popups and the applied-period wait below), so there is exactly +# one list to keep in sync with the live UI's wording. Nominative form — +# what the calendar popups' option labels use ("январь", "декабрь"). +RUSSIAN_MONTHS = "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split() +# Genitive form — what the dynamics table's first cell uses for daily/weekly +# rows ("22 июня", "24 декабря 2018": "of June", "of December"). Confirmed +# live (issue #6 phase 1/2 documents) that this differs from RUSSIAN_MONTHS; +# a fixed nominative-month match against this genitive text never matches. +RUSSIAN_MONTHS_GENITIVE = ( + "января февраля марта апреля мая июня июля августа сентября октября ноября декабря".split() +) def _is_untrustworthy_empty_export(view: WordstatView, dataset: CsvDataset) -> bool: @@ -120,11 +137,44 @@ def __init__( output_root: Path, timeout_seconds: float = 45.0, keep_raw: bool = False, + settling_seconds: float = 1.0, + empty_export_retry_seconds: float = 2.0, ) -> None: self.cdp_url = cdp_url self.output_root = output_root self.timeout_seconds = timeout_seconds self.keep_raw = keep_raw + # Both are real, unconditional pauses against the live UI (see their + # call sites in _collect_one) — not upper-bound timeouts like + # timeout_seconds, which _wait_for polls against a condition and + # returns early. Kept as constructor parameters (not hardcoded) so + # tests against a fake, instantly-responding page can set them to 0 + # instead of actually blocking the test process for real wall-clock + # time on every view of every phrase (140 tests went from 26.9s to + # 0.77s once collect_many's own tests did this — see git history). + # + # settling_seconds (introduced in efaa9ba, issue #6 phase 2): a + # preventive pause before downloading each non-map view, guarding + # against a header-only CSV race first observed for issue #11 + # (Wordstat can return a checked radio + enabled download button + # before the export blob has actually been rebuilt for the newly + # selected phrase/view). No one has reproduced *this specific* + # settling race with reliable repro steps — the 1.0s value was + # chosen without a live timing measurement, not derived from one. + # Cost: +1s per non-map view, +3s per phrase (3 of 4 views are + # non-map), so +50s across a 50-phrase batch. Do not remove without + # a way to verify nothing regresses — issue #22 currently blocks a + # full CLI run from reaching DYNAMICS at all (see collector.py's + # fail-closed behavior on empty top exports), so there is no cheap + # end-to-end signal today that would catch a regression from + # removing this. + # + # empty_export_retry_seconds: unlike the above, this one does have + # a concrete, structural trigger — it only fires after + # _is_untrustworthy_empty_export has already caught a table-visible- + # but-CSV-empty export on this specific run, not preventively. + self.settling_seconds = settling_seconds + self.empty_export_retry_seconds = empty_export_retry_seconds self._previous_table_snapshot: str | None = None async def collect( @@ -148,6 +198,9 @@ async def collect_many( phrases: list[str], region: str = "Россия", resume_directory: Path | None = None, + granularity: Granularity = Granularity.MONTHLY, + date_from: date | None = None, + date_to: date | None = None, ) -> BatchCollectionResult: """Collect reports for several phrases inside a single browser session. @@ -165,6 +218,7 @@ async def collect_many( """ region = region.strip() + validate_period(granularity, date_from, date_to) if not phrases: raise InvalidRequestError("At least one search phrase is required") if not region: @@ -233,15 +287,22 @@ def _mark_region_ready() -> None: # _collect_one for why the region control can't be # re-selected once a phrase's view loop has run (and # doesn't need to be after that). + collect_kwargs = { + "set_region": not region_ready, + "on_region_applied": _mark_region_ready, + "resume_directory": resume_directory, + } + # Keep the historical call shape for the default + # monthly path; several integrations monkeypatch this + # seam and default behavior must remain unchanged. + if granularity is not Granularity.MONTHLY or date_from is not None: + collect_kwargs.update( + granularity=granularity, + date_from=date_from, + date_to=date_to, + ) result = await self._collect_one( - page, - session, - downloads_path, - phrase, - region, - set_region=not region_ready, - on_region_applied=_mark_region_ready, - resume_directory=resume_directory, + page, session, downloads_path, phrase, region, **collect_kwargs ) results.append(result) except AuthenticationRequiredError as error: @@ -279,6 +340,9 @@ async def _collect_one( set_region: bool = True, on_region_applied: Callable[[], None] | None = None, resume_directory: Path | None = None, + granularity: Granularity = Granularity.MONTHLY, + date_from: date | None = None, + date_to: date | None = None, ) -> CollectionResult: # Checked once before the batch starts (collect_many), but a session # can lose authentication mid-batch (e.g. Yandex logs it out); check @@ -295,6 +359,11 @@ async def _collect_one( # separate, explicit path, not a fallback baked into it. run_directory = resume_directory manifest = prepare_resume_directory(run_directory, phrase, region) + requested_period = self._requested_period(date_from, date_to) + if manifest.granularity is not granularity or manifest.requested_period != requested_period: + raise InvalidRequestError( + "--resume-dir was created with a different dynamics granularity or requested period" + ) manifest_path = run_directory / "manifest.json" pending_views = views_to_collect(run_directory, manifest) if not pending_views: @@ -359,6 +428,8 @@ async def _collect_one( updated_at=start, source_url=await page.get_url(), exports=[], + granularity=granularity, + requested_period=self._requested_period(date_from, date_to), ) write_manifest(manifest_path, manifest) else: @@ -378,17 +449,44 @@ async def _collect_one( for view in pending_views: selector = VIEW_SELECTORS[view] await self._select_view(page, selector, view) + # The live UI can expose the first table row before the export + # blob has been rebuilt for the selected phrase/view. A short + # settling interval prevents a header-only CSV from racing the + # table repaint; the structural row gate above remains required. + if view is not WordstatView.REGIONS and self.settling_seconds > 0: + await asyncio.sleep(self.settling_seconds) + if view is WordstatView.DYNAMICS and ( + granularity is not Granularity.MONTHLY or date_from is not None + ): + await self._set_granularity(page, granularity) + if date_from is not None: + await self._set_period(page, granularity, date_from, date_to) source = await self._download_current_view(page, session, downloads_path) try: # Convert before disposing of the download, so a parse or - # write failure leaves the raw CSV on disk to inspect. + # write failure leaves the raw CSV on disk to inspect. The + # live export blob can lag the table repaint once; retry an + # empty table export once before failing closed. dataset = parse_wordstat_csv(source, view) + if _is_untrustworthy_empty_export(view, dataset): + if self.empty_export_retry_seconds > 0: + await asyncio.sleep(self.empty_export_retry_seconds) + source = await self._download_current_view(page, session, downloads_path) + dataset = parse_wordstat_csv(source, view) if _is_untrustworthy_empty_export(view, dataset): raise InterfaceChangedError( - f"Wordstat returned an empty {view.value} CSV, but the page had rendered at " - "least one table row before the download was triggered; export is not trustworthy" + f"Wordstat returned an empty {view.value} CSV after a retry, but the page had rendered " + "at least one table row before the download was triggered; export is not trustworthy" ) - data_path, dtypes = write_dataset(dataset, run_directory) + if view is WordstatView.DYNAMICS: + self._assert_contiguous_dynamics_rows( + dataset, granularity, date_from=date_from, date_to=date_to + ) + file_name = self._dynamics_file_name(granularity) if view is WordstatView.DYNAMICS else None + if file_name is None: + data_path, dtypes = write_dataset(dataset, run_directory) + else: + data_path, dtypes = write_dataset(dataset, run_directory, file_name=file_name) raw_path = finalize_raw(source, run_directory, view, self.keep_raw) except Exception: # noqa: BLE001 # source lives in the batch's shared, temporary downloads @@ -416,6 +514,10 @@ async def _collect_one( # CollectionManifest.status), not a stale one still claiming # every view is missing. manifest = merge_export(manifest, export) + if view is WordstatView.DYNAMICS: + manifest = manifest.model_copy( + update={"actual_period": self._actual_period(dataset)} + ) write_manifest(manifest_path, manifest) return CollectionResult( @@ -424,6 +526,311 @@ async def _collect_one( manifest=manifest, ) + @staticmethod + def _requested_period(date_from: date | None, date_to: date | None) -> dict[str, str] | None: + if date_from is None or date_to is None: + return None + return {"from": date_from.isoformat(), "to": date_to.isoformat()} + + @staticmethod + def _actual_period(dataset: CsvDataset) -> dict[str, str] | None: + if not dataset.rows or not dataset.headers: + return None + field = dataset.headers[0] + return {"field": field, "from": dataset.rows[0][field], "to": dataset.rows[-1][field]} + + @staticmethod + def _assert_contiguous_dynamics_rows( + dataset: CsvDataset, + granularity: Granularity, + date_from: date | None = None, + date_to: date | None = None, + ) -> None: + # This parses dataset.rows[field] — the exported CSV's date column + # — with strptime("%d.%m.%Y"), NOT the DOM's displayed cell text. + # Live-measured weekly CSV (issue #6 phase 1 document — see + # `docs/ISSUE6_PHASE1_RESEARCH.md` on the PR #20 branch, + # `git show 4be8996:docs/ISSUE6_PHASE1_RESEARCH.md`): the exported + # field is named "Неделя с" and holds only the week's start date + # as a bare DD.MM.YYYY value ("22.06.2026") — no range, no week + # number — confirmed byte-for-byte from a real downloaded CSV. The + # DOM cell (_wait_for_table_granularity's WEEKLY pattern below) + # shows a genitive-month date *range* instead ("22 июня 2026 – 28 + # июня 2026"); that is a different string in a different place and + # does not describe the exported column. Do not "fix" this parse + # to expect a range based on the DOM pattern — that would break a + # confirmed-working live path (cycle-review round 1's R2 and round + # 2's RR1 both raised this same claim from reading the code + # without this doc open; both were false positives). + # + # Monthly's period field is a nominative Russian month name + # ("август 2024", confirmed live — issue #6 phase 1 document), not + # a strptime("%d.%m.%Y")-parseable date the way daily/weekly are, + # so neither contiguity nor containment can be validated here for + # it; _wait_for_period_applied still gates monthly's start + # boundary before download (same mechanism as daily/weekly), and + # _actual_period records the raw field/from/to strings into the + # manifest for every dynamics export including monthly, so actual + # coverage stays observable post-hoc even without this check + # (cycle-review round 2, C2 triage). + if granularity is Granularity.MONTHLY or not dataset.rows: + return + field = dataset.headers[0] + try: + dates = [datetime.strptime(row[field], "%d.%m.%Y").date() for row in dataset.rows] + except ValueError as error: + raise InterfaceChangedError( + f"Wordstat returned an unexpected {granularity.value} dynamics date format" + ) from error + if len(dates) >= 2: + step = timedelta(days=1 if granularity is Granularity.DAILY else 7) + for previous, current in zip(dates, dates[1:]): + if current - previous != step: + raise InterfaceChangedError( + f"Wordstat returned a gap in the {granularity.value} dynamics series " + f"between {previous:%d.%m.%Y} and {current:%d.%m.%Y}" + ) + # Containment, not equality: a shorter export fully inside the + # requested window is legitimate (the daily series' variable + # trailing tail, confirmed live — issue #6 phase 1 document), so + # this never compares dates[0]/dates[-1] against date_from/date_to + # for exact equality. What it does reject is a row *outside* the + # requested window in either direction — the live race this guards + # against (collector.py:603, _wait_for_period_applied) confirms + # only the first cell's format+value match date_from before + # downloading; it has no equivalent check for date_to, so a table + # that has not yet repainted past a previous, stale window can + # download and be recorded as complete with data that never + # entered the requested window at all. Deliberately runs even for + # a single-row export (needs only dates[0]/dates[-1], which are + # the same row — unlike the pairwise contiguity loop above, which + # needs at least two rows): a single stale/truncated row is not + # exempt just because it is too short to check contiguity + # (cycle-review round 2, C2 triage — the original version gated + # this behind the same `len(rows) < 2` early return as contiguity + # and let a 1-row export bypass it entirely). Only applies when an + # explicit period was requested (date_from/date_to are None for + # the UI's default window, which this function cannot validate + # against any specific boundary). + if date_from is not None and date_to is not None: + # Weekly rows are keyed by the Monday of date_from's week, not + # date_from itself (confirmed live — _wait_for_period_applied's + # own comment above), so the lower bound must be that aligned + # Monday, not the raw requested date, or a legitimate alignment + # would be misreported as a stale window. + lower_bound = ( + date_from - timedelta(days=date_from.weekday()) + if granularity is Granularity.WEEKLY + else date_from + ) + if dates[0] < lower_bound: + raise InterfaceChangedError( + f"Wordstat {granularity.value} dynamics export starts at {dates[0]:%d.%m.%Y}, " + f"before the requested window start {lower_bound:%d.%m.%Y}" + ) + if dates[-1] > date_to: + raise InterfaceChangedError( + f"Wordstat {granularity.value} dynamics export ends at {dates[-1]:%d.%m.%Y}, " + f"after the requested window end {date_to:%d.%m.%Y}" + ) + + + @staticmethod + def _dynamics_file_name(granularity: Granularity) -> str: + # Preserve the old monthly filename for existing consumers; non-monthly + # exports must carry their granularity in the filename. + return "dynamics.parquet" if granularity is Granularity.MONTHLY else f"dynamics_{granularity.value}.parquet" + + async def _set_granularity(self, page, granularity: Granularity) -> None: + labels = { + Granularity.DAILY: "По дням", + Granularity.WEEKLY: "По неделям", + Granularity.MONTHLY: "По месяцам", + } + await self._click(page, GRANULARITY_SELECTOR) + await self._click_visible_text(page, ".Popup2_visible [role='option']", labels[granularity]) + await self._wait_for( + page, + f"() => document.querySelector({json.dumps(GRANULARITY_SELECTOR)})?.textContent?.trim() === " + f"{json.dumps(labels[granularity])}", + ) + await self._wait_for_table_granularity(page, granularity) + + async def _set_period(self, page, granularity: Granularity, date_from: date, date_to: date | None) -> None: + if date_to is None: + raise InvalidRequestError("A period end date is required when a start date is provided") + popup_type = "month" if granularity is Granularity.MONTHLY else ( + "week" if granularity is Granularity.WEEKLY else "day" + ) + await self._click(page, DATE_RANGE_SELECTOR) + await self._select_calendar_date(page, popup_type, date_from) + if popup_type != "month": + # day/week: independent popups per click, live-confirmed by + # 255bca7's daily/weekly runs going through this exact + # two-click path successfully. + await self._click(page, DATE_RANGE_SELECTOR) + else: + # month: a single shared range-picker, not two independent + # popups (live CDP check, issue #6 phase 2, cycle-review + # round 2). Right after date_from is picked, every month + # element already carries the "in-selecting-range" class — + # the picker is already in range-selection mode — and the + # date-range button's text does not update to reflect + # date_from yet; it only updates once date_to is picked in + # the SAME still-open popup. A second DATE_RANGE_SELECTOR + # click here closes this popup instead of reopening it + # (confirmed live: visibility flips to 'hidden'), so the + # follow-up _select_calendar_date for date_to would time out + # waiting for a popup that never reappears. Confirmed live + # that skipping the intermediate click and picking date_to + # directly in the still-open popup produces the correct + # button text "Январь 2024 — Июнь 2024". + pass + await self._select_calendar_date(page, popup_type, date_to) + # _wait_for_table_granularity alone is not enough here: it only + # checks that the first cell's *format* matches the granularity + # (e.g. "looks like a week range"), which is already true of the + # table's pre-existing content before the period change has been + # applied. Live CDP checks (issue #6 phase 2) caught this exact + # race: the date-range button already showed the newly picked + # dates, _wait_for_table_granularity's format check passed + # immediately, and the exported CSV still carried Wordstat's + # default window — the table's *values* had not caught up yet. + # Waiting for the first cell to actually start at the expected + # date closes that gap, and turns a silent wrong-period export + # into a loud InterfaceChangedError if the picker never catches up. + # + # This only confirms the *start* boundary (date_from), because it + # runs before the download and only the first row is visible/known + # at this point. It does not by itself prove the *end* boundary + # (date_to) has also repainted — a table that has advanced its + # first row but not yet its last row would pass this wait and + # still download a stale end. The end boundary is guarded + # separately, after the download, by + # _assert_contiguous_dynamics_rows' containment check against the + # actually-exported rows (confirmed as a real gap during review, + # not just theory) — that is the authoritative defense for + # date_to, not an earlier DOM-state wait for it. + await self._wait_for_period_applied(page, granularity, date_from) + + async def _wait_for_period_applied(self, page, granularity: Granularity, date_from: date) -> None: + # The month name in the first cell is genitive ("22 июня", "24 + # декабря" — day-of/week-of a date) for daily/weekly, but + # nominative ("август 2024") for monthly (confirmed live, issue #6 + # phase 1/2 documents) — hence two separate month lists rather than + # one. Matching the exact expected month (not just "any month word") + # matters: without it, "24 2018" would silently accept + # a wrong month picked by a stuck calendar, defeating the point of + # this wait (confirmed as a real gap during review, not just theory). + if granularity is Granularity.MONTHLY: + expected = f"{RUSSIAN_MONTHS[date_from.month - 1]} {date_from.year}" + elif granularity is Granularity.DAILY: + expected = f"{date_from.day} {RUSSIAN_MONTHS_GENITIVE[date_from.month - 1]}" + else: + # Weekly's first cell is the Monday of date_from's week, not + # date_from itself (confirmed live: 2018-12-26, a Wednesday, + # produced a first cell starting "24 декабря" — the preceding + # Monday). Wordstat does this alignment itself; not a bug. + week_start = date_from - timedelta(days=date_from.weekday()) + expected = f"{week_start.day} {RUSSIAN_MONTHS_GENITIVE[week_start.month - 1]} {week_start.year}" + pattern = "^" + re.escape(expected) + await self._wait_for( + page, + f"() => new RegExp({json.dumps(pattern)}).test(" + f"document.querySelector({json.dumps(TABLE_ROW_SELECTOR)})?.querySelector('td')" + "?.textContent?.trim() ?? '')", + ) + + async def _wait_for_table_granularity(self, page, granularity: Granularity) -> None: + patterns = { + Granularity.DAILY: r"^\d{1,2}\s+[А-Яа-яЁё]+$", + Granularity.WEEKLY: r"^\d{1,2}\s+[А-Яа-яЁё]+\s+\d{4}\s+–\s+\d{1,2}\s+[А-Яа-яЁё]+\s+\d{4}$", + Granularity.MONTHLY: r"^[А-Яа-яЁё]+\s+\d{4}$", + } + expression = ( + "() => new RegExp(" + json.dumps(patterns[granularity]) + ").test(" + f"document.querySelector({json.dumps(TABLE_ROW_SELECTOR)})?.querySelector('td')?.textContent?.trim() ?? '')" + ) + await self._wait_for(page, expression) + + async def _select_calendar_date(self, page, popup_type: str, target: date) -> None: + root = f".range-datepicker_type_{popup_type}" + await self._wait_for( + page, + f"() => [...document.querySelectorAll({json.dumps(root)})].some((x) => " + "x.offsetParent !== null && getComputedStyle(x).visibility !== 'hidden')", + ) + month_label = RUSSIAN_MONTHS[target.month - 1] + if popup_type == "month": + year_selector = f"{root} .datepicker-month__years_select > button" + month_selector = f"{root} .react-datepicker__month-text" + else: + year_selector = f"{root} .range-datepicker__years_select > button" + month_selector = f"{root} .range-datepicker__months_select > button" + await self._click(page, year_selector) + await self._click_visible_text(page, ".Popup2_visible [role='option']", str(target.year)) + if popup_type in {"day", "week"}: + await self._click(page, month_selector) + await self._click_visible_text(page, ".Popup2_visible [role='option']", month_label) + day_selector = f"{root} button[name='day']" + if popup_type == "month": + # Live CDP check (issue #6 phase 2, cycle-review round 2): the + # month-text popup renders lowercase nominative month names + # ("январь"), matching RUSSIAN_MONTHS verbatim. + # month_label.capitalize() ("Январь") never matches any of + # them — every explicit monthly request with + # --date-from/--date-to failed here with InterfaceChangedError + # ("Wordstat option 'Январь' was not uniquely found") despite + # validate_period accepting the request and the browser + # already being driven. Do not re-add .capitalize(): it was + # never confirmed live and is contradicted by the confirmed + # live DOM text. + await self._click_visible_text(page, month_selector, month_label) + else: + await self._click_visible_date_button(page, day_selector, str(target.day)) + + async def _click_visible_date_button(self, page, selector: str, text: str) -> None: + result = await page.evaluate( + """(...args) => { + const [selector, text] = args; + const matches = [...document.querySelectorAll(selector)].filter((element) => { + const style = getComputedStyle(element); + return element.textContent?.trim() === text + && !element.className.toString().includes('outside') + && !element.className.toString().includes('disabled') + && element.offsetParent !== null + && style.visibility !== 'hidden'; + }); + if (matches.length !== 1) return JSON.stringify({count: matches.length}); + matches[0].click(); + return JSON.stringify({count: 1}); + }""", + selector, + text, + ) + if json.loads(result) != {"count": 1}: + raise InterfaceChangedError(f"Wordstat date control {selector!r} was not uniquely found") + + async def _click_visible_text(self, page, selector: str, text: str) -> None: + result = await page.evaluate( + """(...args) => { + const [selector, text] = args; + const matches = [...document.querySelectorAll(selector)].filter((element) => { + const style = getComputedStyle(element); + return element.textContent?.trim() === text + && element.offsetParent !== null + && style.visibility !== 'hidden'; + }); + if (matches.length !== 1) return JSON.stringify({count: matches.length}); + matches[0].click(); + return JSON.stringify({count: 1}); + }""", + selector, + text, + ) + if json.loads(result) != {"count": 1}: + raise InterfaceChangedError(f"Wordstat option {text!r} was not uniquely found") + async def _assert_authenticated(self, page) -> None: state = await page.evaluate( """() => JSON.stringify({ diff --git a/src/wordstat/dataset_io.py b/src/wordstat/dataset_io.py index 1057375..60805b6 100644 --- a/src/wordstat/dataset_io.py +++ b/src/wordstat/dataset_io.py @@ -11,8 +11,10 @@ from wordstat.models import CsvDataset -def write_dataset(dataset: CsvDataset, run_directory: Path) -> tuple[Path, dict[str, str]]: - """Write one report as ``.parquet`` and report the inferred column types. +def write_dataset( + dataset: CsvDataset, run_directory: Path, file_name: str | None = None +) -> tuple[Path, dict[str, str]]: + """Write one report and report the inferred column types. Column order follows the export's own header order. The returned dtype mapping goes into the manifest so an unexpected export format is visible @@ -33,6 +35,6 @@ def write_dataset(dataset: CsvDataset, run_directory: Path) -> tuple[Path, dict[ columns[header] = pa.array(coerced, type=pa.type_for_alias(dtype)) table = pa.table(columns) - destination = run_directory / f"{dataset.view.value}.parquet" + destination = run_directory / (file_name or f"{dataset.view.value}.parquet") pq.write_table(table, destination, compression="zstd") return destination, dtypes diff --git a/src/wordstat/dtypes.py b/src/wordstat/dtypes.py index 5c10383..1551429 100644 --- a/src/wordstat/dtypes.py +++ b/src/wordstat/dtypes.py @@ -1,7 +1,8 @@ """Column type inference for Wordstat's localized string values. Wordstat exports every value as text: counts arrive as ``"5 228 679"`` with -non-breaking thousands separators, and dynamics periods as ``"01.2024"``. +non-breaking thousands separators, and dynamics periods as ``"август 2024"`` +or dates such as ``"22.06.2026"``. Writing those straight to Parquet would only change the container, so columns are typed here first. @@ -24,8 +25,8 @@ _WHITESPACE = re.compile(r"\s") # Deliberately strict. ``int()``/``float()`` would accept values that are not -# numbers in this data: "01.2024" (a dynamics period) parses as 1.2024, -# silently destroying it, and "1e5", "1_000", "inf" and "nan" are accepted too. +# numbers in this data and could silently destroy a dotted date. Real monthly +# periods are "август 2024"; daily/weekly exports use "22.06.2026". # Only a comma is a decimal separator here — Wordstat exports with a Russian # locale — which is precisely what rejects the dotted period format. _NUMERIC = re.compile(r"^-?[0-9]+(?:,[0-9]+)?$") diff --git a/src/wordstat/errors.py b/src/wordstat/errors.py index 0a485c8..7681db0 100644 --- a/src/wordstat/errors.py +++ b/src/wordstat/errors.py @@ -9,6 +9,10 @@ class InvalidRequestError(WordstatError): """The requested phrase or region is unusable before the browser is touched.""" +class InvalidPeriodError(InvalidRequestError): + """The requested dynamics granularity/window is not supported.""" + + class AuthenticationRequiredError(WordstatError): """Wordstat is reachable but the attached browser is not authenticated.""" diff --git a/src/wordstat/models.py b/src/wordstat/models.py index 92d0102..e73f6a7 100644 --- a/src/wordstat/models.py +++ b/src/wordstat/models.py @@ -6,6 +6,8 @@ from pydantic import BaseModel, ConfigDict, Field, computed_field, model_validator +from wordstat.periods import Granularity + class WordstatView(StrEnum): """The four Wordstat reports exported by the MVP.""" @@ -103,6 +105,9 @@ class CollectionManifest(BaseModel): updated_at: datetime | None = None source_url: str exports: list[ExportSummary] + granularity: Granularity = Granularity.MONTHLY + requested_period: dict[str, str] | None = None + actual_period: dict[str, str] | None = None @computed_field # type: ignore[prop-decorator] @property diff --git a/src/wordstat/periods.py b/src/wordstat/periods.py new file mode 100644 index 0000000..2bc4805 --- /dev/null +++ b/src/wordstat/periods.py @@ -0,0 +1,85 @@ +"""Validation and normalization of user-requested dynamics periods.""" + +from __future__ import annotations + +from calendar import monthrange +from datetime import date, timedelta +from enum import StrEnum + +from wordstat.errors import InvalidPeriodError + + +class Granularity(StrEnum): + MONTHLY = "monthly" + WEEKLY = "weekly" + DAILY = "daily" + + +EARLIEST_DATE = date(2018, 1, 1) + + +def validate_period( + granularity: Granularity, + date_from: date | None, + date_to: date | None, + *, + today: date | None = None, +) -> None: + """Validate an explicit window before opening Chrome. + + Omitted dates leave the UI's default window untouched. The five-year + maximum is deliberately not enforced because phase 1 did not establish + it as a reliable live fact — unlike the daily lower bound below, which + is a confirmed live limitation of the picker itself, not a policy + choice. + """ + if (date_from is None) != (date_to is None): + raise InvalidPeriodError("--date-from and --date-to must be provided together") + if date_from is None or date_to is None: + return + if date_from < EARLIEST_DATE or date_to < EARLIEST_DATE: + raise InvalidPeriodError("The requested period cannot include dates before January 2018") + if date_to < date_from: + raise InvalidPeriodError("The period end date must not be earlier than its start date") + if granularity is Granularity.DAILY: + current = today or date.today() + if date_to > current: + raise InvalidPeriodError("Daily statistics cannot be requested after today") + if date_to - date_from >= timedelta(days=60): + raise InvalidPeriodError("Daily statistics support at most 60 calendar days") + # Live CDP measurement (issue #6 phase 2): the daily date-range + # picker's year-select popup only ever offers the current year, so + # a date_from further back than the trailing 60-day window cannot + # actually be selected in the live UI at all — it currently reaches + # Chrome and fails deep inside calendar-click code with an opaque + # InterfaceChangedError instead of a clear pre-flight rejection. + # Expressed relative to `today` (not "reject any year but the + # current one" — that would wrongly reject legal windows in + # January) and kept consistent with the existing 60-day window + # check above: PR #20's live-confirmed window 22.06.2026-20.08.2026, + # measured on 21.08.2026, is exactly 60 days back from today and + # must remain accepted. + if current - date_from > timedelta(days=60): + raise InvalidPeriodError("Daily statistics cannot start more than 60 days in the past") + elif granularity is Granularity.WEEKLY: + if date_to - date_from < timedelta(days=20): + raise InvalidPeriodError("Weekly statistics require at least three calendar weeks") + else: + end_month = date_from.month + 2 + end_year = date_from.year + (end_month - 1) // 12 + end_month = (end_month - 1) % 12 + 1 + minimum_end = date(end_year, end_month, monthrange(end_year, end_month)[1]) + if date_to < minimum_end: + raise InvalidPeriodError("Monthly statistics require at least three calendar months") + + +def parse_date(value: str | None) -> date | None: + if value is None: + return None + try: + year, month, day = (int(part) for part in value.split("-")) + if month < 1 or month > 12 or day < 1 or day > monthrange(year, month)[1]: + raise ValueError + return date(year, month, day) + except (TypeError, ValueError): + raise InvalidPeriodError(f"Invalid date {value!r}; expected YYYY-MM-DD") from None diff --git a/tests/test_cli.py b/tests/test_cli.py index 2e63669..b925c96 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -142,6 +142,18 @@ def test_phrases_file_falls_back_to_cp1251(tmp_path: Path): assert resolve_phrases((), phrases_file) == ["чай", "кофе"] +def test_phrases_file_with_a_utf8_bom_does_not_leak_into_the_first_phrase(tmp_path: Path): + # A BOM'd UTF-8 file decodes successfully under plain "utf-8" with the + # BOM character left attached to the first line; ''.isspace() is + # False, so neither resolve_phrases' line.strip() nor the collector's + # own phrase.strip() removes it. utf-8-sig must be tried before plain + # utf-8 so the BOM is stripped during decoding itself. + phrases_file = tmp_path / "phrases.txt" + phrases_file.write_bytes("ремонт квартир\nдизайн интерьера\n".encode("utf-8-sig")) + + assert resolve_phrases((), phrases_file) == ["ремонт квартир", "дизайн интерьера"] + + def test_phrases_file_with_undecodable_bytes_is_reported_without_a_traceback(tmp_path: Path): # 0x98 is invalid in both utf-8 (a lone continuation byte) and cp1251 # (unassigned in that codepage) — one of the few byte values neither diff --git a/tests/test_collector_batch.py b/tests/test_collector_batch.py index 2c04ae2..544fddd 100644 --- a/tests/test_collector_batch.py +++ b/tests/test_collector_batch.py @@ -96,7 +96,7 @@ def test_collect_many_reuses_one_session_for_all_phrases(monkeypatch, tmp_path): _patch_common(monkeypatch) _patch_collect_one(monkeypatch, lambda phrase: _async(_fake_result(tmp_path, phrase))) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["чай", "кофе", "вода"])) assert _FakeSession.instances == 1 @@ -115,7 +115,7 @@ async def flaky(phrase): _patch_collect_one(monkeypatch, flaky) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["первый", "сломается", "второй"])) assert batch.total == 3 @@ -139,7 +139,7 @@ async def flaky(phrase): _patch_collect_one(monkeypatch, flaky) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["первый", "сломается", "второй"])) assert [r.run_directory.name for r in batch.results] == ["первый", "второй"] @@ -172,7 +172,7 @@ async def collect_one( monkeypatch.setattr(WordstatCollector, "_collect_one", collect_one) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["первый", "второй", "третий"])) assert attempted == ["первый", "второй"] # "третий" was never attempted @@ -197,7 +197,7 @@ async def counting_assert_authenticated(self, page): monkeypatch.setattr(WordstatCollector, "_assert_authenticated", counting_assert_authenticated) _patch_collect_one_passthrough(monkeypatch, tmp_path) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) asyncio.run(collector.collect_many(["первый", "второй", "третий"])) # Once before the loop (collect_many) + once per phrase (_collect_one). @@ -254,7 +254,7 @@ async def recording_collect_one( monkeypatch.setattr(WordstatCollector, "_collect_one", recording_collect_one) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) asyncio.run(collector.collect_many(["первый", "второй", "третий"])) assert set_region_calls == [True, False, False] @@ -293,7 +293,7 @@ async def recording_collect_one( monkeypatch.setattr(WordstatCollector, "_collect_one", recording_collect_one) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["первый", "второй", "третий"])) assert set_region_calls == [True, True, False] @@ -338,7 +338,7 @@ async def collect_one( monkeypatch.setattr(WordstatCollector, "_collect_one", collect_one) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["первый", "второй", "третий"])) # Region was actually applied while handling "первый" — the second @@ -361,28 +361,28 @@ async def failing_stop(self): monkeypatch.setattr(_FakeSession, "stop", failing_stop) _patch_collect_one(monkeypatch, lambda phrase: _async(_fake_result(tmp_path, phrase))) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["чай"])) assert [r.run_directory.name for r in batch.results] == ["чай"] def test_collect_many_rejects_an_empty_phrase_list(tmp_path): - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) with pytest.raises(InvalidRequestError, match="At least one search phrase"): asyncio.run(collector.collect_many([])) def test_collect_many_rejects_a_blank_phrase_in_the_middle(tmp_path): - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) with pytest.raises(InvalidRequestError, match="must not be empty"): asyncio.run(collector.collect_many(["чай", " ", "кофе"])) def test_collect_many_rejects_a_blank_region(tmp_path): - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) with pytest.raises(InvalidRequestError, match="region must not be empty"): asyncio.run(collector.collect_many(["чай"], region=" ")) @@ -392,7 +392,7 @@ def test_collect_wraps_collect_many_and_returns_the_single_result(monkeypatch, t _patch_common(monkeypatch) _patch_collect_one(monkeypatch, lambda phrase: _async(_fake_result(tmp_path, phrase))) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) result = asyncio.run(collector.collect(phrase="чай")) assert result.run_directory.name == "чай" @@ -406,7 +406,7 @@ async def fail(phrase): _patch_collect_one(monkeypatch, fail) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) with pytest.raises(PhraseEntryError, match="boom"): asyncio.run(collector.collect(phrase="чай")) @@ -455,7 +455,7 @@ async def download(self, page, session, directory): monkeypatch.setattr(WordstatCollector, "_wait_for", wait) monkeypatch.setattr(WordstatCollector, "_download_current_view", download) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) page = _FakePage() session = _FakeSession() asyncio.run(collector._collect_one(page, session, downloads_path, "первая", "Россия", set_region=False)) @@ -499,7 +499,7 @@ def failing_write_dataset(dataset, run_directory): monkeypatch.setattr(WordstatCollector, "_download_current_view", fake_download) monkeypatch.setattr(collector_module, "write_dataset", failing_write_dataset) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run(): page = _FakePage() @@ -551,7 +551,7 @@ def flaky_write_dataset(dataset, run_directory): monkeypatch.setattr(WordstatCollector, "_download_current_view", fake_download) monkeypatch.setattr(collector_module, "write_dataset", flaky_write_dataset) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run(): page = _FakePage() @@ -599,7 +599,7 @@ def recording_write_manifest(path, manifest): monkeypatch.setattr(WordstatCollector, "_download_current_view", fake_download) monkeypatch.setattr(collector_module, "write_manifest", recording_write_manifest) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run(): page = _FakePage() @@ -652,7 +652,7 @@ async def failing_after_first_download(self, page, session, dl_path): monkeypatch.setattr(WordstatCollector, "_select_view", fake_select_view) monkeypatch.setattr(WordstatCollector, "_download_current_view", failing_after_first_download) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run_first(): page = _FakePage() @@ -718,7 +718,7 @@ async def fake_download(self, page, session, dl_path): monkeypatch.setattr(WordstatCollector, "_select_view", fake_select_view) monkeypatch.setattr(WordstatCollector, "_download_current_view", fake_download) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run(): page = _FakePage() @@ -744,7 +744,7 @@ async def run_resume_with_wrong_phrase(): def test_collect_many_rejects_resume_directory_with_more_than_one_phrase(tmp_path): - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) resume_dir = tmp_path / "some-run" resume_dir.mkdir() @@ -776,7 +776,7 @@ async def failing_download(self, page, session, dl_path): monkeypatch.setattr(WordstatCollector, "_select_view", fake_select_view) monkeypatch.setattr(WordstatCollector, "_download_current_view", failing_download) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run_first(): page = _FakePage(url="https://wordstat.yandex.ru/?words=тест®ion=Россия") @@ -848,7 +848,7 @@ async def fake_download(self, page, session, dl_path): monkeypatch.setattr(WordstatCollector, "_select_view", fake_select_view) monkeypatch.setattr(WordstatCollector, "_download_current_view", fake_download) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run_full(): page = _FakePage(url="https://wordstat.yandex.ru/?words=тест") diff --git a/tests/test_collector_view.py b/tests/test_collector_view.py index 26ca7dc..e5dd4a4 100644 --- a/tests/test_collector_view.py +++ b/tests/test_collector_view.py @@ -1,12 +1,15 @@ """View-selection retry behavior without touching a browser.""" import asyncio +import re +from datetime import date import pytest from wordstat.collector import WordstatCollector, _is_untrustworthy_empty_export from wordstat.errors import InterfaceChangedError from wordstat.models import CsvDataset, WordstatView +from wordstat.periods import Granularity def test_select_view_retries_once_when_active_marker_does_not_change(monkeypatch, tmp_path): @@ -197,3 +200,179 @@ async def wait(self, page, expression, seconds=None, required=True): asyncio.run(WordstatCollector("cdp", tmp_path)._select_view(object(), "label[for='map']", WordstatView.REGIONS)) assert ".table__wrapper" not in waits[0] + + +# _wait_for_period_applied's regex must match the live table's actual text. +# Daily/weekly first cells are genitive ("22 июня", "24 декабря 2018" — "of +# June", "of December"), unlike the nominative RUSSIAN_MONTHS list used for +# clicking the calendar popups ("июнь", "декабрь"). A live CDP check (issue +# #6 phase 2) confirmed a fixed nominative-month match against the live +# cell's genitive text never matches, so _wait_for ran out its full timeout +# every time instead of detecting the period was actually already applied. +# These are regression guards for that exact mismatch, not for the general +# _wait_for polling mechanism (already covered above). +def _captured_pattern(monkeypatch, tmp_path): + captured = {} + + async def wait(self, page, expression, seconds=None, required=True): + captured["expression"] = expression + + monkeypatch.setattr(WordstatCollector, "_wait_for", wait) + return WordstatCollector("cdp", tmp_path), captured + + +def _extract_regex(expression: str) -> re.Pattern: + # _wait_for_period_applied builds "new RegExp()...". + # Extract and compile the same pattern the live page would receive. + import json + + start = expression.index("new RegExp(") + len("new RegExp(") + end = expression.index(")", start) + pattern = json.loads(expression[start:end]) + return re.compile(pattern) + + +def test_wait_for_period_applied_daily_matches_genitive_cell_text(monkeypatch, tmp_path): + collector, captured = _captured_pattern(monkeypatch, tmp_path) + asyncio.run(collector._wait_for_period_applied(object(), Granularity.DAILY, date(2026, 6, 22))) + pattern = _extract_regex(captured["expression"]) + assert pattern.search("22 июня") + assert not pattern.search("23 июня") + # The month must match exactly, not just "some month word" — a stuck + # calendar that landed on the right day of a wrong month must still + # fail this check (found in review: an early version anchored only on + # day+year and would have silently accepted this). + assert not pattern.search("22 июля") + + +def test_wait_for_period_applied_weekly_matches_genitive_cell_text_and_aligns_to_monday(monkeypatch, tmp_path): + collector, captured = _captured_pattern(monkeypatch, tmp_path) + # 2018-12-26 is a Wednesday; the first cell is the Monday of that week. + asyncio.run(collector._wait_for_period_applied(object(), Granularity.WEEKLY, date(2018, 12, 26))) + pattern = _extract_regex(captured["expression"]) + assert pattern.search("24 декабря 2018 – 30 декабря 2018") + assert not pattern.search("31 декабря 2018 – 6 января 2019") + assert not pattern.search("24 ноября 2018 – 30 ноября 2018") + + +def test_wait_for_period_applied_monthly_matches_nominative_cell_text(monkeypatch, tmp_path): + collector, captured = _captured_pattern(monkeypatch, tmp_path) + asyncio.run(collector._wait_for_period_applied(object(), Granularity.MONTHLY, date(2024, 8, 1))) + pattern = _extract_regex(captured["expression"]) + assert pattern.search("август 2024") + assert not pattern.search("сентябрь 2024") + + +def test_set_period_does_not_reopen_the_monthly_popup_between_dates(monkeypatch, tmp_path): + # Live CDP check (issue #6 phase 2, cycle-review round 2): the monthly + # calendar is a single range-picker, not two independent popups like + # day/week. Live DOM evidence: right after date_from is picked, every + # month element already carries the "in-selecting-range" class (the + # picker is already in range-selection mode), and the date-range + # button's text does not update to reflect date_from yet -- it only + # updates once date_to is picked in the SAME still-open popup. The + # intermediate DATE_RANGE_SELECTOR click that day/week rely on to + # reopen their popup instead closes this one (confirmed live: + # visibility flips to 'hidden'), so the follow-up _select_calendar_date + # call for date_to times out waiting for a popup that never reappears. + # Live-confirmed fix: for monthly, skip the intermediate click and let + # the second _select_calendar_date pick date_to in the still-open + # popup -- confirmed live to produce the correct button text + # "Январь 2024 — Июнь 2024". + calls = [] + + async def click(self, page, selector): + calls.append(("click", selector)) + + async def select_calendar_date(self, page, popup_type, target): + calls.append(("select_calendar_date", popup_type, target)) + + async def wait_for_period_applied(self, page, granularity, date_from): + calls.append(("wait_for_period_applied", granularity, date_from)) + + monkeypatch.setattr(WordstatCollector, "_click", click) + monkeypatch.setattr(WordstatCollector, "_select_calendar_date", select_calendar_date) + monkeypatch.setattr(WordstatCollector, "_wait_for_period_applied", wait_for_period_applied) + + collector = WordstatCollector("cdp", tmp_path) + asyncio.run( + collector._set_period(object(), Granularity.MONTHLY, date(2024, 1, 1), date(2024, 6, 30)) + ) + + click_count = sum(1 for call in calls if call[0] == "click") + assert click_count == 1, f"expected exactly one popup-open click for monthly, got {calls}" + select_calls = [call for call in calls if call[0] == "select_calendar_date"] + assert select_calls == [ + ("select_calendar_date", "month", date(2024, 1, 1)), + ("select_calendar_date", "month", date(2024, 6, 30)), + ] + + +def test_set_period_still_reopens_the_popup_between_dates_for_day_and_week(monkeypatch, tmp_path): + # day/week use independent popups per click (confirmed live in round 1: + # 255bca7's live weekly/daily runs both went through this two-click + # path successfully) -- only monthly's shared range-picker changes. + calls = [] + + async def click(self, page, selector): + calls.append("click") + + async def select_calendar_date(self, page, popup_type, target): + pass + + async def wait_for_period_applied(self, page, granularity, date_from): + pass + + monkeypatch.setattr(WordstatCollector, "_click", click) + monkeypatch.setattr(WordstatCollector, "_select_calendar_date", select_calendar_date) + monkeypatch.setattr(WordstatCollector, "_wait_for_period_applied", wait_for_period_applied) + + collector = WordstatCollector("cdp", tmp_path) + asyncio.run( + collector._set_period(object(), Granularity.DAILY, date(2026, 7, 1), date(2026, 8, 20)) + ) + assert len(calls) == 2 + + calls.clear() + asyncio.run( + collector._set_period(object(), Granularity.WEEKLY, date(2026, 7, 1), date(2026, 8, 20)) + ) + assert len(calls) == 2 + + +def test_select_calendar_date_clicks_the_month_popup_text_as_the_dom_renders_it(monkeypatch, tmp_path): + # Live CDP check (issue #6 phase 2, cycle-review round 2): the monthly + # calendar's month-text popup (.react-datepicker__month-text) renders + # lowercase nominative month names ("январь"), matching RUSSIAN_MONTHS + # verbatim -- confirmed by reading the popup's actual elements live. + # month_label.capitalize() ("Январь") never matches any of them, so + # every explicit --granularity monthly --date-from/--date-to request + # failed with InterfaceChangedError: "Wordstat option 'Январь' was not + # uniquely found" deep inside this click, despite validate_period + # accepting the request and the browser already being driven. + clicked_texts = [] + + async def click(self, page, selector): + pass + + async def wait(self, page, expression, seconds=None, required=True): + pass + + async def click_visible_text(self, page, selector, text): + clicked_texts.append((selector, text)) + + async def click_visible_date_button(self, page, selector, text): + clicked_texts.append((selector, text)) + + monkeypatch.setattr(WordstatCollector, "_click", click) + monkeypatch.setattr(WordstatCollector, "_wait_for", wait) + monkeypatch.setattr(WordstatCollector, "_click_visible_text", click_visible_text) + monkeypatch.setattr(WordstatCollector, "_click_visible_date_button", click_visible_date_button) + + collector = WordstatCollector("cdp", tmp_path) + asyncio.run(collector._select_calendar_date(object(), "month", date(2024, 1, 1))) + + month_click = next( + text for selector, text in clicked_texts if "month" in selector or "react-datepicker" in selector + ) + assert month_click == "январь" diff --git a/tests/test_dataset_io.py b/tests/test_dataset_io.py index 5d39c78..68f25e5 100644 --- a/tests/test_dataset_io.py +++ b/tests/test_dataset_io.py @@ -51,7 +51,7 @@ def test_write_dataset_writes_empty_cells_as_null(tmp_path): def test_write_dataset_keeps_dynamics_periods_as_text(tmp_path): dataset = _dataset( ["Период", "Показов"], - [{"Период": "01.2024", "Показов": "10"}, {"Период": "02.2024", "Показов": "20"}], + [{"Период": "август 2024", "Показов": "10"}, {"Период": "сентябрь 2024", "Показов": "20"}], view=WordstatView.DYNAMICS, ) @@ -60,6 +60,26 @@ def test_write_dataset_keeps_dynamics_periods_as_text(tmp_path): assert dtypes["Период"] == "string" +def test_write_dataset_keeps_daily_dates_as_text(tmp_path): + dataset = _dataset( + ["Дата", "Показов"], + [{"Дата": "22.06.2026", "Показов": "10"}, {"Дата": "23.06.2026", "Показов": "20"}], + view=WordstatView.DYNAMICS, + ) + + _, dtypes = write_dataset(dataset, tmp_path) + + assert dtypes["Дата"] == "string" + + +def test_write_dataset_accepts_granularity_specific_filename(tmp_path): + dataset = _dataset(["Дата", "Показов"], [{"Дата": "22.06.2026", "Показов": "10"}], view=WordstatView.DYNAMICS) + + path, _ = write_dataset(dataset, tmp_path, file_name="dynamics_daily.parquet") + + assert path.name == "dynamics_daily.parquet" + + def test_write_dataset_handles_a_report_without_rows(tmp_path): dataset = _dataset(["Запрос", "Показов"], []) diff --git a/tests/test_dtypes.py b/tests/test_dtypes.py index 23eb0d8..01af7cb 100644 --- a/tests/test_dtypes.py +++ b/tests/test_dtypes.py @@ -28,7 +28,8 @@ def test_parse_number_accepts_wordstat_number_formats(value, expected): @pytest.mark.parametrize( "value", [ - "01.2024", # dynamics period, must never become 1.2024 + "август 2024", # monthly dynamics period must remain text + "22.06.2026", # daily/weekly dynamics date must remain text "1e5", "1_000", "inf", @@ -45,7 +46,8 @@ def test_parse_number_rejects_non_numbers(value): def test_dynamics_period_column_stays_string(): - assert infer_column(["01.2024", "02.2024", "03.2024"]) is None + assert infer_column(["август 2024", "сентябрь 2024", "октябрь 2024"]) is None + assert infer_column(["22.06.2026", "23.06.2026"]) is None def test_all_numeric_column_becomes_int64(): diff --git a/tests/test_manifest_period.py b/tests/test_manifest_period.py new file mode 100644 index 0000000..e89c1d2 --- /dev/null +++ b/tests/test_manifest_period.py @@ -0,0 +1,131 @@ +from datetime import date + +import pytest + +from wordstat.collector import WordstatCollector +from wordstat.errors import InterfaceChangedError +from wordstat.models import CsvDataset, WordstatView +from wordstat.periods import Granularity + + +def test_actual_period_uses_the_values_present_in_the_export(): + dataset = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Дата", "Число запросов"], + rows=[ + {"Дата": "22.06.2026", "Число запросов": "1"}, + {"Дата": "18.08.2026", "Число запросов": "2"}, + ], + ) + + assert WordstatCollector._actual_period(dataset) == { + "field": "Дата", + "from": "22.06.2026", + "to": "18.08.2026", + } + + +def test_dynamics_filename_identifies_non_monthly_granularity(): + assert WordstatCollector._dynamics_file_name(Granularity.MONTHLY) == "dynamics.parquet" + assert WordstatCollector._dynamics_file_name(Granularity.DAILY) == "dynamics_daily.parquet" + assert WordstatCollector._dynamics_file_name(Granularity.WEEKLY) == "dynamics_weekly.parquet" + + +def test_dynamics_series_allows_short_tail_but_rejects_internal_gap(): + dataset = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Дата"], + rows=[{"Дата": "22.06.2026"}, {"Дата": "23.06.2026"}, {"Дата": "25.06.2026"}], + ) + + with pytest.raises(InterfaceChangedError, match="gap"): + WordstatCollector._assert_contiguous_dynamics_rows(dataset, Granularity.DAILY) + + short_tail = dataset.model_copy(update={"rows": [{"Дата": "22.06.2026"}, {"Дата": "23.06.2026"}]}) + WordstatCollector._assert_contiguous_dynamics_rows(short_tail, Granularity.DAILY) + + +def test_dynamics_series_rejects_a_stale_window_that_predates_the_request(): + # A readiness gate that only confirmed the first row matched date_from + # could pass for a table that still shows an earlier, stale window + # whose *first* row happens to coincide with date_from by chance, while + # its last row never advanced to cover date_to. This exercises the + # containment check directly on the exported rows (the actual defense + # against a stale end boundary), independent of the live DOM race that + # motivated it (collector.py:603). + dataset = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Дата"], + rows=[ + {"Дата": "22.06.2026"}, + {"Дата": "23.06.2026"}, + {"Дата": "24.06.2026"}, + ], + ) + # A shorter export fully inside the requested window is legitimate + # (documented variable trailing tail) and must not be rejected. + WordstatCollector._assert_contiguous_dynamics_rows( + dataset, Granularity.DAILY, date_from=date(2026, 6, 22), date_to=date(2026, 8, 20) + ) + # A row before the requested start is never legitimate: it means the + # table had not advanced past a previous, stale window. + with pytest.raises(InterfaceChangedError, match="requested window"): + WordstatCollector._assert_contiguous_dynamics_rows( + dataset, Granularity.DAILY, date_from=date(2026, 6, 23), date_to=date(2026, 8, 20) + ) + # A row after the requested end is never legitimate either: it means + # the table is still showing a different, later window than requested. + with pytest.raises(InterfaceChangedError, match="requested window"): + WordstatCollector._assert_contiguous_dynamics_rows( + dataset, Granularity.DAILY, date_from=date(2026, 6, 22), date_to=date(2026, 6, 23) + ) + + +def test_single_row_export_is_not_exempt_from_the_containment_check(): + # Cycle-review round 2, C2: the containment check originally sat behind + # the same `len(rows) < 2` early return as the pairwise contiguity loop + # -- but containment only needs dates[0]/dates[-1] (the same row, for a + # 1-row export), not a pair. A genuine 1-row daily/weekly export whose + # single date lies entirely outside the requested window passed + # _is_untrustworthy_empty_export (which only rejects 0 rows) and was + # then written to Parquet and marked complete, unguarded. + stale_single_row = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Дата"], + rows=[{"Дата": "20.06.2026"}], + ) + with pytest.raises(InterfaceChangedError, match="requested window"): + WordstatCollector._assert_contiguous_dynamics_rows( + stale_single_row, Granularity.DAILY, date_from=date(2026, 6, 22), date_to=date(2026, 8, 20) + ) + # A single row genuinely inside the requested window remains legitimate + # (e.g. a one-day window, or the daily series' documented variable + # trailing tail collapsed to one row) and must not be rejected. + in_window_single_row = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Дата"], + rows=[{"Дата": "22.06.2026"}], + ) + WordstatCollector._assert_contiguous_dynamics_rows( + in_window_single_row, Granularity.DAILY, date_from=date(2026, 6, 22), date_to=date(2026, 8, 20) + ) + # No explicit period requested (the UI's default window) still skips + # the check entirely -- nothing to validate a single row against. + WordstatCollector._assert_contiguous_dynamics_rows(stale_single_row, Granularity.DAILY) + + +def test_weekly_containment_uses_the_aligned_week_start_not_the_raw_request(): + # Wordstat snaps a weekly window's first row to the Monday of + # date_from's week (confirmed live: 2018-12-26, a Wednesday, produced + # a first cell starting 24.12.2018 -- the preceding Monday). A + # containment check that compared against the raw date_from instead of + # that aligned Monday would misreport this legitimate alignment as a + # stale window. + dataset = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Неделя с"], + rows=[{"Неделя с": "24.12.2018"}, {"Неделя с": "31.12.2018"}], + ) + WordstatCollector._assert_contiguous_dynamics_rows( + dataset, Granularity.WEEKLY, date_from=date(2018, 12, 26), date_to=date(2019, 1, 13) + ) diff --git a/tests/test_periods.py b/tests/test_periods.py new file mode 100644 index 0000000..8aeddb9 --- /dev/null +++ b/tests/test_periods.py @@ -0,0 +1,71 @@ +from datetime import date + +import pytest + +from wordstat.errors import InvalidPeriodError +from wordstat.periods import Granularity, parse_date, validate_period + + +def test_parse_date_uses_iso_calendar_dates(): + assert parse_date("2026-08-21") == date(2026, 8, 21) + + +@pytest.mark.parametrize("value", ["2026-02-30", "21.08.2026", "2026-13-01"]) +def test_parse_date_rejects_invalid_values(value): + with pytest.raises(InvalidPeriodError): + parse_date(value) + + +def test_period_requires_both_bounds(): + with pytest.raises(InvalidPeriodError, match="provided together"): + validate_period(Granularity.DAILY, date(2026, 8, 1), None) + + +def test_period_rejects_reversed_and_pre_2018_ranges(): + with pytest.raises(InvalidPeriodError, match="earlier"): + validate_period(Granularity.DAILY, date(2026, 8, 20), date(2026, 8, 1)) + with pytest.raises(InvalidPeriodError, match="January 2018"): + validate_period(Granularity.MONTHLY, date(2017, 12, 1), date(2018, 3, 1)) + + +def test_daily_period_is_at_most_60_days_and_not_future(): + with pytest.raises(InvalidPeriodError, match="60"): + validate_period(Granularity.DAILY, date(2026, 6, 1), date(2026, 7, 31), today=date(2026, 8, 21)) + with pytest.raises(InvalidPeriodError, match="after today"): + validate_period(Granularity.DAILY, date(2026, 8, 20), date(2026, 8, 22), today=date(2026, 8, 21)) + validate_period(Granularity.DAILY, date(2026, 7, 1), date(2026, 8, 21), today=date(2026, 8, 21)) + + +def test_daily_period_rejects_a_start_date_further_back_than_the_trailing_60_day_window(): + # Live CDP measurement (issue #6 phase 2): the daily date-range + # picker's own year-select popup only ever offers the current year — + # a request for a valid-length window that starts far in the past + # (e.g. 2019) is accepted by validate_period today, reaches the live + # UI, and fails deep inside calendar-click code with an + # InterfaceChangedError instead of being rejected up front. issue #6 + # step 3 explicitly asks for "daily granularity outside the 60-day + # window" to be rejected before opening Chrome, which this closes. + # + # The bound is expressed relative to `today`, not the picker's year + # list (a year-based rule would wrongly reject legal windows early in + # January) and shares the existing 60-calendar-day convention: PR #20's + # live-confirmed window 22.06.2026-20.08.2026, measured on 21.08.2026, + # is exactly 60 days and must remain accepted. + validate_period(Granularity.DAILY, date(2026, 6, 22), date(2026, 8, 20), today=date(2026, 8, 21)) + with pytest.raises(InvalidPeriodError, match="60 days in the past"): + validate_period(Granularity.DAILY, date(2019, 1, 1), date(2019, 2, 15), today=date(2026, 8, 21)) + with pytest.raises(InvalidPeriodError, match="60 days in the past"): + validate_period(Granularity.DAILY, date(2026, 6, 21), date(2026, 6, 25), today=date(2026, 8, 21)) + + +def test_weekly_and_monthly_minimums(): + with pytest.raises(InvalidPeriodError, match="three calendar weeks"): + validate_period(Granularity.WEEKLY, date(2026, 8, 1), date(2026, 8, 20)) + validate_period(Granularity.WEEKLY, date(2026, 8, 1), date(2026, 8, 21)) + with pytest.raises(InvalidPeriodError, match="three calendar months"): + validate_period(Granularity.MONTHLY, date(2026, 8, 1), date(2026, 10, 1)) + validate_period(Granularity.MONTHLY, date(2026, 8, 1), date(2026, 10, 31)) + + +def test_five_year_limit_is_not_assumed(): + validate_period(Granularity.MONTHLY, date(2018, 1, 1), date(2026, 8, 21))