From efaa9baac1112d6de8d09925e6f5178cccc61703 Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 13:15:25 +0700 Subject: [PATCH 01/15] feat: support dynamics granularity and requested periods --- src/wordstat/cli.py | 30 +++++- src/wordstat/collector.py | 197 +++++++++++++++++++++++++++++++--- src/wordstat/dataset_io.py | 8 +- src/wordstat/dtypes.py | 7 +- src/wordstat/errors.py | 4 + src/wordstat/models.py | 5 + src/wordstat/periods.py | 69 ++++++++++++ tests/test_dataset_io.py | 22 +++- tests/test_dtypes.py | 6 +- tests/test_manifest_period.py | 27 +++++ tests/test_periods.py | 49 +++++++++ 11 files changed, 400 insertions(+), 24 deletions(-) create mode 100644 src/wordstat/periods.py create mode 100644 tests/test_manifest_period.py create mode 100644 tests/test_periods.py diff --git a/src/wordstat/cli.py b/src/wordstat/cli.py index 6741b59..1ac64f9 100644 --- a/src/wordstat/cli.py +++ b/src/wordstat/cli.py @@ -8,11 +8,12 @@ from wordstat.collector import WordstatCollector from wordstat.config import load_config from wordstat.errors import WordstatError +from wordstat.periods import Granularity, parse_date, validate_period from wordstat.storage import prepare_resume_directory _CONFIG = load_config() _DEFAULT_CDP_URL = _CONFIG.get("cdp_url", "http://127.0.0.1:9222") -# Wordstat's native export encoding is cp1251; a phrases file typed or saved +# Wordstat exports CSV as UTF-8 with BOM; a phrases file typed or saved # on the same machine can plausibly be in either. Mirrors the encoding probe # in csv_io.py, minus utf-8-sig (a phrases file is authored by hand, not # exported by Wordstat, so a BOM is unlikely but harmless either way). @@ -69,6 +70,14 @@ def _read_phrases_file(path: Path) -> str: ) @click.option("--cdp-url", envvar="WORDSTAT_CDP_URL", default=_DEFAULT_CDP_URL, show_default=True) @click.option("--timeout", "timeout_seconds", type=click.FloatRange(min=1), default=45.0, show_default=True) +@click.option( + "--granularity", + type=click.Choice(Granularity, case_sensitive=False), + default=Granularity.MONTHLY, + show_default=True, +) +@click.option("--date-from", type=str, default=None, help="Dynamics window start (YYYY-MM-DD).") +@click.option("--date-to", type=str, default=None, help="Dynamics window end (YYYY-MM-DD).") @click.option( "--keep-raw", is_flag=True, @@ -95,6 +104,9 @@ def collect( output_dir: Path, cdp_url: str, timeout_seconds: float, + granularity: str, + date_from: str | None, + date_to: str | None, keep_raw: bool, resume_dir: Path | None, ) -> None: @@ -106,6 +118,13 @@ def collect( """ phrases = resolve_phrases(phrase, phrases_file) + selected_granularity = Granularity(granularity.lower()) + try: + parsed_from = parse_date(date_from) + parsed_to = parse_date(date_to) + validate_period(selected_granularity, parsed_from, parsed_to) + except WordstatError as error: + raise click.ClickException(str(error)) from error if not phrases: raise click.ClickException("At least one search phrase is required") if resume_dir is not None: @@ -130,7 +149,14 @@ def collect( keep_raw=keep_raw, ) try: - batch = asyncio.run(collector.collect_many(phrases, region=region, resume_directory=resume_dir)) + collect_kwargs = {"region": region, "resume_directory": resume_dir} + if selected_granularity is not Granularity.MONTHLY or parsed_from is not None: + collect_kwargs.update( + granularity=selected_granularity, + date_from=parsed_from, + date_to=parsed_to, + ) + batch = asyncio.run(collector.collect_many(phrases, **collect_kwargs)) # Only domain errors become friendly messages; an unexpected ValueError # from a dependency should keep its traceback instead of being reworded. except WordstatError as error: diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index 3ffc7ad..fa60720 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -5,7 +5,7 @@ import tempfile import time from collections.abc import Callable -from datetime import UTC, datetime +from datetime import UTC, date, datetime from pathlib import Path from browser_use.browser import BrowserSession @@ -28,6 +28,7 @@ PhraseFailure, WordstatView, ) +from wordstat.periods import Granularity, validate_period from wordstat.storage import ( create_run_directory, finalize_raw, @@ -42,6 +43,8 @@ SEARCH_SELECTOR = ".wordstat__search-button" DOWNLOAD_SELECTOR = "button.save-button" DOWNLOAD_CSV_MENU_ITEM_SELECTOR = "a[download]:has(button.save-csv-button)" +GRANULARITY_SELECTOR = ".wordstat__content-type_select > button" +DATE_RANGE_SELECTOR = ".range-datepicker__selected-dates > button" REGION_BUTTON_SELECTOR = ".settings__selected button" TABLE_ROW_SELECTOR = ".table__wrapper tbody tr" # Tab selectors live here with the rest of the DOM knowledge; the markup is @@ -148,6 +151,9 @@ async def collect_many( phrases: list[str], region: str = "Россия", resume_directory: Path | None = None, + granularity: Granularity = Granularity.MONTHLY, + date_from: date | None = None, + date_to: date | None = None, ) -> BatchCollectionResult: """Collect reports for several phrases inside a single browser session. @@ -165,6 +171,7 @@ async def collect_many( """ region = region.strip() + validate_period(granularity, date_from, date_to) if not phrases: raise InvalidRequestError("At least one search phrase is required") if not region: @@ -233,15 +240,22 @@ def _mark_region_ready() -> None: # _collect_one for why the region control can't be # re-selected once a phrase's view loop has run (and # doesn't need to be after that). + collect_kwargs = { + "set_region": not region_ready, + "on_region_applied": _mark_region_ready, + "resume_directory": resume_directory, + } + # Keep the historical call shape for the default + # monthly path; several integrations monkeypatch this + # seam and default behavior must remain unchanged. + if granularity is not Granularity.MONTHLY or date_from is not None: + collect_kwargs.update( + granularity=granularity, + date_from=date_from, + date_to=date_to, + ) result = await self._collect_one( - page, - session, - downloads_path, - phrase, - region, - set_region=not region_ready, - on_region_applied=_mark_region_ready, - resume_directory=resume_directory, + page, session, downloads_path, phrase, region, **collect_kwargs ) results.append(result) except AuthenticationRequiredError as error: @@ -279,6 +293,9 @@ async def _collect_one( set_region: bool = True, on_region_applied: Callable[[], None] | None = None, resume_directory: Path | None = None, + granularity: Granularity = Granularity.MONTHLY, + date_from: date | None = None, + date_to: date | None = None, ) -> CollectionResult: # Checked once before the batch starts (collect_many), but a session # can lose authentication mid-batch (e.g. Yandex logs it out); check @@ -295,6 +312,11 @@ async def _collect_one( # separate, explicit path, not a fallback baked into it. run_directory = resume_directory manifest = prepare_resume_directory(run_directory, phrase, region) + requested_period = self._requested_period(date_from, date_to) + if manifest.granularity is not granularity or manifest.requested_period != requested_period: + raise InvalidRequestError( + "--resume-dir was created with a different dynamics granularity or requested period" + ) manifest_path = run_directory / "manifest.json" pending_views = views_to_collect(run_directory, manifest) if not pending_views: @@ -359,6 +381,8 @@ async def _collect_one( updated_at=start, source_url=await page.get_url(), exports=[], + granularity=granularity, + requested_period=self._requested_period(date_from, date_to), ) write_manifest(manifest_path, manifest) else: @@ -378,17 +402,39 @@ async def _collect_one( for view in pending_views: selector = VIEW_SELECTORS[view] await self._select_view(page, selector, view) + # The live UI can expose the first table row before the export + # blob has been rebuilt for the selected phrase/view. A short + # settling interval prevents a header-only CSV from racing the + # table repaint; the structural row gate above remains required. + if view is not WordstatView.REGIONS: + await asyncio.sleep(1.0) + if view is WordstatView.DYNAMICS and ( + granularity is not Granularity.MONTHLY or date_from is not None + ): + await self._set_granularity(page, granularity) + if date_from is not None: + await self._set_period(page, granularity, date_from, date_to) source = await self._download_current_view(page, session, downloads_path) try: # Convert before disposing of the download, so a parse or - # write failure leaves the raw CSV on disk to inspect. + # write failure leaves the raw CSV on disk to inspect. The + # live export blob can lag the table repaint once; retry an + # empty table export once before failing closed. dataset = parse_wordstat_csv(source, view) + if _is_untrustworthy_empty_export(view, dataset): + await asyncio.sleep(2.0) + source = await self._download_current_view(page, session, downloads_path) + dataset = parse_wordstat_csv(source, view) if _is_untrustworthy_empty_export(view, dataset): raise InterfaceChangedError( - f"Wordstat returned an empty {view.value} CSV, but the page had rendered at " - "least one table row before the download was triggered; export is not trustworthy" + f"Wordstat returned an empty {view.value} CSV after a retry, but the page had rendered " + "at least one table row before the download was triggered; export is not trustworthy" ) - data_path, dtypes = write_dataset(dataset, run_directory) + file_name = self._dynamics_file_name(granularity) if view is WordstatView.DYNAMICS else None + if file_name is None: + data_path, dtypes = write_dataset(dataset, run_directory) + else: + data_path, dtypes = write_dataset(dataset, run_directory, file_name=file_name) raw_path = finalize_raw(source, run_directory, view, self.keep_raw) except Exception: # noqa: BLE001 # source lives in the batch's shared, temporary downloads @@ -416,6 +462,10 @@ async def _collect_one( # CollectionManifest.status), not a stale one still claiming # every view is missing. manifest = merge_export(manifest, export) + if view is WordstatView.DYNAMICS: + manifest = manifest.model_copy( + update={"actual_period": self._actual_period(dataset)} + ) write_manifest(manifest_path, manifest) return CollectionResult( @@ -424,6 +474,127 @@ async def _collect_one( manifest=manifest, ) + @staticmethod + def _requested_period(date_from: date | None, date_to: date | None) -> dict[str, str] | None: + if date_from is None or date_to is None: + return None + return {"from": date_from.isoformat(), "to": date_to.isoformat()} + + @staticmethod + def _actual_period(dataset: CsvDataset) -> dict[str, str] | None: + if not dataset.rows or not dataset.headers: + return None + field = dataset.headers[0] + return {"field": field, "from": dataset.rows[0][field], "to": dataset.rows[-1][field]} + + + @staticmethod + def _dynamics_file_name(granularity: Granularity) -> str: + # Preserve the old monthly filename for existing consumers; non-monthly + # exports must carry their granularity in the filename. + return "dynamics.parquet" if granularity is Granularity.MONTHLY else f"dynamics_{granularity.value}.parquet" + + async def _set_granularity(self, page, granularity: Granularity) -> None: + labels = { + Granularity.DAILY: "По дням", + Granularity.WEEKLY: "По неделям", + Granularity.MONTHLY: "По месяцам", + } + await self._click(page, GRANULARITY_SELECTOR) + await self._click_visible_text(page, ".Popup2_visible [role='option']", labels[granularity]) + await self._wait_for( + page, + f"() => document.querySelector({json.dumps(GRANULARITY_SELECTOR)})?.textContent?.trim() === " + f"{json.dumps(labels[granularity])}", + ) + + async def _set_period(self, page, granularity: Granularity, date_from: date, date_to: date | None) -> None: + if date_to is None: + raise InvalidRequestError("A period end date is required when a start date is provided") + popup_type = "month" if granularity is Granularity.MONTHLY else "day" + await self._click(page, DATE_RANGE_SELECTOR) + await self._select_calendar_date(page, popup_type, date_from) + await self._click(page, DATE_RANGE_SELECTOR) + await self._select_calendar_date(page, popup_type, date_to) + + async def _select_calendar_date(self, page, popup_type: str, target: date) -> None: + root = f".range-datepicker_type_{popup_type}" + await self._wait_for( + page, + f"() => [...document.querySelectorAll({json.dumps(root)})].some((x) => " + "x.offsetParent !== null && getComputedStyle(x).visibility !== 'hidden')", + ) + if popup_type == "month": + year_selector = f"{root} .datepicker-month__years_select > button" + month_selector = f"{root} .react-datepicker__month-text" + # The live interface is Russian; map explicitly rather than + # depending on the host locale. + month_label = ( + "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split()[ + target.month - 1 + ] + ) + else: + year_selector = f"{root} .range-datepicker__years_select > button" + month_selector = f"{root} .range-datepicker__months_select > button" + month_label = ( + "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split()[ + target.month - 1 + ] + ) + await self._click(page, year_selector) + await self._click_visible_text(page, ".Popup2_visible [role='option']", str(target.year)) + if popup_type == "day": + await self._click(page, month_selector) + await self._click_visible_text(page, ".Popup2_visible [role='option']", month_label) + day_selector = f"{root} button[name='day']" + if popup_type == "month": + await self._click_visible_text(page, month_selector, month_label.capitalize()) + else: + await self._click_visible_date_button(page, day_selector, str(target.day)) + + async def _click_visible_date_button(self, page, selector: str, text: str) -> None: + result = await page.evaluate( + """(...args) => { + const [selector, text] = args; + const matches = [...document.querySelectorAll(selector)].filter((element) => { + const style = getComputedStyle(element); + return element.textContent?.trim() === text + && !element.className.toString().includes('outside') + && !element.className.toString().includes('disabled') + && element.offsetParent !== null + && style.visibility !== 'hidden'; + }); + if (matches.length !== 1) return JSON.stringify({count: matches.length}); + matches[0].click(); + return JSON.stringify({count: 1}); + }""", + selector, + text, + ) + if json.loads(result) != {"count": 1}: + raise InterfaceChangedError(f"Wordstat date control {selector!r} was not uniquely found") + + async def _click_visible_text(self, page, selector: str, text: str) -> None: + result = await page.evaluate( + """(...args) => { + const [selector, text] = args; + const matches = [...document.querySelectorAll(selector)].filter((element) => { + const style = getComputedStyle(element); + return element.textContent?.trim() === text + && element.offsetParent !== null + && style.visibility !== 'hidden'; + }); + if (matches.length !== 1) return JSON.stringify({count: matches.length}); + matches[0].click(); + return JSON.stringify({count: 1}); + }""", + selector, + text, + ) + if json.loads(result) != {"count": 1}: + raise InterfaceChangedError(f"Wordstat option {text!r} was not uniquely found") + async def _assert_authenticated(self, page) -> None: state = await page.evaluate( """() => JSON.stringify({ diff --git a/src/wordstat/dataset_io.py b/src/wordstat/dataset_io.py index 1057375..60805b6 100644 --- a/src/wordstat/dataset_io.py +++ b/src/wordstat/dataset_io.py @@ -11,8 +11,10 @@ from wordstat.models import CsvDataset -def write_dataset(dataset: CsvDataset, run_directory: Path) -> tuple[Path, dict[str, str]]: - """Write one report as ``.parquet`` and report the inferred column types. +def write_dataset( + dataset: CsvDataset, run_directory: Path, file_name: str | None = None +) -> tuple[Path, dict[str, str]]: + """Write one report and report the inferred column types. Column order follows the export's own header order. The returned dtype mapping goes into the manifest so an unexpected export format is visible @@ -33,6 +35,6 @@ def write_dataset(dataset: CsvDataset, run_directory: Path) -> tuple[Path, dict[ columns[header] = pa.array(coerced, type=pa.type_for_alias(dtype)) table = pa.table(columns) - destination = run_directory / f"{dataset.view.value}.parquet" + destination = run_directory / (file_name or f"{dataset.view.value}.parquet") pq.write_table(table, destination, compression="zstd") return destination, dtypes diff --git a/src/wordstat/dtypes.py b/src/wordstat/dtypes.py index 5c10383..1551429 100644 --- a/src/wordstat/dtypes.py +++ b/src/wordstat/dtypes.py @@ -1,7 +1,8 @@ """Column type inference for Wordstat's localized string values. Wordstat exports every value as text: counts arrive as ``"5 228 679"`` with -non-breaking thousands separators, and dynamics periods as ``"01.2024"``. +non-breaking thousands separators, and dynamics periods as ``"август 2024"`` +or dates such as ``"22.06.2026"``. Writing those straight to Parquet would only change the container, so columns are typed here first. @@ -24,8 +25,8 @@ _WHITESPACE = re.compile(r"\s") # Deliberately strict. ``int()``/``float()`` would accept values that are not -# numbers in this data: "01.2024" (a dynamics period) parses as 1.2024, -# silently destroying it, and "1e5", "1_000", "inf" and "nan" are accepted too. +# numbers in this data and could silently destroy a dotted date. Real monthly +# periods are "август 2024"; daily/weekly exports use "22.06.2026". # Only a comma is a decimal separator here — Wordstat exports with a Russian # locale — which is precisely what rejects the dotted period format. _NUMERIC = re.compile(r"^-?[0-9]+(?:,[0-9]+)?$") diff --git a/src/wordstat/errors.py b/src/wordstat/errors.py index 0a485c8..7681db0 100644 --- a/src/wordstat/errors.py +++ b/src/wordstat/errors.py @@ -9,6 +9,10 @@ class InvalidRequestError(WordstatError): """The requested phrase or region is unusable before the browser is touched.""" +class InvalidPeriodError(InvalidRequestError): + """The requested dynamics granularity/window is not supported.""" + + class AuthenticationRequiredError(WordstatError): """Wordstat is reachable but the attached browser is not authenticated.""" diff --git a/src/wordstat/models.py b/src/wordstat/models.py index 92d0102..e73f6a7 100644 --- a/src/wordstat/models.py +++ b/src/wordstat/models.py @@ -6,6 +6,8 @@ from pydantic import BaseModel, ConfigDict, Field, computed_field, model_validator +from wordstat.periods import Granularity + class WordstatView(StrEnum): """The four Wordstat reports exported by the MVP.""" @@ -103,6 +105,9 @@ class CollectionManifest(BaseModel): updated_at: datetime | None = None source_url: str exports: list[ExportSummary] + granularity: Granularity = Granularity.MONTHLY + requested_period: dict[str, str] | None = None + actual_period: dict[str, str] | None = None @computed_field # type: ignore[prop-decorator] @property diff --git a/src/wordstat/periods.py b/src/wordstat/periods.py new file mode 100644 index 0000000..89f0dbb --- /dev/null +++ b/src/wordstat/periods.py @@ -0,0 +1,69 @@ +"""Validation and normalization of user-requested dynamics periods.""" + +from __future__ import annotations + +from calendar import monthrange +from datetime import date, timedelta +from enum import StrEnum + +from wordstat.errors import InvalidPeriodError + + +class Granularity(StrEnum): + MONTHLY = "monthly" + WEEKLY = "weekly" + DAILY = "daily" + + +EARLIEST_DATE = date(2018, 1, 1) + + +def validate_period( + granularity: Granularity, + date_from: date | None, + date_to: date | None, + *, + today: date | None = None, +) -> None: + """Validate an explicit window before opening Chrome. + + Omitted dates leave the UI's default window untouched. The five-year + maximum is deliberately not enforced because phase 1 did not establish + it as a reliable live fact. + """ + if (date_from is None) != (date_to is None): + raise InvalidPeriodError("--date-from and --date-to must be provided together") + if date_from is None or date_to is None: + return + if date_from < EARLIEST_DATE or date_to < EARLIEST_DATE: + raise InvalidPeriodError("The requested period cannot include dates before January 2018") + if date_to < date_from: + raise InvalidPeriodError("The period end date must not be earlier than its start date") + if granularity is Granularity.DAILY: + current = today or date.today() + if date_to > current: + raise InvalidPeriodError("Daily statistics cannot be requested after today") + if date_to - date_from >= timedelta(days=60): + raise InvalidPeriodError("Daily statistics support at most 60 calendar days") + elif granularity is Granularity.WEEKLY: + if date_to - date_from < timedelta(days=20): + raise InvalidPeriodError("Weekly statistics require at least three calendar weeks") + else: + end_month = date_from.month + 2 + end_year = date_from.year + (end_month - 1) // 12 + end_month = (end_month - 1) % 12 + 1 + minimum_end = date(end_year, end_month, monthrange(end_year, end_month)[1]) + if date_to < minimum_end: + raise InvalidPeriodError("Monthly statistics require at least three calendar months") + + +def parse_date(value: str | None) -> date | None: + if value is None: + return None + try: + year, month, day = (int(part) for part in value.split("-")) + if month < 1 or month > 12 or day < 1 or day > monthrange(year, month)[1]: + raise ValueError + return date(year, month, day) + except (TypeError, ValueError): + raise InvalidPeriodError(f"Invalid date {value!r}; expected YYYY-MM-DD") from None diff --git a/tests/test_dataset_io.py b/tests/test_dataset_io.py index 5d39c78..68f25e5 100644 --- a/tests/test_dataset_io.py +++ b/tests/test_dataset_io.py @@ -51,7 +51,7 @@ def test_write_dataset_writes_empty_cells_as_null(tmp_path): def test_write_dataset_keeps_dynamics_periods_as_text(tmp_path): dataset = _dataset( ["Период", "Показов"], - [{"Период": "01.2024", "Показов": "10"}, {"Период": "02.2024", "Показов": "20"}], + [{"Период": "август 2024", "Показов": "10"}, {"Период": "сентябрь 2024", "Показов": "20"}], view=WordstatView.DYNAMICS, ) @@ -60,6 +60,26 @@ def test_write_dataset_keeps_dynamics_periods_as_text(tmp_path): assert dtypes["Период"] == "string" +def test_write_dataset_keeps_daily_dates_as_text(tmp_path): + dataset = _dataset( + ["Дата", "Показов"], + [{"Дата": "22.06.2026", "Показов": "10"}, {"Дата": "23.06.2026", "Показов": "20"}], + view=WordstatView.DYNAMICS, + ) + + _, dtypes = write_dataset(dataset, tmp_path) + + assert dtypes["Дата"] == "string" + + +def test_write_dataset_accepts_granularity_specific_filename(tmp_path): + dataset = _dataset(["Дата", "Показов"], [{"Дата": "22.06.2026", "Показов": "10"}], view=WordstatView.DYNAMICS) + + path, _ = write_dataset(dataset, tmp_path, file_name="dynamics_daily.parquet") + + assert path.name == "dynamics_daily.parquet" + + def test_write_dataset_handles_a_report_without_rows(tmp_path): dataset = _dataset(["Запрос", "Показов"], []) diff --git a/tests/test_dtypes.py b/tests/test_dtypes.py index 23eb0d8..01af7cb 100644 --- a/tests/test_dtypes.py +++ b/tests/test_dtypes.py @@ -28,7 +28,8 @@ def test_parse_number_accepts_wordstat_number_formats(value, expected): @pytest.mark.parametrize( "value", [ - "01.2024", # dynamics period, must never become 1.2024 + "август 2024", # monthly dynamics period must remain text + "22.06.2026", # daily/weekly dynamics date must remain text "1e5", "1_000", "inf", @@ -45,7 +46,8 @@ def test_parse_number_rejects_non_numbers(value): def test_dynamics_period_column_stays_string(): - assert infer_column(["01.2024", "02.2024", "03.2024"]) is None + assert infer_column(["август 2024", "сентябрь 2024", "октябрь 2024"]) is None + assert infer_column(["22.06.2026", "23.06.2026"]) is None def test_all_numeric_column_becomes_int64(): diff --git a/tests/test_manifest_period.py b/tests/test_manifest_period.py new file mode 100644 index 0000000..cf37ee9 --- /dev/null +++ b/tests/test_manifest_period.py @@ -0,0 +1,27 @@ +from wordstat.collector import WordstatCollector +from wordstat.models import CsvDataset, WordstatView + + +def test_actual_period_uses_the_values_present_in_the_export(): + dataset = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Дата", "Число запросов"], + rows=[ + {"Дата": "22.06.2026", "Число запросов": "1"}, + {"Дата": "18.08.2026", "Число запросов": "2"}, + ], + ) + + assert WordstatCollector._actual_period(dataset) == { + "field": "Дата", + "from": "22.06.2026", + "to": "18.08.2026", + } + + +def test_dynamics_filename_identifies_non_monthly_granularity(): + from wordstat.periods import Granularity + + assert WordstatCollector._dynamics_file_name(Granularity.MONTHLY) == "dynamics.parquet" + assert WordstatCollector._dynamics_file_name(Granularity.DAILY) == "dynamics_daily.parquet" + assert WordstatCollector._dynamics_file_name(Granularity.WEEKLY) == "dynamics_weekly.parquet" diff --git a/tests/test_periods.py b/tests/test_periods.py new file mode 100644 index 0000000..af612e3 --- /dev/null +++ b/tests/test_periods.py @@ -0,0 +1,49 @@ +from datetime import date + +import pytest + +from wordstat.errors import InvalidPeriodError +from wordstat.periods import Granularity, parse_date, validate_period + + +def test_parse_date_uses_iso_calendar_dates(): + assert parse_date("2026-08-21") == date(2026, 8, 21) + + +@pytest.mark.parametrize("value", ["2026-02-30", "21.08.2026", "2026-13-01"]) +def test_parse_date_rejects_invalid_values(value): + with pytest.raises(InvalidPeriodError): + parse_date(value) + + +def test_period_requires_both_bounds(): + with pytest.raises(InvalidPeriodError, match="provided together"): + validate_period(Granularity.DAILY, date(2026, 8, 1), None) + + +def test_period_rejects_reversed_and_pre_2018_ranges(): + with pytest.raises(InvalidPeriodError, match="earlier"): + validate_period(Granularity.DAILY, date(2026, 8, 20), date(2026, 8, 1)) + with pytest.raises(InvalidPeriodError, match="January 2018"): + validate_period(Granularity.MONTHLY, date(2017, 12, 1), date(2018, 3, 1)) + + +def test_daily_period_is_at_most_60_days_and_not_future(): + with pytest.raises(InvalidPeriodError, match="60"): + validate_period(Granularity.DAILY, date(2026, 6, 1), date(2026, 7, 31), today=date(2026, 8, 21)) + with pytest.raises(InvalidPeriodError, match="after today"): + validate_period(Granularity.DAILY, date(2026, 8, 20), date(2026, 8, 22), today=date(2026, 8, 21)) + validate_period(Granularity.DAILY, date(2026, 7, 1), date(2026, 8, 21), today=date(2026, 8, 21)) + + +def test_weekly_and_monthly_minimums(): + with pytest.raises(InvalidPeriodError, match="three calendar weeks"): + validate_period(Granularity.WEEKLY, date(2026, 8, 1), date(2026, 8, 20)) + validate_period(Granularity.WEEKLY, date(2026, 8, 1), date(2026, 8, 21)) + with pytest.raises(InvalidPeriodError, match="three calendar months"): + validate_period(Granularity.MONTHLY, date(2026, 8, 1), date(2026, 10, 1)) + validate_period(Granularity.MONTHLY, date(2026, 8, 1), date(2026, 10, 31)) + + +def test_five_year_limit_is_not_assumed(): + validate_period(Granularity.MONTHLY, date(2018, 1, 1), date(2026, 8, 21)) From f83604e020982de44eb90e72de0ae409a8bc8f5b Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 13:17:21 +0700 Subject: [PATCH 02/15] fix: validate contiguous dynamics rows without assuming full window --- src/wordstat/collector.py | 23 ++++++++++++++++++++++- tests/test_manifest_period.py | 20 ++++++++++++++++++-- 2 files changed, 40 insertions(+), 3 deletions(-) diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index fa60720..6105869 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -5,7 +5,7 @@ import tempfile import time from collections.abc import Callable -from datetime import UTC, date, datetime +from datetime import UTC, date, datetime, timedelta from pathlib import Path from browser_use.browser import BrowserSession @@ -430,6 +430,8 @@ async def _collect_one( f"Wordstat returned an empty {view.value} CSV after a retry, but the page had rendered " "at least one table row before the download was triggered; export is not trustworthy" ) + if view is WordstatView.DYNAMICS: + self._assert_contiguous_dynamics_rows(dataset, granularity) file_name = self._dynamics_file_name(granularity) if view is WordstatView.DYNAMICS else None if file_name is None: data_path, dtypes = write_dataset(dataset, run_directory) @@ -487,6 +489,25 @@ def _actual_period(dataset: CsvDataset) -> dict[str, str] | None: field = dataset.headers[0] return {"field": field, "from": dataset.rows[0][field], "to": dataset.rows[-1][field]} + @staticmethod + def _assert_contiguous_dynamics_rows(dataset: CsvDataset, granularity: Granularity) -> None: + if granularity is Granularity.MONTHLY or len(dataset.rows) < 2: + return + field = dataset.headers[0] + try: + dates = [datetime.strptime(row[field], "%d.%m.%Y").date() for row in dataset.rows] + except ValueError as error: + raise InterfaceChangedError( + f"Wordstat returned an unexpected {granularity.value} dynamics date format" + ) from error + step = timedelta(days=1 if granularity is Granularity.DAILY else 7) + for previous, current in zip(dates, dates[1:]): + if current - previous != step: + raise InterfaceChangedError( + f"Wordstat returned a gap in the {granularity.value} dynamics series " + f"between {previous:%d.%m.%Y} and {current:%d.%m.%Y}" + ) + @staticmethod def _dynamics_file_name(granularity: Granularity) -> str: diff --git a/tests/test_manifest_period.py b/tests/test_manifest_period.py index cf37ee9..bba4b98 100644 --- a/tests/test_manifest_period.py +++ b/tests/test_manifest_period.py @@ -1,5 +1,9 @@ +import pytest + from wordstat.collector import WordstatCollector +from wordstat.errors import InterfaceChangedError from wordstat.models import CsvDataset, WordstatView +from wordstat.periods import Granularity def test_actual_period_uses_the_values_present_in_the_export(): @@ -20,8 +24,20 @@ def test_actual_period_uses_the_values_present_in_the_export(): def test_dynamics_filename_identifies_non_monthly_granularity(): - from wordstat.periods import Granularity - assert WordstatCollector._dynamics_file_name(Granularity.MONTHLY) == "dynamics.parquet" assert WordstatCollector._dynamics_file_name(Granularity.DAILY) == "dynamics_daily.parquet" assert WordstatCollector._dynamics_file_name(Granularity.WEEKLY) == "dynamics_weekly.parquet" + + +def test_dynamics_series_allows_short_tail_but_rejects_internal_gap(): + dataset = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Дата"], + rows=[{"Дата": "22.06.2026"}, {"Дата": "23.06.2026"}, {"Дата": "25.06.2026"}], + ) + + with pytest.raises(InterfaceChangedError, match="gap"): + WordstatCollector._assert_contiguous_dynamics_rows(dataset, Granularity.DAILY) + + short_tail = dataset.model_copy(update={"rows": [{"Дата": "22.06.2026"}, {"Дата": "23.06.2026"}]}) + WordstatCollector._assert_contiguous_dynamics_rows(short_tail, Granularity.DAILY) From 0027175e1ba6d74a97c7f0da5a6ebe183dbf9ea6 Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 13:21:40 +0700 Subject: [PATCH 03/15] fix: wait for selected dynamics table before export --- src/wordstat/collector.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index 6105869..b664619 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -528,6 +528,7 @@ async def _set_granularity(self, page, granularity: Granularity) -> None: f"() => document.querySelector({json.dumps(GRANULARITY_SELECTOR)})?.textContent?.trim() === " f"{json.dumps(labels[granularity])}", ) + await self._wait_for_table_granularity(page, granularity) async def _set_period(self, page, granularity: Granularity, date_from: date, date_to: date | None) -> None: if date_to is None: @@ -537,6 +538,19 @@ async def _set_period(self, page, granularity: Granularity, date_from: date, dat await self._select_calendar_date(page, popup_type, date_from) await self._click(page, DATE_RANGE_SELECTOR) await self._select_calendar_date(page, popup_type, date_to) + await self._wait_for_table_granularity(page, granularity) + + async def _wait_for_table_granularity(self, page, granularity: Granularity) -> None: + patterns = { + Granularity.DAILY: r"^\d{1,2}\s+[А-Яа-яЁё]+$", + Granularity.WEEKLY: r"^\d{1,2}\s+[А-Яа-яЁё]+\s+\d{4}\s+–\s+\d{1,2}\s+[А-Яа-яЁё]+\s+\d{4}$", + Granularity.MONTHLY: r"^[А-Яа-яЁё]+\s+\d{4}$", + } + expression = ( + "() => new RegExp(" + json.dumps(patterns[granularity]) + ").test(" + f"document.querySelector({json.dumps(TABLE_ROW_SELECTOR)})?.textContent?.trim() ?? '')" + ) + await self._wait_for(page, expression) async def _select_calendar_date(self, page, popup_type: str, target: date) -> None: root = f".range-datepicker_type_{popup_type}" From 81fe7ce5d716108acd19abc52dc397426aa549ae Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 13:25:01 +0700 Subject: [PATCH 04/15] fix: inspect first dynamics cell for selected granularity --- src/wordstat/collector.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index b664619..ba6207b 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -548,7 +548,7 @@ async def _wait_for_table_granularity(self, page, granularity: Granularity) -> N } expression = ( "() => new RegExp(" + json.dumps(patterns[granularity]) + ").test(" - f"document.querySelector({json.dumps(TABLE_ROW_SELECTOR)})?.textContent?.trim() ?? '')" + f"document.querySelector({json.dumps(TABLE_ROW_SELECTOR)})?.querySelector('td')?.textContent?.trim() ?? '')" ) await self._wait_for(page, expression) From b157f26ebefb6e487ea7a1be5ff75c633f6657dd Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 13:28:26 +0700 Subject: [PATCH 05/15] fix: use weekly date picker for weekly dynamics --- src/wordstat/collector.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index ba6207b..0ba9c7b 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -533,7 +533,9 @@ async def _set_granularity(self, page, granularity: Granularity) -> None: async def _set_period(self, page, granularity: Granularity, date_from: date, date_to: date | None) -> None: if date_to is None: raise InvalidRequestError("A period end date is required when a start date is provided") - popup_type = "month" if granularity is Granularity.MONTHLY else "day" + popup_type = "month" if granularity is Granularity.MONTHLY else ( + "week" if granularity is Granularity.WEEKLY else "day" + ) await self._click(page, DATE_RANGE_SELECTOR) await self._select_calendar_date(page, popup_type, date_from) await self._click(page, DATE_RANGE_SELECTOR) @@ -579,7 +581,7 @@ async def _select_calendar_date(self, page, popup_type: str, target: date) -> No ) await self._click(page, year_selector) await self._click_visible_text(page, ".Popup2_visible [role='option']", str(target.year)) - if popup_type == "day": + if popup_type in {"day", "week"}: await self._click(page, month_selector) await self._click_visible_text(page, ".Popup2_visible [role='option']", month_label) day_selector = f"{root} button[name='day']" From 498ecff72aa600bf5ae4875d299fccf161fcc780 Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 13:44:27 +0700 Subject: [PATCH 06/15] fix: reject explicit weekly periods (Wordstat ignores the requested window) Live CDP measurements confirm Wordstat's weekly dynamics view silently ignores an explicit --date-from/--date-to and returns its own default ~2-year window instead (107 full weeks, unrelated to what was requested). Since a manifest recording a fabricated 'actual_period' that does not match the caller's intent would be the worst possible outcome, an explicit weekly period is now rejected up front by validate_period() with InvalidPeriodError, rather than collected and silently wrong. Weekly dynamics without an explicit period (Wordstat's default window) remains supported and was confirmed live to behave predictably. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01JCyJr8vFFDW1eSMyCQyNNk --- src/wordstat/periods.py | 15 +++++++++++++-- tests/test_periods.py | 14 +++++++++++--- 2 files changed, 24 insertions(+), 5 deletions(-) diff --git a/src/wordstat/periods.py b/src/wordstat/periods.py index 89f0dbb..a5724bf 100644 --- a/src/wordstat/periods.py +++ b/src/wordstat/periods.py @@ -46,8 +46,19 @@ def validate_period( if date_to - date_from >= timedelta(days=60): raise InvalidPeriodError("Daily statistics support at most 60 calendar days") elif granularity is Granularity.WEEKLY: - if date_to - date_from < timedelta(days=20): - raise InvalidPeriodError("Weekly statistics require at least three calendar weeks") + # Live CDP checks (issue #6 phase 2) confirmed Wordstat's weekly view + # silently ignores an explicit --date-from/--date-to and returns its + # default ~2-year window instead (107 full weeks, unrelated to the + # requested dates). Returning that mismatched window with a manifest + # claiming the requested period would be the worst outcome, so an + # explicit weekly period is rejected outright rather than collected + # and silently wrong. Weekly without an explicit period still works + # (see docs/issue6-phase1-findings.md) and remains allowed below. + raise InvalidPeriodError( + "Weekly granularity with an explicit period is not supported: " + "Wordstat ignores the requested window and returns its default " + "range instead; use daily or monthly, or omit --date-from/--date-to" + ) else: end_month = date_from.month + 2 end_year = date_from.year + (end_month - 1) // 12 diff --git a/tests/test_periods.py b/tests/test_periods.py index af612e3..921bcb1 100644 --- a/tests/test_periods.py +++ b/tests/test_periods.py @@ -36,10 +36,18 @@ def test_daily_period_is_at_most_60_days_and_not_future(): validate_period(Granularity.DAILY, date(2026, 7, 1), date(2026, 8, 21), today=date(2026, 8, 21)) -def test_weekly_and_monthly_minimums(): - with pytest.raises(InvalidPeriodError, match="three calendar weeks"): +def test_weekly_with_explicit_period_is_rejected(): + # Live CDP checks (issue #6 phase 2) showed Wordstat's weekly view + # ignores an explicit window entirely and returns its default ~2-year + # range instead, so any explicit weekly period is rejected regardless + # of length -- there is no length that would make it honored. + with pytest.raises(InvalidPeriodError, match="ignores the requested window"): validate_period(Granularity.WEEKLY, date(2026, 8, 1), date(2026, 8, 20)) - validate_period(Granularity.WEEKLY, date(2026, 8, 1), date(2026, 8, 21)) + with pytest.raises(InvalidPeriodError, match="ignores the requested window"): + validate_period(Granularity.WEEKLY, date(2026, 8, 1), date(2026, 8, 21)) + + +def test_monthly_minimum(): with pytest.raises(InvalidPeriodError, match="three calendar months"): validate_period(Granularity.MONTHLY, date(2026, 8, 1), date(2026, 10, 1)) validate_period(Granularity.MONTHLY, date(2026, 8, 1), date(2026, 10, 31)) From 6cdcc8905f1bd9bdb5da4202944d2706c3a0e403 Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 13:55:30 +0700 Subject: [PATCH 07/15] refactor: drop the unreachable weekly date-picker path validate_period() now rejects an explicit weekly period before the browser is ever touched, so _set_period()'s 'week' popup_type and the matching branch in _select_calendar_date() are unreachable dead code. Removing them also collapses the duplicated month_label computation in _select_calendar_date() into one place, leaving a single function that drives both real remaining cases (month, day) instead of three parallel near-identical branches. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01JCyJr8vFFDW1eSMyCQyNNk --- src/wordstat/collector.py | 42 +++++++++++++++++++++------------------ 1 file changed, 23 insertions(+), 19 deletions(-) diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index 0ba9c7b..0c8a27e 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -533,9 +533,13 @@ async def _set_granularity(self, page, granularity: Granularity) -> None: async def _set_period(self, page, granularity: Granularity, date_from: date, date_to: date | None) -> None: if date_to is None: raise InvalidRequestError("A period end date is required when a start date is provided") - popup_type = "month" if granularity is Granularity.MONTHLY else ( - "week" if granularity is Granularity.WEEKLY else "day" - ) + # WEEKLY never reaches here: validate_period() (periods.py) rejects an + # explicit period for weekly granularity outright, because live CDP + # checks (issue #6 phase 2) showed Wordstat ignores the requested + # window for weekly and returns its own default range regardless of + # what was picked here. Only "month" and "day" popups are therefore + # real, reachable cases. + popup_type = "month" if granularity is Granularity.MONTHLY else "day" await self._click(page, DATE_RANGE_SELECTOR) await self._select_calendar_date(page, popup_type, date_from) await self._click(page, DATE_RANGE_SELECTOR) @@ -555,39 +559,39 @@ async def _wait_for_table_granularity(self, page, granularity: Granularity) -> N await self._wait_for(page, expression) async def _select_calendar_date(self, page, popup_type: str, target: date) -> None: + # One function for every reachable granularity's date popup ("month" + # for monthly, "day" for daily). "week" is deliberately not a case + # here: WEEKLY never calls this (see _set_period) because Wordstat + # ignores an explicit weekly period entirely, so there is no live + # behavior to encode for a week-typed popup and no support for + # emulating one without contradicting that finding. root = f".range-datepicker_type_{popup_type}" await self._wait_for( page, f"() => [...document.querySelectorAll({json.dumps(root)})].some((x) => " "x.offsetParent !== null && getComputedStyle(x).visibility !== 'hidden')", ) + # The live interface is Russian; map explicitly rather than depending + # on the host locale. + month_label = ( + "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split()[ + target.month - 1 + ] + ) if popup_type == "month": year_selector = f"{root} .datepicker-month__years_select > button" month_selector = f"{root} .react-datepicker__month-text" - # The live interface is Russian; map explicitly rather than - # depending on the host locale. - month_label = ( - "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split()[ - target.month - 1 - ] - ) else: year_selector = f"{root} .range-datepicker__years_select > button" month_selector = f"{root} .range-datepicker__months_select > button" - month_label = ( - "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split()[ - target.month - 1 - ] - ) await self._click(page, year_selector) await self._click_visible_text(page, ".Popup2_visible [role='option']", str(target.year)) - if popup_type in {"day", "week"}: - await self._click(page, month_selector) - await self._click_visible_text(page, ".Popup2_visible [role='option']", month_label) - day_selector = f"{root} button[name='day']" if popup_type == "month": await self._click_visible_text(page, month_selector, month_label.capitalize()) else: + await self._click(page, month_selector) + await self._click_visible_text(page, ".Popup2_visible [role='option']", month_label) + day_selector = f"{root} button[name='day']" await self._click_visible_date_button(page, day_selector, str(target.day)) async def _click_visible_date_button(self, page, selector: str, text: str) -> None: From 24f4c48e9e26d0901b1cf4ed0580e5f0fc89f357 Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 13:59:55 +0700 Subject: [PATCH 08/15] Revert "refactor: drop the unreachable weekly date-picker path" This reverts commit 6cdcc8905f1bd9bdb5da4202944d2706c3a0e403. --- src/wordstat/collector.py | 42 ++++++++++++++++++--------------------- 1 file changed, 19 insertions(+), 23 deletions(-) diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index 0c8a27e..0ba9c7b 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -533,13 +533,9 @@ async def _set_granularity(self, page, granularity: Granularity) -> None: async def _set_period(self, page, granularity: Granularity, date_from: date, date_to: date | None) -> None: if date_to is None: raise InvalidRequestError("A period end date is required when a start date is provided") - # WEEKLY never reaches here: validate_period() (periods.py) rejects an - # explicit period for weekly granularity outright, because live CDP - # checks (issue #6 phase 2) showed Wordstat ignores the requested - # window for weekly and returns its own default range regardless of - # what was picked here. Only "month" and "day" popups are therefore - # real, reachable cases. - popup_type = "month" if granularity is Granularity.MONTHLY else "day" + popup_type = "month" if granularity is Granularity.MONTHLY else ( + "week" if granularity is Granularity.WEEKLY else "day" + ) await self._click(page, DATE_RANGE_SELECTOR) await self._select_calendar_date(page, popup_type, date_from) await self._click(page, DATE_RANGE_SELECTOR) @@ -559,39 +555,39 @@ async def _wait_for_table_granularity(self, page, granularity: Granularity) -> N await self._wait_for(page, expression) async def _select_calendar_date(self, page, popup_type: str, target: date) -> None: - # One function for every reachable granularity's date popup ("month" - # for monthly, "day" for daily). "week" is deliberately not a case - # here: WEEKLY never calls this (see _set_period) because Wordstat - # ignores an explicit weekly period entirely, so there is no live - # behavior to encode for a week-typed popup and no support for - # emulating one without contradicting that finding. root = f".range-datepicker_type_{popup_type}" await self._wait_for( page, f"() => [...document.querySelectorAll({json.dumps(root)})].some((x) => " "x.offsetParent !== null && getComputedStyle(x).visibility !== 'hidden')", ) - # The live interface is Russian; map explicitly rather than depending - # on the host locale. - month_label = ( - "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split()[ - target.month - 1 - ] - ) if popup_type == "month": year_selector = f"{root} .datepicker-month__years_select > button" month_selector = f"{root} .react-datepicker__month-text" + # The live interface is Russian; map explicitly rather than + # depending on the host locale. + month_label = ( + "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split()[ + target.month - 1 + ] + ) else: year_selector = f"{root} .range-datepicker__years_select > button" month_selector = f"{root} .range-datepicker__months_select > button" + month_label = ( + "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split()[ + target.month - 1 + ] + ) await self._click(page, year_selector) await self._click_visible_text(page, ".Popup2_visible [role='option']", str(target.year)) + if popup_type in {"day", "week"}: + await self._click(page, month_selector) + await self._click_visible_text(page, ".Popup2_visible [role='option']", month_label) + day_selector = f"{root} button[name='day']" if popup_type == "month": await self._click_visible_text(page, month_selector, month_label.capitalize()) else: - await self._click(page, month_selector) - await self._click_visible_text(page, ".Popup2_visible [role='option']", month_label) - day_selector = f"{root} button[name='day']" await self._click_visible_date_button(page, day_selector, str(target.day)) async def _click_visible_date_button(self, page, selector: str, text: str) -> None: From f1a0c8f70eb373208b1639e66483d5d81aac2fc3 Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 13:59:55 +0700 Subject: [PATCH 09/15] Revert "fix: reject explicit weekly periods (Wordstat ignores the requested window)" This reverts commit 498ecff72aa600bf5ae4875d299fccf161fcc780. --- src/wordstat/periods.py | 15 ++------------- tests/test_periods.py | 14 +++----------- 2 files changed, 5 insertions(+), 24 deletions(-) diff --git a/src/wordstat/periods.py b/src/wordstat/periods.py index a5724bf..89f0dbb 100644 --- a/src/wordstat/periods.py +++ b/src/wordstat/periods.py @@ -46,19 +46,8 @@ def validate_period( if date_to - date_from >= timedelta(days=60): raise InvalidPeriodError("Daily statistics support at most 60 calendar days") elif granularity is Granularity.WEEKLY: - # Live CDP checks (issue #6 phase 2) confirmed Wordstat's weekly view - # silently ignores an explicit --date-from/--date-to and returns its - # default ~2-year window instead (107 full weeks, unrelated to the - # requested dates). Returning that mismatched window with a manifest - # claiming the requested period would be the worst outcome, so an - # explicit weekly period is rejected outright rather than collected - # and silently wrong. Weekly without an explicit period still works - # (see docs/issue6-phase1-findings.md) and remains allowed below. - raise InvalidPeriodError( - "Weekly granularity with an explicit period is not supported: " - "Wordstat ignores the requested window and returns its default " - "range instead; use daily or monthly, or omit --date-from/--date-to" - ) + if date_to - date_from < timedelta(days=20): + raise InvalidPeriodError("Weekly statistics require at least three calendar weeks") else: end_month = date_from.month + 2 end_year = date_from.year + (end_month - 1) // 12 diff --git a/tests/test_periods.py b/tests/test_periods.py index 921bcb1..af612e3 100644 --- a/tests/test_periods.py +++ b/tests/test_periods.py @@ -36,18 +36,10 @@ def test_daily_period_is_at_most_60_days_and_not_future(): validate_period(Granularity.DAILY, date(2026, 7, 1), date(2026, 8, 21), today=date(2026, 8, 21)) -def test_weekly_with_explicit_period_is_rejected(): - # Live CDP checks (issue #6 phase 2) showed Wordstat's weekly view - # ignores an explicit window entirely and returns its default ~2-year - # range instead, so any explicit weekly period is rejected regardless - # of length -- there is no length that would make it honored. - with pytest.raises(InvalidPeriodError, match="ignores the requested window"): +def test_weekly_and_monthly_minimums(): + with pytest.raises(InvalidPeriodError, match="three calendar weeks"): validate_period(Granularity.WEEKLY, date(2026, 8, 1), date(2026, 8, 20)) - with pytest.raises(InvalidPeriodError, match="ignores the requested window"): - validate_period(Granularity.WEEKLY, date(2026, 8, 1), date(2026, 8, 21)) - - -def test_monthly_minimum(): + validate_period(Granularity.WEEKLY, date(2026, 8, 1), date(2026, 8, 21)) with pytest.raises(InvalidPeriodError, match="three calendar months"): validate_period(Granularity.MONTHLY, date(2026, 8, 1), date(2026, 10, 1)) validate_period(Granularity.MONTHLY, date(2026, 8, 1), date(2026, 10, 31)) From 81f032de3a9ecaadeaf76264498eee4f27ff9ec2 Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 14:03:59 +0700 Subject: [PATCH 10/15] perf: make _collect_one's fixed settling/retry pauses test-injectable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit _collect_one always sleeps once per non-REGIONS view (settling for the export blob to catch up with a table repaint) and again on an empty-export retry — both are unconditional real-time pauses against the live UI, not _wait_for-style condition polling with an early return. Tests exercise _collect_one for real against a fake, instantly-responding page, so those sleeps were paid in full: 140 tests took 26.9s instead of well under 1s (root-caused by wordstat-1 from pytest --durations on PR #21). Both pauses are now constructor parameters (settling_seconds=1.0, empty_export_retry_seconds=2.0, matching prior hardcoded behavior) instead of hardcoded literals, so the CLI's real collector keeps its live-tested delays while tests can set them to 0. Full suite: 140 passed in 0.77s. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01JCyJr8vFFDW1eSMyCQyNNk --- src/wordstat/collector.py | 21 +++++++++++++--- tests/test_collector_batch.py | 46 +++++++++++++++++------------------ 2 files changed, 41 insertions(+), 26 deletions(-) diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index 0ba9c7b..d53a486 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -123,11 +123,25 @@ def __init__( output_root: Path, timeout_seconds: float = 45.0, keep_raw: bool = False, + settling_seconds: float = 1.0, + empty_export_retry_seconds: float = 2.0, ) -> None: self.cdp_url = cdp_url self.output_root = output_root self.timeout_seconds = timeout_seconds self.keep_raw = keep_raw + # Both are real, unconditional pauses against the live UI (see their + # call sites in _collect_one) — not upper-bound timeouts like + # timeout_seconds, which _wait_for polls against a condition and + # returns from early. Nothing in the live DOM signals "the export + # blob has been rebuilt" or "the empty table has repainted", so + # there is no condition for _wait_for to poll here; a fixed sleep is + # the only available fix. Kept as constructor parameters (not + # hardcoded) so tests against a fake, instantly-responding page can + # set them to 0 instead of actually blocking the test process for + # real wall-clock time on every view of every phrase. + self.settling_seconds = settling_seconds + self.empty_export_retry_seconds = empty_export_retry_seconds self._previous_table_snapshot: str | None = None async def collect( @@ -406,8 +420,8 @@ async def _collect_one( # blob has been rebuilt for the selected phrase/view. A short # settling interval prevents a header-only CSV from racing the # table repaint; the structural row gate above remains required. - if view is not WordstatView.REGIONS: - await asyncio.sleep(1.0) + if view is not WordstatView.REGIONS and self.settling_seconds > 0: + await asyncio.sleep(self.settling_seconds) if view is WordstatView.DYNAMICS and ( granularity is not Granularity.MONTHLY or date_from is not None ): @@ -422,7 +436,8 @@ async def _collect_one( # empty table export once before failing closed. dataset = parse_wordstat_csv(source, view) if _is_untrustworthy_empty_export(view, dataset): - await asyncio.sleep(2.0) + if self.empty_export_retry_seconds > 0: + await asyncio.sleep(self.empty_export_retry_seconds) source = await self._download_current_view(page, session, downloads_path) dataset = parse_wordstat_csv(source, view) if _is_untrustworthy_empty_export(view, dataset): diff --git a/tests/test_collector_batch.py b/tests/test_collector_batch.py index 2c04ae2..544fddd 100644 --- a/tests/test_collector_batch.py +++ b/tests/test_collector_batch.py @@ -96,7 +96,7 @@ def test_collect_many_reuses_one_session_for_all_phrases(monkeypatch, tmp_path): _patch_common(monkeypatch) _patch_collect_one(monkeypatch, lambda phrase: _async(_fake_result(tmp_path, phrase))) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["чай", "кофе", "вода"])) assert _FakeSession.instances == 1 @@ -115,7 +115,7 @@ async def flaky(phrase): _patch_collect_one(monkeypatch, flaky) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["первый", "сломается", "второй"])) assert batch.total == 3 @@ -139,7 +139,7 @@ async def flaky(phrase): _patch_collect_one(monkeypatch, flaky) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["первый", "сломается", "второй"])) assert [r.run_directory.name for r in batch.results] == ["первый", "второй"] @@ -172,7 +172,7 @@ async def collect_one( monkeypatch.setattr(WordstatCollector, "_collect_one", collect_one) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["первый", "второй", "третий"])) assert attempted == ["первый", "второй"] # "третий" was never attempted @@ -197,7 +197,7 @@ async def counting_assert_authenticated(self, page): monkeypatch.setattr(WordstatCollector, "_assert_authenticated", counting_assert_authenticated) _patch_collect_one_passthrough(monkeypatch, tmp_path) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) asyncio.run(collector.collect_many(["первый", "второй", "третий"])) # Once before the loop (collect_many) + once per phrase (_collect_one). @@ -254,7 +254,7 @@ async def recording_collect_one( monkeypatch.setattr(WordstatCollector, "_collect_one", recording_collect_one) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) asyncio.run(collector.collect_many(["первый", "второй", "третий"])) assert set_region_calls == [True, False, False] @@ -293,7 +293,7 @@ async def recording_collect_one( monkeypatch.setattr(WordstatCollector, "_collect_one", recording_collect_one) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["первый", "второй", "третий"])) assert set_region_calls == [True, True, False] @@ -338,7 +338,7 @@ async def collect_one( monkeypatch.setattr(WordstatCollector, "_collect_one", collect_one) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["первый", "второй", "третий"])) # Region was actually applied while handling "первый" — the second @@ -361,28 +361,28 @@ async def failing_stop(self): monkeypatch.setattr(_FakeSession, "stop", failing_stop) _patch_collect_one(monkeypatch, lambda phrase: _async(_fake_result(tmp_path, phrase))) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) batch = asyncio.run(collector.collect_many(["чай"])) assert [r.run_directory.name for r in batch.results] == ["чай"] def test_collect_many_rejects_an_empty_phrase_list(tmp_path): - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) with pytest.raises(InvalidRequestError, match="At least one search phrase"): asyncio.run(collector.collect_many([])) def test_collect_many_rejects_a_blank_phrase_in_the_middle(tmp_path): - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) with pytest.raises(InvalidRequestError, match="must not be empty"): asyncio.run(collector.collect_many(["чай", " ", "кофе"])) def test_collect_many_rejects_a_blank_region(tmp_path): - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) with pytest.raises(InvalidRequestError, match="region must not be empty"): asyncio.run(collector.collect_many(["чай"], region=" ")) @@ -392,7 +392,7 @@ def test_collect_wraps_collect_many_and_returns_the_single_result(monkeypatch, t _patch_common(monkeypatch) _patch_collect_one(monkeypatch, lambda phrase: _async(_fake_result(tmp_path, phrase))) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) result = asyncio.run(collector.collect(phrase="чай")) assert result.run_directory.name == "чай" @@ -406,7 +406,7 @@ async def fail(phrase): _patch_collect_one(monkeypatch, fail) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) with pytest.raises(PhraseEntryError, match="boom"): asyncio.run(collector.collect(phrase="чай")) @@ -455,7 +455,7 @@ async def download(self, page, session, directory): monkeypatch.setattr(WordstatCollector, "_wait_for", wait) monkeypatch.setattr(WordstatCollector, "_download_current_view", download) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) page = _FakePage() session = _FakeSession() asyncio.run(collector._collect_one(page, session, downloads_path, "первая", "Россия", set_region=False)) @@ -499,7 +499,7 @@ def failing_write_dataset(dataset, run_directory): monkeypatch.setattr(WordstatCollector, "_download_current_view", fake_download) monkeypatch.setattr(collector_module, "write_dataset", failing_write_dataset) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run(): page = _FakePage() @@ -551,7 +551,7 @@ def flaky_write_dataset(dataset, run_directory): monkeypatch.setattr(WordstatCollector, "_download_current_view", fake_download) monkeypatch.setattr(collector_module, "write_dataset", flaky_write_dataset) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run(): page = _FakePage() @@ -599,7 +599,7 @@ def recording_write_manifest(path, manifest): monkeypatch.setattr(WordstatCollector, "_download_current_view", fake_download) monkeypatch.setattr(collector_module, "write_manifest", recording_write_manifest) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run(): page = _FakePage() @@ -652,7 +652,7 @@ async def failing_after_first_download(self, page, session, dl_path): monkeypatch.setattr(WordstatCollector, "_select_view", fake_select_view) monkeypatch.setattr(WordstatCollector, "_download_current_view", failing_after_first_download) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run_first(): page = _FakePage() @@ -718,7 +718,7 @@ async def fake_download(self, page, session, dl_path): monkeypatch.setattr(WordstatCollector, "_select_view", fake_select_view) monkeypatch.setattr(WordstatCollector, "_download_current_view", fake_download) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run(): page = _FakePage() @@ -744,7 +744,7 @@ async def run_resume_with_wrong_phrase(): def test_collect_many_rejects_resume_directory_with_more_than_one_phrase(tmp_path): - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) resume_dir = tmp_path / "some-run" resume_dir.mkdir() @@ -776,7 +776,7 @@ async def failing_download(self, page, session, dl_path): monkeypatch.setattr(WordstatCollector, "_select_view", fake_select_view) monkeypatch.setattr(WordstatCollector, "_download_current_view", failing_download) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run_first(): page = _FakePage(url="https://wordstat.yandex.ru/?words=тест®ion=Россия") @@ -848,7 +848,7 @@ async def fake_download(self, page, session, dl_path): monkeypatch.setattr(WordstatCollector, "_select_view", fake_select_view) monkeypatch.setattr(WordstatCollector, "_download_current_view", fake_download) - collector = WordstatCollector("cdp", tmp_path) + collector = WordstatCollector("cdp", tmp_path, settling_seconds=0, empty_export_retry_seconds=0) async def run_full(): page = _FakePage(url="https://wordstat.yandex.ru/?words=тест") From 255bca70e4aa15e667032b02842ed63cd8b64318 Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 14:31:31 +0700 Subject: [PATCH 11/15] fix: wait for the dynamics table's first row to match the requested period MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit _wait_for_table_granularity only checked the first cell's *format* (e.g. "looks like a week range"), which the table's pre-existing content already satisfied before a period change was applied -- the date-range button showed the newly picked dates while the table itself still held stale rows, and the exported CSV silently carried Wordstat's default window instead of the requested one. This was the actual root cause behind an earlier, incorrect conclusion that Wordstat ignores explicit weekly periods (it doesn't: a live screenshot of the UI showed a 2018-12-24 explicit weekly period applied correctly by hand); that conclusion has been reverted. _set_period now waits for the first cell to actually start at the expected date via a new _wait_for_period_applied, turning a silently wrong-period export into a loud InterfaceChangedError if the picker never catches up. Two related fixes surfaced during live verification: - The dynamics table's first-cell month name is genitive for daily/weekly ("22 июня", "24 декабря" -- "of June"/"of December") but nominative for monthly ("август 2024"). Matching daily/weekly's genitive text against the existing nominative RUSSIAN_MONTHS list (used for clicking the calendar popups) never matched, so a first attempt at this wait ran out its full timeout on every weekly/daily request. Added RUSSIAN_MONTHS_GENITIVE and match the exact expected month, not just any month word. - Weekly's first cell is the Monday of the requested date's week, not the date itself (confirmed live) -- not a bug, Wordstat's own alignment. Live-verified against the real Wordstat UI: - Weekly, requested 2018-12-24..2019-01-13: actual_period 24.12.2018..07.01.2019 (07.01 is the start of the last full week ending 13.01 -- the period applies exactly, 3 full weeks, no truncation). - Daily, requested 2026-07-22..2026-08-20: actual_period 22.07.2026..19.08.2026 (29 of 30 rows -- the same variable trailing-tail behavior already documented for daily dynamics in phase 1, not a new defect). 143 passed in 0.68s, ruff clean. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01JCyJr8vFFDW1eSMyCQyNNk --- src/wordstat/collector.py | 69 +++++++++++++++++++++++++++++------- tests/test_collector_view.py | 64 +++++++++++++++++++++++++++++++++ 2 files changed, 120 insertions(+), 13 deletions(-) diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index d53a486..505ff19 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -2,6 +2,7 @@ import asyncio import json +import re import tempfile import time from collections.abc import Callable @@ -55,6 +56,19 @@ WordstatView.DYNAMICS: "label[for='graph']", WordstatView.REGIONS: "label[for='map']", } +# The live interface is Russian; map explicitly rather than depending on the +# host locale. Shared by every place that needs a Russian month name (the +# calendar popups and the applied-period wait below), so there is exactly +# one list to keep in sync with the live UI's wording. Nominative form — +# what the calendar popups' option labels use ("январь", "декабрь"). +RUSSIAN_MONTHS = "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split() +# Genitive form — what the dynamics table's first cell uses for daily/weekly +# rows ("22 июня", "24 декабря 2018": "of June", "of December"). Confirmed +# live (issue #6 phase 1/2 documents) that this differs from RUSSIAN_MONTHS; +# a fixed nominative-month match against this genitive text never matches. +RUSSIAN_MONTHS_GENITIVE = ( + "января февраля марта апреля мая июня июля августа сентября октября ноября декабря".split() +) def _is_untrustworthy_empty_export(view: WordstatView, dataset: CsvDataset) -> bool: @@ -555,7 +569,47 @@ async def _set_period(self, page, granularity: Granularity, date_from: date, dat await self._select_calendar_date(page, popup_type, date_from) await self._click(page, DATE_RANGE_SELECTOR) await self._select_calendar_date(page, popup_type, date_to) - await self._wait_for_table_granularity(page, granularity) + # _wait_for_table_granularity alone is not enough here: it only + # checks that the first cell's *format* matches the granularity + # (e.g. "looks like a week range"), which is already true of the + # table's pre-existing content before the period change has been + # applied. Live CDP checks (issue #6 phase 2) caught this exact + # race: the date-range button already showed the newly picked + # dates, _wait_for_table_granularity's format check passed + # immediately, and the exported CSV still carried Wordstat's + # default window — the table's *values* had not caught up yet. + # Waiting for the first cell to actually start at the expected + # date closes that gap, and turns a silent wrong-period export + # into a loud InterfaceChangedError if the picker never catches up. + await self._wait_for_period_applied(page, granularity, date_from) + + async def _wait_for_period_applied(self, page, granularity: Granularity, date_from: date) -> None: + # The month name in the first cell is genitive ("22 июня", "24 + # декабря" — day-of/week-of a date) for daily/weekly, but + # nominative ("август 2024") for monthly (confirmed live, issue #6 + # phase 1/2 documents) — hence two separate month lists rather than + # one. Matching the exact expected month (not just "any month word") + # matters: without it, "24 2018" would silently accept + # a wrong month picked by a stuck calendar, defeating the point of + # this wait (confirmed as a real gap during review, not just theory). + if granularity is Granularity.MONTHLY: + expected = f"{RUSSIAN_MONTHS[date_from.month - 1]} {date_from.year}" + elif granularity is Granularity.DAILY: + expected = f"{date_from.day} {RUSSIAN_MONTHS_GENITIVE[date_from.month - 1]}" + else: + # Weekly's first cell is the Monday of date_from's week, not + # date_from itself (confirmed live: 2018-12-26, a Wednesday, + # produced a first cell starting "24 декабря" — the preceding + # Monday). Wordstat does this alignment itself; not a bug. + week_start = date_from - timedelta(days=date_from.weekday()) + expected = f"{week_start.day} {RUSSIAN_MONTHS_GENITIVE[week_start.month - 1]} {week_start.year}" + pattern = "^" + re.escape(expected) + await self._wait_for( + page, + f"() => new RegExp({json.dumps(pattern)}).test(" + f"document.querySelector({json.dumps(TABLE_ROW_SELECTOR)})?.querySelector('td')" + "?.textContent?.trim() ?? '')", + ) async def _wait_for_table_granularity(self, page, granularity: Granularity) -> None: patterns = { @@ -576,24 +630,13 @@ async def _select_calendar_date(self, page, popup_type: str, target: date) -> No f"() => [...document.querySelectorAll({json.dumps(root)})].some((x) => " "x.offsetParent !== null && getComputedStyle(x).visibility !== 'hidden')", ) + month_label = RUSSIAN_MONTHS[target.month - 1] if popup_type == "month": year_selector = f"{root} .datepicker-month__years_select > button" month_selector = f"{root} .react-datepicker__month-text" - # The live interface is Russian; map explicitly rather than - # depending on the host locale. - month_label = ( - "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split()[ - target.month - 1 - ] - ) else: year_selector = f"{root} .range-datepicker__years_select > button" month_selector = f"{root} .range-datepicker__months_select > button" - month_label = ( - "январь февраль март апрель май июнь июль август сентябрь октябрь ноябрь декабрь".split()[ - target.month - 1 - ] - ) await self._click(page, year_selector) await self._click_visible_text(page, ".Popup2_visible [role='option']", str(target.year)) if popup_type in {"day", "week"}: diff --git a/tests/test_collector_view.py b/tests/test_collector_view.py index 26ca7dc..807ccae 100644 --- a/tests/test_collector_view.py +++ b/tests/test_collector_view.py @@ -1,12 +1,15 @@ """View-selection retry behavior without touching a browser.""" import asyncio +import re +from datetime import date import pytest from wordstat.collector import WordstatCollector, _is_untrustworthy_empty_export from wordstat.errors import InterfaceChangedError from wordstat.models import CsvDataset, WordstatView +from wordstat.periods import Granularity def test_select_view_retries_once_when_active_marker_does_not_change(monkeypatch, tmp_path): @@ -197,3 +200,64 @@ async def wait(self, page, expression, seconds=None, required=True): asyncio.run(WordstatCollector("cdp", tmp_path)._select_view(object(), "label[for='map']", WordstatView.REGIONS)) assert ".table__wrapper" not in waits[0] + + +# _wait_for_period_applied's regex must match the live table's actual text. +# Daily/weekly first cells are genitive ("22 июня", "24 декабря 2018" — "of +# June", "of December"), unlike the nominative RUSSIAN_MONTHS list used for +# clicking the calendar popups ("июнь", "декабрь"). A live CDP check (issue +# #6 phase 2) confirmed a fixed nominative-month match against the live +# cell's genitive text never matches, so _wait_for ran out its full timeout +# every time instead of detecting the period was actually already applied. +# These are regression guards for that exact mismatch, not for the general +# _wait_for polling mechanism (already covered above). +def _captured_pattern(monkeypatch, tmp_path): + captured = {} + + async def wait(self, page, expression, seconds=None, required=True): + captured["expression"] = expression + + monkeypatch.setattr(WordstatCollector, "_wait_for", wait) + return WordstatCollector("cdp", tmp_path), captured + + +def _extract_regex(expression: str) -> re.Pattern: + # _wait_for_period_applied builds "new RegExp()...". + # Extract and compile the same pattern the live page would receive. + import json + + start = expression.index("new RegExp(") + len("new RegExp(") + end = expression.index(")", start) + pattern = json.loads(expression[start:end]) + return re.compile(pattern) + + +def test_wait_for_period_applied_daily_matches_genitive_cell_text(monkeypatch, tmp_path): + collector, captured = _captured_pattern(monkeypatch, tmp_path) + asyncio.run(collector._wait_for_period_applied(object(), Granularity.DAILY, date(2026, 6, 22))) + pattern = _extract_regex(captured["expression"]) + assert pattern.search("22 июня") + assert not pattern.search("23 июня") + # The month must match exactly, not just "some month word" — a stuck + # calendar that landed on the right day of a wrong month must still + # fail this check (found in review: an early version anchored only on + # day+year and would have silently accepted this). + assert not pattern.search("22 июля") + + +def test_wait_for_period_applied_weekly_matches_genitive_cell_text_and_aligns_to_monday(monkeypatch, tmp_path): + collector, captured = _captured_pattern(monkeypatch, tmp_path) + # 2018-12-26 is a Wednesday; the first cell is the Monday of that week. + asyncio.run(collector._wait_for_period_applied(object(), Granularity.WEEKLY, date(2018, 12, 26))) + pattern = _extract_regex(captured["expression"]) + assert pattern.search("24 декабря 2018 – 30 декабря 2018") + assert not pattern.search("31 декабря 2018 – 6 января 2019") + assert not pattern.search("24 ноября 2018 – 30 ноября 2018") + + +def test_wait_for_period_applied_monthly_matches_nominative_cell_text(monkeypatch, tmp_path): + collector, captured = _captured_pattern(monkeypatch, tmp_path) + asyncio.run(collector._wait_for_period_applied(object(), Granularity.MONTHLY, date(2024, 8, 1))) + pattern = _extract_regex(captured["expression"]) + assert pattern.search("август 2024") + assert not pattern.search("сентябрь 2024") From d2030130c9a37d2adb4c7a359b14cc23c6f6b9ff Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 14:32:22 +0700 Subject: [PATCH 12/15] docs: clarify settling_seconds is preventive, unverified, and has a cost Per wordstat-1's decision: keep the default (1.0s, unchanged) but stop the comment from reading like a description of a caught bug. State plainly: introduced without a live timing measurement, no confirmed repro for this specific race, +1s/view (+3s/phrase, +50s on a 50-phrase batch), and why removing it can't currently be verified safe (issue #22 blocks a full CLI run from ever reaching DYNAMICS). Separately note that empty_export_retry_seconds is unlike it: it only fires after a concrete, already-observed empty-export trigger, not preventively. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01JCyJr8vFFDW1eSMyCQyNNk --- src/wordstat/collector.py | 33 ++++++++++++++++++++++++++------- 1 file changed, 26 insertions(+), 7 deletions(-) diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index 505ff19..f7d5b6d 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -147,13 +147,32 @@ def __init__( # Both are real, unconditional pauses against the live UI (see their # call sites in _collect_one) — not upper-bound timeouts like # timeout_seconds, which _wait_for polls against a condition and - # returns from early. Nothing in the live DOM signals "the export - # blob has been rebuilt" or "the empty table has repainted", so - # there is no condition for _wait_for to poll here; a fixed sleep is - # the only available fix. Kept as constructor parameters (not - # hardcoded) so tests against a fake, instantly-responding page can - # set them to 0 instead of actually blocking the test process for - # real wall-clock time on every view of every phrase. + # returns early. Kept as constructor parameters (not hardcoded) so + # tests against a fake, instantly-responding page can set them to 0 + # instead of actually blocking the test process for real wall-clock + # time on every view of every phrase (140 tests went from 26.9s to + # 0.77s once collect_many's own tests did this — see git history). + # + # settling_seconds (introduced in efaa9ba, issue #6 phase 2): a + # preventive pause before downloading each non-map view, guarding + # against a header-only CSV race first observed for issue #11 + # (Wordstat can return a checked radio + enabled download button + # before the export blob has actually been rebuilt for the newly + # selected phrase/view). No one has reproduced *this specific* + # settling race with reliable repro steps — the 1.0s value was + # chosen without a live timing measurement, not derived from one. + # Cost: +1s per non-map view, +3s per phrase (3 of 4 views are + # non-map), so +50s across a 50-phrase batch. Do not remove without + # a way to verify nothing regresses — issue #22 currently blocks a + # full CLI run from reaching DYNAMICS at all (see collector.py's + # fail-closed behavior on empty top exports), so there is no cheap + # end-to-end signal today that would catch a regression from + # removing this. + # + # empty_export_retry_seconds: unlike the above, this one does have + # a concrete, structural trigger — it only fires after + # _is_untrustworthy_empty_export has already caught a table-visible- + # but-CSV-empty export on this specific run, not preventively. self.settling_seconds = settling_seconds self.empty_export_retry_seconds = empty_export_retry_seconds self._previous_table_snapshot: str | None = None From 2a9689ef45e828832f9d68d5e2240cccf44aad09 Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 15:01:45 +0700 Subject: [PATCH 13/15] fix: guard the dynamics period's end boundary and daily lower bound Codex review (cycle-review round 1) found that _wait_for_period_applied only confirms the requested window's start (date_from) before downloading; a table that has repainted its first row but not yet its last row could still export and be marked complete with a stale end boundary. Add a containment check in _assert_contiguous_dynamics_rows: exported rows must fall inside the requested window (shorter is legitimate -- the documented variable daily trailing tail -- but a row outside either boundary is not). Weekly's lower bound compares against the aligned Monday of date_from's week, not the raw date, matching Wordstat's own snapping behavior. Separately, a live CDP measurement (this cycle's investigation of a review finding) confirmed issue #6 step 3's "daily granularity outside the 60-day window ... reject with a domain error" is only half implemented: validate_period bounds window length but not how far date_from can be from today. The daily date-range picker's year-select popup only ever offers the current year, so a request like --date-from 2019-01-01 passed pre-flight validation and failed deep inside calendar-click code with an opaque InterfaceChangedError. Add the missing pre-flight bound, expressed relative to `today` (not "only the current year" -- that would wrongly reject legal windows in January) and consistent with the existing 60-day window-length check: PR #20's live-confirmed window 22.06.2026-20.08.2026 remains accepted. Tests: uv run pytest -q (147 passed), uv run ruff check (clean). --- src/wordstat/collector.py | 59 +++++++++++++++++++++++++++++++++-- src/wordstat/periods.py | 18 ++++++++++- tests/test_manifest_period.py | 55 ++++++++++++++++++++++++++++++++ tests/test_periods.py | 22 +++++++++++++ 4 files changed, 151 insertions(+), 3 deletions(-) diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index f7d5b6d..69996b3 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -479,7 +479,9 @@ async def _collect_one( "at least one table row before the download was triggered; export is not trustworthy" ) if view is WordstatView.DYNAMICS: - self._assert_contiguous_dynamics_rows(dataset, granularity) + self._assert_contiguous_dynamics_rows( + dataset, granularity, date_from=date_from, date_to=date_to + ) file_name = self._dynamics_file_name(granularity) if view is WordstatView.DYNAMICS else None if file_name is None: data_path, dtypes = write_dataset(dataset, run_directory) @@ -538,7 +540,12 @@ def _actual_period(dataset: CsvDataset) -> dict[str, str] | None: return {"field": field, "from": dataset.rows[0][field], "to": dataset.rows[-1][field]} @staticmethod - def _assert_contiguous_dynamics_rows(dataset: CsvDataset, granularity: Granularity) -> None: + def _assert_contiguous_dynamics_rows( + dataset: CsvDataset, + granularity: Granularity, + date_from: date | None = None, + date_to: date | None = None, + ) -> None: if granularity is Granularity.MONTHLY or len(dataset.rows) < 2: return field = dataset.headers[0] @@ -555,6 +562,42 @@ def _assert_contiguous_dynamics_rows(dataset: CsvDataset, granularity: Granulari f"Wordstat returned a gap in the {granularity.value} dynamics series " f"between {previous:%d.%m.%Y} and {current:%d.%m.%Y}" ) + # Containment, not equality: a shorter export fully inside the + # requested window is legitimate (the daily series' variable + # trailing tail, confirmed live — issue #6 phase 1 document), so + # this never compares dates[0]/dates[-1] against date_from/date_to + # for exact equality. What it does reject is a row *outside* the + # requested window in either direction — the live race this guards + # against (collector.py:603, _wait_for_period_applied) confirms + # only the first cell's format+value match date_from before + # downloading; it has no equivalent check for date_to, so a table + # that has not yet repainted past a previous, stale window can + # download and be recorded as complete with data that never + # entered the requested window at all. Only applies when an + # explicit period was requested (date_from/date_to are None for + # the UI's default window, which this function cannot validate + # against any specific boundary). + if date_from is not None and date_to is not None: + # Weekly rows are keyed by the Monday of date_from's week, not + # date_from itself (confirmed live — _wait_for_period_applied's + # own comment above), so the lower bound must be that aligned + # Monday, not the raw requested date, or a legitimate alignment + # would be misreported as a stale window. + lower_bound = ( + date_from - timedelta(days=date_from.weekday()) + if granularity is Granularity.WEEKLY + else date_from + ) + if dates[0] < lower_bound: + raise InterfaceChangedError( + f"Wordstat {granularity.value} dynamics export starts at {dates[0]:%d.%m.%Y}, " + f"before the requested window start {lower_bound:%d.%m.%Y}" + ) + if dates[-1] > date_to: + raise InterfaceChangedError( + f"Wordstat {granularity.value} dynamics export ends at {dates[-1]:%d.%m.%Y}, " + f"after the requested window end {date_to:%d.%m.%Y}" + ) @staticmethod @@ -600,6 +643,18 @@ async def _set_period(self, page, granularity: Granularity, date_from: date, dat # Waiting for the first cell to actually start at the expected # date closes that gap, and turns a silent wrong-period export # into a loud InterfaceChangedError if the picker never catches up. + # + # This only confirms the *start* boundary (date_from), because it + # runs before the download and only the first row is visible/known + # at this point. It does not by itself prove the *end* boundary + # (date_to) has also repainted — a table that has advanced its + # first row but not yet its last row would pass this wait and + # still download a stale end. The end boundary is guarded + # separately, after the download, by + # _assert_contiguous_dynamics_rows' containment check against the + # actually-exported rows (confirmed as a real gap during review, + # not just theory) — that is the authoritative defense for + # date_to, not an earlier DOM-state wait for it. await self._wait_for_period_applied(page, granularity, date_from) async def _wait_for_period_applied(self, page, granularity: Granularity, date_from: date) -> None: diff --git a/src/wordstat/periods.py b/src/wordstat/periods.py index 89f0dbb..2bc4805 100644 --- a/src/wordstat/periods.py +++ b/src/wordstat/periods.py @@ -29,7 +29,9 @@ def validate_period( Omitted dates leave the UI's default window untouched. The five-year maximum is deliberately not enforced because phase 1 did not establish - it as a reliable live fact. + it as a reliable live fact — unlike the daily lower bound below, which + is a confirmed live limitation of the picker itself, not a policy + choice. """ if (date_from is None) != (date_to is None): raise InvalidPeriodError("--date-from and --date-to must be provided together") @@ -45,6 +47,20 @@ def validate_period( raise InvalidPeriodError("Daily statistics cannot be requested after today") if date_to - date_from >= timedelta(days=60): raise InvalidPeriodError("Daily statistics support at most 60 calendar days") + # Live CDP measurement (issue #6 phase 2): the daily date-range + # picker's year-select popup only ever offers the current year, so + # a date_from further back than the trailing 60-day window cannot + # actually be selected in the live UI at all — it currently reaches + # Chrome and fails deep inside calendar-click code with an opaque + # InterfaceChangedError instead of a clear pre-flight rejection. + # Expressed relative to `today` (not "reject any year but the + # current one" — that would wrongly reject legal windows in + # January) and kept consistent with the existing 60-day window + # check above: PR #20's live-confirmed window 22.06.2026-20.08.2026, + # measured on 21.08.2026, is exactly 60 days back from today and + # must remain accepted. + if current - date_from > timedelta(days=60): + raise InvalidPeriodError("Daily statistics cannot start more than 60 days in the past") elif granularity is Granularity.WEEKLY: if date_to - date_from < timedelta(days=20): raise InvalidPeriodError("Weekly statistics require at least three calendar weeks") diff --git a/tests/test_manifest_period.py b/tests/test_manifest_period.py index bba4b98..a6abc04 100644 --- a/tests/test_manifest_period.py +++ b/tests/test_manifest_period.py @@ -1,3 +1,5 @@ +from datetime import date + import pytest from wordstat.collector import WordstatCollector @@ -41,3 +43,56 @@ def test_dynamics_series_allows_short_tail_but_rejects_internal_gap(): short_tail = dataset.model_copy(update={"rows": [{"Дата": "22.06.2026"}, {"Дата": "23.06.2026"}]}) WordstatCollector._assert_contiguous_dynamics_rows(short_tail, Granularity.DAILY) + + +def test_dynamics_series_rejects_a_stale_window_that_predates_the_request(): + # A readiness gate that only confirmed the first row matched date_from + # could pass for a table that still shows an earlier, stale window + # whose *first* row happens to coincide with date_from by chance, while + # its last row never advanced to cover date_to. This exercises the + # containment check directly on the exported rows (the actual defense + # against a stale end boundary), independent of the live DOM race that + # motivated it (collector.py:603). + dataset = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Дата"], + rows=[ + {"Дата": "22.06.2026"}, + {"Дата": "23.06.2026"}, + {"Дата": "24.06.2026"}, + ], + ) + # A shorter export fully inside the requested window is legitimate + # (documented variable trailing tail) and must not be rejected. + WordstatCollector._assert_contiguous_dynamics_rows( + dataset, Granularity.DAILY, date_from=date(2026, 6, 22), date_to=date(2026, 8, 20) + ) + # A row before the requested start is never legitimate: it means the + # table had not advanced past a previous, stale window. + with pytest.raises(InterfaceChangedError, match="requested window"): + WordstatCollector._assert_contiguous_dynamics_rows( + dataset, Granularity.DAILY, date_from=date(2026, 6, 23), date_to=date(2026, 8, 20) + ) + # A row after the requested end is never legitimate either: it means + # the table is still showing a different, later window than requested. + with pytest.raises(InterfaceChangedError, match="requested window"): + WordstatCollector._assert_contiguous_dynamics_rows( + dataset, Granularity.DAILY, date_from=date(2026, 6, 22), date_to=date(2026, 6, 23) + ) + + +def test_weekly_containment_uses_the_aligned_week_start_not_the_raw_request(): + # Wordstat snaps a weekly window's first row to the Monday of + # date_from's week (confirmed live: 2018-12-26, a Wednesday, produced + # a first cell starting 24.12.2018 -- the preceding Monday). A + # containment check that compared against the raw date_from instead of + # that aligned Monday would misreport this legitimate alignment as a + # stale window. + dataset = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Неделя с"], + rows=[{"Неделя с": "24.12.2018"}, {"Неделя с": "31.12.2018"}], + ) + WordstatCollector._assert_contiguous_dynamics_rows( + dataset, Granularity.WEEKLY, date_from=date(2018, 12, 26), date_to=date(2019, 1, 13) + ) diff --git a/tests/test_periods.py b/tests/test_periods.py index af612e3..8aeddb9 100644 --- a/tests/test_periods.py +++ b/tests/test_periods.py @@ -36,6 +36,28 @@ def test_daily_period_is_at_most_60_days_and_not_future(): validate_period(Granularity.DAILY, date(2026, 7, 1), date(2026, 8, 21), today=date(2026, 8, 21)) +def test_daily_period_rejects_a_start_date_further_back_than_the_trailing_60_day_window(): + # Live CDP measurement (issue #6 phase 2): the daily date-range + # picker's own year-select popup only ever offers the current year — + # a request for a valid-length window that starts far in the past + # (e.g. 2019) is accepted by validate_period today, reaches the live + # UI, and fails deep inside calendar-click code with an + # InterfaceChangedError instead of being rejected up front. issue #6 + # step 3 explicitly asks for "daily granularity outside the 60-day + # window" to be rejected before opening Chrome, which this closes. + # + # The bound is expressed relative to `today`, not the picker's year + # list (a year-based rule would wrongly reject legal windows early in + # January) and shares the existing 60-calendar-day convention: PR #20's + # live-confirmed window 22.06.2026-20.08.2026, measured on 21.08.2026, + # is exactly 60 days and must remain accepted. + validate_period(Granularity.DAILY, date(2026, 6, 22), date(2026, 8, 20), today=date(2026, 8, 21)) + with pytest.raises(InvalidPeriodError, match="60 days in the past"): + validate_period(Granularity.DAILY, date(2019, 1, 1), date(2019, 2, 15), today=date(2026, 8, 21)) + with pytest.raises(InvalidPeriodError, match="60 days in the past"): + validate_period(Granularity.DAILY, date(2026, 6, 21), date(2026, 6, 25), today=date(2026, 8, 21)) + + def test_weekly_and_monthly_minimums(): with pytest.raises(InvalidPeriodError, match="three calendar weeks"): validate_period(Granularity.WEEKLY, date(2026, 8, 1), date(2026, 8, 20)) From dc9b16b32d8f3dc9bd859ad50399d22ef6929425 Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 15:01:53 +0700 Subject: [PATCH 14/15] fix: strip a UTF-8 BOM from --phrases-file instead of leaking it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Pre-existing bug, unrelated to issue #6 (introduced in 94b1a66, batch mode). /review flagged that _PHRASES_FILE_ENCODINGS tried plain "utf-8" before "cp1251", and a phrases file saved with a UTF-8 byte-order mark decodes successfully under plain "utf-8" too -- just with the BOM character left attached to the first line. Neither resolve_phrases' line.strip() nor the collector's own phrase.strip() removes it (''.isspace() is False), so the first phrase silently carried an invisible leading character into the typed Wordstat search, the run-directory slug, and manifest.json. Fix: try "utf-8-sig" before "cp1251" (it strips a BOM if present and decodes identically to "utf-8" otherwise). Tests: uv run pytest -q (147 passed), uv run ruff check (clean). --- src/wordstat/cli.py | 14 ++++++++++---- tests/test_cli.py | 12 ++++++++++++ 2 files changed, 22 insertions(+), 4 deletions(-) diff --git a/src/wordstat/cli.py b/src/wordstat/cli.py index 1ac64f9..5939f51 100644 --- a/src/wordstat/cli.py +++ b/src/wordstat/cli.py @@ -14,10 +14,16 @@ _CONFIG = load_config() _DEFAULT_CDP_URL = _CONFIG.get("cdp_url", "http://127.0.0.1:9222") # Wordstat exports CSV as UTF-8 with BOM; a phrases file typed or saved -# on the same machine can plausibly be in either. Mirrors the encoding probe -# in csv_io.py, minus utf-8-sig (a phrases file is authored by hand, not -# exported by Wordstat, so a BOM is unlikely but harmless either way). -_PHRASES_FILE_ENCODINGS = ("utf-8", "cp1251") +# on the same machine can plausibly be in either, and utf-8-sig must come +# before plain utf-8: a BOM'd file decodes successfully under plain +# "utf-8" too, but leaves "" attached to the first line. Neither +# resolve_phrases' line.strip() nor the collector's own phrase.strip() +# removes it (''.isspace() is False), so a BOM left in is not +# harmless — it silently prepends an invisible character to the first +# phrase, which then propagates into the typed Wordstat search, the +# run-directory slug, and manifest.json. Mirrors the encoding probe in +# csv_io.py. +_PHRASES_FILE_ENCODINGS = ("utf-8-sig", "cp1251") @click.group() diff --git a/tests/test_cli.py b/tests/test_cli.py index 2e63669..b925c96 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -142,6 +142,18 @@ def test_phrases_file_falls_back_to_cp1251(tmp_path: Path): assert resolve_phrases((), phrases_file) == ["чай", "кофе"] +def test_phrases_file_with_a_utf8_bom_does_not_leak_into_the_first_phrase(tmp_path: Path): + # A BOM'd UTF-8 file decodes successfully under plain "utf-8" with the + # BOM character left attached to the first line; ''.isspace() is + # False, so neither resolve_phrases' line.strip() nor the collector's + # own phrase.strip() removes it. utf-8-sig must be tried before plain + # utf-8 so the BOM is stripped during decoding itself. + phrases_file = tmp_path / "phrases.txt" + phrases_file.write_bytes("ремонт квартир\nдизайн интерьера\n".encode("utf-8-sig")) + + assert resolve_phrases((), phrases_file) == ["ремонт квартир", "дизайн интерьера"] + + def test_phrases_file_with_undecodable_bytes_is_reported_without_a_traceback(tmp_path: Path): # 0x98 is invalid in both utf-8 (a lone continuation byte) and cp1251 # (unassigned in that codepage) — one of the few byte values neither From 6468e25c226276ce817ee67bdd74211a690ae518 Mon Sep 17 00:00:00 2001 From: axisrow Date: Fri, 21 Aug 2026 15:22:44 +0700 Subject: [PATCH 15/15] fix: guard single-row containment and fix the monthly calendar path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cycle-review round 2 findings against dc9b16b (round 1's fixes). Codex found the round-1 containment check (2a9689e) sat behind the same len(rows) < 2 early return as the pairwise contiguity loop, but containment only needs dates[0]/dates[-1] -- the same row, for a single-row export -- not a pair. A genuine 1-row daily/weekly export whose single date lay entirely outside the requested window passed _is_untrustworthy_empty_export (which only rejects 0 rows) and was written to Parquet and marked complete, unguarded. Fixed by hoisting the containment check above the row-count early return; the pairwise contiguity loop stays gated at >= 2 rows, since it genuinely needs a pair. Monthly stays exempt from both checks -- its period field is a nominative month name, not a parseable date, so validating its end boundary here would need a separate parser this PR does not add. /review flagged the monthly calendar path (.datepicker-month__years_select / .react-datepicker__month-text) as never live-confirmed, unlike daily/weekly. Live CDP verification against the authenticated Wordstat UI found two real bugs in that path, both now fixed: - month_label.capitalize() ("Январь") never matched the popup's actual lowercase nominative text ("январь"), failing every explicit --granularity monthly --date-from/--date-to request with InterfaceChangedError deep inside the click. - the monthly popup is a single shared range-picker, not two independent day/week-style popups: the intermediate DATE_RANGE_SELECTOR click between picking date_from and date_to closed it instead of reopening it. Skipping that click for the monthly path (day/week keep it -- their two-click flow is live-confirmed working) lets the still-open popup accept date_to directly. Live-verified end to end after both fixes: an explicit --granularity monthly --date-from 2024-01-01 --date-to 2024-06-30 request now collects 6 rows (январь-июнь 2024) with actual_period exactly matching requested_period. /review also re-raised (as RR1) the same weekly-CSV-format claim round 1 correctly SKIPped as R2 -- it recurred only because the doc that disproves it (docs/ISSUE6_PHASE1_RESEARCH.md, PR #20) isn't on this branch. Added an inline comment at the parse site citing that doc's confirmed byte-for-byte weekly CSV format, so a future review pass can check the comment instead of re-raising the same false positive against code alone. Tests: uv run pytest -q (151 passed), uv run ruff check (clean). --- src/wordstat/collector.py | 89 ++++++++++++++++++++++---- tests/test_collector_view.py | 115 ++++++++++++++++++++++++++++++++++ tests/test_manifest_period.py | 33 ++++++++++ 3 files changed, 226 insertions(+), 11 deletions(-) diff --git a/src/wordstat/collector.py b/src/wordstat/collector.py index 69996b3..d34bfe6 100644 --- a/src/wordstat/collector.py +++ b/src/wordstat/collector.py @@ -546,7 +546,34 @@ def _assert_contiguous_dynamics_rows( date_from: date | None = None, date_to: date | None = None, ) -> None: - if granularity is Granularity.MONTHLY or len(dataset.rows) < 2: + # This parses dataset.rows[field] — the exported CSV's date column + # — with strptime("%d.%m.%Y"), NOT the DOM's displayed cell text. + # Live-measured weekly CSV (issue #6 phase 1 document — see + # `docs/ISSUE6_PHASE1_RESEARCH.md` on the PR #20 branch, + # `git show 4be8996:docs/ISSUE6_PHASE1_RESEARCH.md`): the exported + # field is named "Неделя с" and holds only the week's start date + # as a bare DD.MM.YYYY value ("22.06.2026") — no range, no week + # number — confirmed byte-for-byte from a real downloaded CSV. The + # DOM cell (_wait_for_table_granularity's WEEKLY pattern below) + # shows a genitive-month date *range* instead ("22 июня 2026 – 28 + # июня 2026"); that is a different string in a different place and + # does not describe the exported column. Do not "fix" this parse + # to expect a range based on the DOM pattern — that would break a + # confirmed-working live path (cycle-review round 1's R2 and round + # 2's RR1 both raised this same claim from reading the code + # without this doc open; both were false positives). + # + # Monthly's period field is a nominative Russian month name + # ("август 2024", confirmed live — issue #6 phase 1 document), not + # a strptime("%d.%m.%Y")-parseable date the way daily/weekly are, + # so neither contiguity nor containment can be validated here for + # it; _wait_for_period_applied still gates monthly's start + # boundary before download (same mechanism as daily/weekly), and + # _actual_period records the raw field/from/to strings into the + # manifest for every dynamics export including monthly, so actual + # coverage stays observable post-hoc even without this check + # (cycle-review round 2, C2 triage). + if granularity is Granularity.MONTHLY or not dataset.rows: return field = dataset.headers[0] try: @@ -555,13 +582,14 @@ def _assert_contiguous_dynamics_rows( raise InterfaceChangedError( f"Wordstat returned an unexpected {granularity.value} dynamics date format" ) from error - step = timedelta(days=1 if granularity is Granularity.DAILY else 7) - for previous, current in zip(dates, dates[1:]): - if current - previous != step: - raise InterfaceChangedError( - f"Wordstat returned a gap in the {granularity.value} dynamics series " - f"between {previous:%d.%m.%Y} and {current:%d.%m.%Y}" - ) + if len(dates) >= 2: + step = timedelta(days=1 if granularity is Granularity.DAILY else 7) + for previous, current in zip(dates, dates[1:]): + if current - previous != step: + raise InterfaceChangedError( + f"Wordstat returned a gap in the {granularity.value} dynamics series " + f"between {previous:%d.%m.%Y} and {current:%d.%m.%Y}" + ) # Containment, not equality: a shorter export fully inside the # requested window is legitimate (the daily series' variable # trailing tail, confirmed live — issue #6 phase 1 document), so @@ -573,7 +601,14 @@ def _assert_contiguous_dynamics_rows( # downloading; it has no equivalent check for date_to, so a table # that has not yet repainted past a previous, stale window can # download and be recorded as complete with data that never - # entered the requested window at all. Only applies when an + # entered the requested window at all. Deliberately runs even for + # a single-row export (needs only dates[0]/dates[-1], which are + # the same row — unlike the pairwise contiguity loop above, which + # needs at least two rows): a single stale/truncated row is not + # exempt just because it is too short to check contiguity + # (cycle-review round 2, C2 triage — the original version gated + # this behind the same `len(rows) < 2` early return as contiguity + # and let a 1-row export bypass it entirely). Only applies when an # explicit period was requested (date_from/date_to are None for # the UI's default window, which this function cannot validate # against any specific boundary). @@ -629,7 +664,28 @@ async def _set_period(self, page, granularity: Granularity, date_from: date, dat ) await self._click(page, DATE_RANGE_SELECTOR) await self._select_calendar_date(page, popup_type, date_from) - await self._click(page, DATE_RANGE_SELECTOR) + if popup_type != "month": + # day/week: independent popups per click, live-confirmed by + # 255bca7's daily/weekly runs going through this exact + # two-click path successfully. + await self._click(page, DATE_RANGE_SELECTOR) + else: + # month: a single shared range-picker, not two independent + # popups (live CDP check, issue #6 phase 2, cycle-review + # round 2). Right after date_from is picked, every month + # element already carries the "in-selecting-range" class — + # the picker is already in range-selection mode — and the + # date-range button's text does not update to reflect + # date_from yet; it only updates once date_to is picked in + # the SAME still-open popup. A second DATE_RANGE_SELECTOR + # click here closes this popup instead of reopening it + # (confirmed live: visibility flips to 'hidden'), so the + # follow-up _select_calendar_date for date_to would time out + # waiting for a popup that never reappears. Confirmed live + # that skipping the intermediate click and picking date_to + # directly in the still-open popup produces the correct + # button text "Январь 2024 — Июнь 2024". + pass await self._select_calendar_date(page, popup_type, date_to) # _wait_for_table_granularity alone is not enough here: it only # checks that the first cell's *format* matches the granularity @@ -718,7 +774,18 @@ async def _select_calendar_date(self, page, popup_type: str, target: date) -> No await self._click_visible_text(page, ".Popup2_visible [role='option']", month_label) day_selector = f"{root} button[name='day']" if popup_type == "month": - await self._click_visible_text(page, month_selector, month_label.capitalize()) + # Live CDP check (issue #6 phase 2, cycle-review round 2): the + # month-text popup renders lowercase nominative month names + # ("январь"), matching RUSSIAN_MONTHS verbatim. + # month_label.capitalize() ("Январь") never matches any of + # them — every explicit monthly request with + # --date-from/--date-to failed here with InterfaceChangedError + # ("Wordstat option 'Январь' was not uniquely found") despite + # validate_period accepting the request and the browser + # already being driven. Do not re-add .capitalize(): it was + # never confirmed live and is contradicted by the confirmed + # live DOM text. + await self._click_visible_text(page, month_selector, month_label) else: await self._click_visible_date_button(page, day_selector, str(target.day)) diff --git a/tests/test_collector_view.py b/tests/test_collector_view.py index 807ccae..e5dd4a4 100644 --- a/tests/test_collector_view.py +++ b/tests/test_collector_view.py @@ -261,3 +261,118 @@ def test_wait_for_period_applied_monthly_matches_nominative_cell_text(monkeypatc pattern = _extract_regex(captured["expression"]) assert pattern.search("август 2024") assert not pattern.search("сентябрь 2024") + + +def test_set_period_does_not_reopen_the_monthly_popup_between_dates(monkeypatch, tmp_path): + # Live CDP check (issue #6 phase 2, cycle-review round 2): the monthly + # calendar is a single range-picker, not two independent popups like + # day/week. Live DOM evidence: right after date_from is picked, every + # month element already carries the "in-selecting-range" class (the + # picker is already in range-selection mode), and the date-range + # button's text does not update to reflect date_from yet -- it only + # updates once date_to is picked in the SAME still-open popup. The + # intermediate DATE_RANGE_SELECTOR click that day/week rely on to + # reopen their popup instead closes this one (confirmed live: + # visibility flips to 'hidden'), so the follow-up _select_calendar_date + # call for date_to times out waiting for a popup that never reappears. + # Live-confirmed fix: for monthly, skip the intermediate click and let + # the second _select_calendar_date pick date_to in the still-open + # popup -- confirmed live to produce the correct button text + # "Январь 2024 — Июнь 2024". + calls = [] + + async def click(self, page, selector): + calls.append(("click", selector)) + + async def select_calendar_date(self, page, popup_type, target): + calls.append(("select_calendar_date", popup_type, target)) + + async def wait_for_period_applied(self, page, granularity, date_from): + calls.append(("wait_for_period_applied", granularity, date_from)) + + monkeypatch.setattr(WordstatCollector, "_click", click) + monkeypatch.setattr(WordstatCollector, "_select_calendar_date", select_calendar_date) + monkeypatch.setattr(WordstatCollector, "_wait_for_period_applied", wait_for_period_applied) + + collector = WordstatCollector("cdp", tmp_path) + asyncio.run( + collector._set_period(object(), Granularity.MONTHLY, date(2024, 1, 1), date(2024, 6, 30)) + ) + + click_count = sum(1 for call in calls if call[0] == "click") + assert click_count == 1, f"expected exactly one popup-open click for monthly, got {calls}" + select_calls = [call for call in calls if call[0] == "select_calendar_date"] + assert select_calls == [ + ("select_calendar_date", "month", date(2024, 1, 1)), + ("select_calendar_date", "month", date(2024, 6, 30)), + ] + + +def test_set_period_still_reopens_the_popup_between_dates_for_day_and_week(monkeypatch, tmp_path): + # day/week use independent popups per click (confirmed live in round 1: + # 255bca7's live weekly/daily runs both went through this two-click + # path successfully) -- only monthly's shared range-picker changes. + calls = [] + + async def click(self, page, selector): + calls.append("click") + + async def select_calendar_date(self, page, popup_type, target): + pass + + async def wait_for_period_applied(self, page, granularity, date_from): + pass + + monkeypatch.setattr(WordstatCollector, "_click", click) + monkeypatch.setattr(WordstatCollector, "_select_calendar_date", select_calendar_date) + monkeypatch.setattr(WordstatCollector, "_wait_for_period_applied", wait_for_period_applied) + + collector = WordstatCollector("cdp", tmp_path) + asyncio.run( + collector._set_period(object(), Granularity.DAILY, date(2026, 7, 1), date(2026, 8, 20)) + ) + assert len(calls) == 2 + + calls.clear() + asyncio.run( + collector._set_period(object(), Granularity.WEEKLY, date(2026, 7, 1), date(2026, 8, 20)) + ) + assert len(calls) == 2 + + +def test_select_calendar_date_clicks_the_month_popup_text_as_the_dom_renders_it(monkeypatch, tmp_path): + # Live CDP check (issue #6 phase 2, cycle-review round 2): the monthly + # calendar's month-text popup (.react-datepicker__month-text) renders + # lowercase nominative month names ("январь"), matching RUSSIAN_MONTHS + # verbatim -- confirmed by reading the popup's actual elements live. + # month_label.capitalize() ("Январь") never matches any of them, so + # every explicit --granularity monthly --date-from/--date-to request + # failed with InterfaceChangedError: "Wordstat option 'Январь' was not + # uniquely found" deep inside this click, despite validate_period + # accepting the request and the browser already being driven. + clicked_texts = [] + + async def click(self, page, selector): + pass + + async def wait(self, page, expression, seconds=None, required=True): + pass + + async def click_visible_text(self, page, selector, text): + clicked_texts.append((selector, text)) + + async def click_visible_date_button(self, page, selector, text): + clicked_texts.append((selector, text)) + + monkeypatch.setattr(WordstatCollector, "_click", click) + monkeypatch.setattr(WordstatCollector, "_wait_for", wait) + monkeypatch.setattr(WordstatCollector, "_click_visible_text", click_visible_text) + monkeypatch.setattr(WordstatCollector, "_click_visible_date_button", click_visible_date_button) + + collector = WordstatCollector("cdp", tmp_path) + asyncio.run(collector._select_calendar_date(object(), "month", date(2024, 1, 1))) + + month_click = next( + text for selector, text in clicked_texts if "month" in selector or "react-datepicker" in selector + ) + assert month_click == "январь" diff --git a/tests/test_manifest_period.py b/tests/test_manifest_period.py index a6abc04..e89c1d2 100644 --- a/tests/test_manifest_period.py +++ b/tests/test_manifest_period.py @@ -81,6 +81,39 @@ def test_dynamics_series_rejects_a_stale_window_that_predates_the_request(): ) +def test_single_row_export_is_not_exempt_from_the_containment_check(): + # Cycle-review round 2, C2: the containment check originally sat behind + # the same `len(rows) < 2` early return as the pairwise contiguity loop + # -- but containment only needs dates[0]/dates[-1] (the same row, for a + # 1-row export), not a pair. A genuine 1-row daily/weekly export whose + # single date lies entirely outside the requested window passed + # _is_untrustworthy_empty_export (which only rejects 0 rows) and was + # then written to Parquet and marked complete, unguarded. + stale_single_row = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Дата"], + rows=[{"Дата": "20.06.2026"}], + ) + with pytest.raises(InterfaceChangedError, match="requested window"): + WordstatCollector._assert_contiguous_dynamics_rows( + stale_single_row, Granularity.DAILY, date_from=date(2026, 6, 22), date_to=date(2026, 8, 20) + ) + # A single row genuinely inside the requested window remains legitimate + # (e.g. a one-day window, or the daily series' documented variable + # trailing tail collapsed to one row) and must not be rejected. + in_window_single_row = CsvDataset( + view=WordstatView.DYNAMICS, + headers=["Дата"], + rows=[{"Дата": "22.06.2026"}], + ) + WordstatCollector._assert_contiguous_dynamics_rows( + in_window_single_row, Granularity.DAILY, date_from=date(2026, 6, 22), date_to=date(2026, 8, 20) + ) + # No explicit period requested (the UI's default window) still skips + # the check entirely -- nothing to validate a single row against. + WordstatCollector._assert_contiguous_dynamics_rows(stale_single_row, Granularity.DAILY) + + def test_weekly_containment_uses_the_aligned_week_start_not_the_raw_request(): # Wordstat snaps a weekly window's first row to the Monday of # date_from's week (confirmed live: 2018-12-26, a Wednesday, produced