diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 2ad2702..e509959 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -68,6 +68,20 @@ jobs: - name: Модули ведут к своим ноутбукам run: python scripts/check_modules.py + # Курс девять раз называет себя «двадцать семь модулей в семи частях» -- + # цифрами, русскими словами и английскими. Ни одно из этих мест не знает, + # сколько модулей на самом деле, и разойтись они могут молча. Считается + # здесь один раз, по каталогам и таблицам программы. + - name: Курс говорит о себе правду + run: python scripts/check_counts.py + + # Теги Open Graph отдаёт шаблон, а не страница: сборка проходит и с + # пустым заголовком (при 404 `page` пуста), и с картинкой, которой уже + # нет, и с заголовком, выведенным дважды. Проверяется по собранному + # каталогу, поэтому шаг идёт после mkdocs build. + - name: У каждой страницы есть что показать, когда ей делятся + run: python scripts/check_social.py site + ci: name: CI runs-on: ubuntu-latest diff --git a/.zenodo.json b/.zenodo.json index 3144ab5..e37267d 100644 --- a/.zenodo.json +++ b/.zenodo.json @@ -1,7 +1,7 @@ { "upload_type": "lesson", "title": "lemma — a free course in machine learning, deep learning and reinforcement learning", - "description": "A complete roadmap through machine learning, neural networks, reinforcement learning and recommender systems: twenty-seven modules in seven parts, from the arithmetic of a mean to reproducing a recent paper. The central skill taught is checking a claim rather than launching a training run, so the first module covers baselines and confidence intervals before any machine learning at all. Written in Russian; the notebooks run on a CPU in seconds and are executed in CI on Linux and Windows.", + "description": "A complete roadmap through machine learning, neural networks, reinforcement learning and recommender systems: twenty-seven modules in seven parts, from the arithmetic of a mean to reproducing a recent paper. The central skill taught is checking a claim rather than launching a training run, so the first module covers baselines and confidence intervals before any machine learning at all. Written in Russian and English, both versions complete; the notebooks run on a CPU in seconds and are executed in CI on Linux and Windows.", "creators": [ { "name": "Drobyshev, Denis", @@ -19,6 +19,7 @@ "reproducibility", "research-methods", "russian", + "english", "jupyter-notebook" ], "access_right": "open", diff --git a/CITATION.cff b/CITATION.cff index ace3846..419cb2d 100644 --- a/CITATION.cff +++ b/CITATION.cff @@ -10,8 +10,8 @@ abstract: >- the arithmetic of a mean to reproducing a recent paper. The central skill taught is checking a claim rather than launching a training run, so the first module covers baselines and confidence intervals before any machine learning - at all. Written in Russian; the notebooks run on a CPU in seconds and are - executed in CI on Linux and Windows. + at all. Written in Russian and English, both versions complete; the notebooks + run on a CPU in seconds and are executed in CI on Linux and Windows. authors: - family-names: Drobyshev given-names: Denis @@ -28,3 +28,4 @@ keywords: - reproducibility - research-methods - russian + - english diff --git a/README.en.md b/README.en.md new file mode 100644 index 0000000..82eee5b --- /dev/null +++ b/README.en.md @@ -0,0 +1,112 @@ +# lemma + +[Русский](README.md) · **English** · [Read the course](https://drobyshevdev.github.io/lemma/en/) + +**A free course in machine learning, neural networks, reinforcement learning and +recommender systems — from nothing to reading and reproducing research.** + +[![Site](https://img.shields.io/badge/site-drobyshevdev.github.io/lemma-4f46e5)](https://drobyshevdev.github.io/lemma/en/) +[![CI](https://github.com/DrobyshevDev/lemma/actions/workflows/ci.yml/badge.svg)](https://github.com/DrobyshevDev/lemma/actions/workflows/ci.yml) +[![Modules](https://img.shields.io/badge/modules-27-8A2BE2.svg)](https://drobyshevdev.github.io/lemma/en/programme/) +[![Text: CC BY 4.0](https://img.shields.io/badge/text-CC%20BY%204.0-lightgrey.svg)](LICENSE-CONTENT) +[![Code: MIT](https://img.shields.io/badge/code-MIT-green.svg)](LICENSE) + +**[Read the course →](https://drobyshevdev.github.io/lemma/en/)** + +## What this is + +A roadmap through ML, DL and RL: what to learn, in what order, why, and how to tell that +you have actually learned it. Twenty-seven modules in seven parts, from the arithmetic of +a mean and a variance to reproducing a recent paper as a capstone. + +Free and open, in full. No sign-up, no "first module free", no 24-month instalment plan. + +A lemma is a statement proved not for its own sake but to prove the next one. The course +is built the same way: no module stands on its own, each one holds up what comes after. + +Both languages are complete — twenty-seven modules in Russian and twenty-seven in English, +the same notebook behind each. + +## What is different about it + +Most courses teach you to train models. This one teaches you to **check claims**. + +The field moves through papers, and the overwhelming majority of the improvements claimed +in them do not reproduce, dissolve under an honest comparison, or come from comparing a +tuned method against an untuned baseline. Someone who can train a model but cannot check +a claim cannot tell progress from noise — and builds on noise. + +So a module does not end with "we got accuracy 0.93" but with "we checked that the +improvement survives a change of random seed and a comparison against an honest baseline". +Module 1 is about exactly that, before any machine learning at all. + +The second difference is the tie to psychology where the tie is real: reinforcement +learning and behavioural psychology describe the same thing twice over, and recommender +systems are applied psychology of attention. + +The third is that the course is written by the people who build the tools it uses. Where +run tracking is needed, that is `mlango`; where an agent loop with a readable trace is +needed, `glia`; where a trained policy has to be compared against a classical baseline, +`decisionrl`, which ships that baseline with every problem. None of the libraries is +required: everywhere the course also shows how to do the same thing by hand. + +## Layout + +``` +docs/ + index.en.md landing page (laid out in overrides/home.en.html) + programme.en.md all 27 modules + capstone.en.md the capstone: choosing a paper, reproducing it, writing it up + prerequisites.en.md what to know before starting + how-to-study.en.md how to study so that it works + modules/ modules, .md in Russian and .en.md in English + assets/theme.css a dark editorial theme over mkdocs-material +overrides/home.en.html the landing template +notebooks/ notebooks, one per module, shared by both languages +``` + +Notebooks run top to bottom with no edits and **on a CPU in reasonable time**. Where a full +run gives a different number, that is said outright. CI executes every notebook on Linux +and Windows: a reader whose notebook does not run does not have the course. + +## Running it + +From any module page, through the **"open in Colab"** link next to the notebook. There is +nothing to install: the course code imports only NumPy and Matplotlib, and Colab already +has both. A reader who cares about one module should not have to build an environment for it. + +Locally, if you would rather keep everything yourself: + +```bash +pip install -r requirements.txt +mkdocs serve # → http://127.0.0.1:8000 +jupyter lab notebooks/ +``` + +## State + +All twenty-seven modules are done — both language versions, a notebook for each, CI green. +The course can be taken end to end, from the first module to the capstone. + +| Part | Modules | State | +|---|---|---| +| I. How claims are checked | 1–4 | **complete** | +| II. Classical ML | 5–7 | **complete** | +| III. Neural networks | 8–11 | **complete** | +| IV. RL and psychology | 12–16 | **complete** | +| V. Recommender systems | 17–20 | **complete** | +| VI. Agents | 21–23 | **complete** | +| VII. Out to the frontier | 24–27 | **complete** | + +## Helping + +The most useful feedback is **"I got stuck here"**. If an explanation did not work, that is +a defect in the text, not in the reader, and it needs to be known about. Open an +[issue](https://github.com/DrobyshevDev/lemma/issues) naming the module and the place. + +Contribution rules — the [organisation's CONTRIBUTING.md](https://github.com/DrobyshevDev/.github/blob/master/CONTRIBUTING.md). + +## Licences + +Text and illustrations — [CC BY 4.0](LICENSE-CONTENT): take them, translate them, use them +in your own teaching, give credit. Code in the notebooks and scripts — [MIT](LICENSE). diff --git a/README.md b/README.md index 81b5102..28edcce 100644 --- a/README.md +++ b/README.md @@ -1,5 +1,7 @@ # lemma +**Русский** · [English](README.en.md) · [Читать курс](https://drobyshevdev.github.io/lemma/) + **Бесплатный курс по машинному обучению, нейросетям, обучению с подкреплением и рекомендательным системам — с нуля и до умения читать и воспроизводить исследования.** @@ -24,6 +26,9 @@ следующее. Курс устроен так же: ни один модуль не стоит отдельно, каждый — опора для того, что идёт после. +Обе языковые версии полные — двадцать семь модулей по-русски и двадцать семь по-английски, +за каждым один и тот же ноутбук. + ## Чем отличается Большинство курсов учат обучать модели. Этот учит **проверять утверждения**. diff --git a/docs/assets/og.png b/docs/assets/og.png new file mode 100644 index 0000000..f4ef4c9 Binary files /dev/null and b/docs/assets/og.png differ diff --git a/docs/programme.en.md b/docs/programme.en.md index 821386c..e5477b3 100644 --- a/docs/programme.en.md +++ b/docs/programme.en.md @@ -97,5 +97,5 @@ How to choose it, do it and write it up — a full walkthrough with a checklist ## In total **About 50 weeks** at ten hours a week — a year at an unhurried pace, or half a year at a -dense one. Parts I–II (8 weeks) make sense on their own: even if you stop there, you will read +dense one. Parts I–II (12 weeks) make sense on their own: even if you stop there, you will read papers more carefully than most of the people who write them. diff --git a/docs/programme.md b/docs/programme.md index 5e55c02..5a4b6ae 100644 --- a/docs/programme.md +++ b/docs/programme.md @@ -98,5 +98,5 @@ ## Итого **Около 50 недель** при десяти часах в неделю — год неспешно или полгода плотно. Части -I–II (8 недель) имеют смысл сами по себе: даже если вы остановитесь на них, вы будете +I–II (12 недель) имеют смысл сами по себе: даже если вы остановитесь на них, вы будете читать статьи внимательнее, чем большинство тех, кто их пишет. diff --git a/overrides/home.en.html b/overrides/home.en.html index 90b779e..c45ea46 100644 --- a/overrides/home.en.html +++ b/overrides/home.en.html @@ -9,11 +9,20 @@ {% block extrahead %} {{ super() }} +{% endblock %} + +{# Replaces the generic tags from main.html rather than adding to them. #} +{% block social_meta %} + - + + + + + {% endblock %} {% block tabs %} diff --git a/overrides/home.html b/overrides/home.html index cf6ee6f..f301ef9 100644 --- a/overrides/home.html +++ b/overrides/home.html @@ -11,11 +11,22 @@ {# The content block below is empty on purpose; without this the theme still reserves a padded, empty article between the landing and the footer. #} +{% endblock %} + +{# Replaces the generic tags from main.html rather than adding to them: this + page's description is written for a reader deciding whether to start, which + a generic one cannot be. #} +{% block social_meta %} + - + + + + + {% endblock %} {% block tabs %} diff --git a/overrides/main.html b/overrides/main.html new file mode 100644 index 0000000..4158d33 --- /dev/null +++ b/overrides/main.html @@ -0,0 +1,40 @@ +{% extends "base.html" %} + +{# + Open Graph on every page, not only on the two landings. + + The theme emits none, and the landings carried hand-written tags, so all + fifty-four module pages shared as a bare link: no title, no description, no + image. The shareable unit of a course is a module -- "вот модуль про дофамин + и TD-обучение" -- and that was the one thing with nothing to show. + + `social_meta` is its own block so a landing can replace these rather than + append to them: appending would emit og:title twice and leave the scraper to + pick. +#} + +{% block extrahead %} + {{ super() }} + + {% block social_meta %} + {#- `page` is None while the theme renders the 404, so every access here is + guarded. Without that the build dies with "'None' has no attribute + 'meta'" and mkdocs does not say which template. -#} + {%- set page_title = (page.title if page else None) | default(config.site_name, true) -%} + {%- set page_description = (page.meta.description if page and page.meta else None) + | default(config.site_description, true) -%} + {%- set page_url = (page.canonical_url if page else config.site_url) -%} + + + + + + + {# summary_large_image, not summary: the card is 1280×640 and a small + thumbnail wastes it. #} + + + + + {% endblock %} +{% endblock %} diff --git a/scripts/check_counts.py b/scripts/check_counts.py new file mode 100644 index 0000000..41f92c3 --- /dev/null +++ b/scripts/check_counts.py @@ -0,0 +1,252 @@ +#!/usr/bin/env python3 +"""Числа, которые курс говорит о себе, сходятся с тем, из чего он состоит. + +Курс называет себя «двадцать семь модулей в семи частях» в обоих README, на +двух лендингах, в двух программах, в двух бейджах, в CITATION.cff и в +.zenodo.json — цифрами, русскими словами и английскими. Ни одно из этих мест не +знает, сколько модулей на самом деле. Добавить модуль — значит попасть в +каждое из них и ни разу не ошибиться, а первая же неправка живёт до тех пор, +пока кто-нибудь не пересчитает руками. + +Сколько мест проверено, скрипт говорит сам: число утверждений — тоже число, +которое ему незачем знать наизусть. + +Здесь считается ровно один раз — по `docs/modules/`, `notebooks/` и таблицам +программы, — а дальше каждое утверждение сверяется с этим счётом. + +Отдельно сверяются недели: программа обещает «около 50 недель» и «части I–II — +столько-то», и оба числа есть сумма колонки «Время» в её же таблицах. Такую +сумму никто не пересчитывает, правя одну строку. + +Пропавшее утверждение — тоже расхождение. Если фразу перепишут так, что шаблон +перестанет находиться, проверка перестанет проверять и промолчит об этом; +поэтому ненайденное место сообщается наравне с разошедшимся числом. + +Только стандартная библиотека. + + python scripts/check_counts.py +""" + +from __future__ import annotations + +import json +import pathlib +import re +import sys + +ROOT = pathlib.Path(__file__).resolve().parent.parent + +# Слова до сорока: курс из сорока модулей — уже другой курс, и падение проверки +# на сорок первом правильно, а не недосмотр. +RU_ONES = ["", "один", "два", "три", "четыре", "пять", "шесть", "семь", "восемь", "девять", + "десять", "одиннадцать", "двенадцать", "тринадцать", "четырнадцать", "пятнадцать", + "шестнадцать", "семнадцать", "восемнадцать", "девятнадцать"] +RU_TENS = {20: "двадцать", 30: "тридцать", 40: "сорок"} +EN_ONES = ["", "one", "two", "three", "four", "five", "six", "seven", "eight", "nine", + "ten", "eleven", "twelve", "thirteen", "fourteen", "fifteen", + "sixteen", "seventeen", "eighteen", "nineteen"] +EN_TENS = {20: "twenty", 30: "thirty", 40: "forty"} + + +def ru_word(n: int) -> str: + if n < 20: + return RU_ONES[n] + tens, ones = divmod(n, 10) + return (RU_TENS[tens * 10] + (f" {RU_ONES[ones]}" if ones else "")).strip() + + +def en_word(n: int) -> str: + if n < 20: + return EN_ONES[n] + tens, ones = divmod(n, 10) + return EN_TENS[tens * 10] + (f"-{EN_ONES[ones]}" if ones else "") + + +#: Предложный падеж тех же числительных: «в семи частях», не «в семь частях». +#: Правило «мягкий знак на -и» ловит пять…двадцать и тридцать, но не первые +#: четыре, не восемь (восьми, а не «восеми») и не сорок — они названы отдельно. +RU_PREP_IRREGULAR = {1: "одном", 2: "двух", 3: "трёх", 4: "четырёх", 8: "восьми", + 40: "сорока"} + + +def ru_word_prep(n: int) -> str: + if n in RU_PREP_IRREGULAR: + return RU_PREP_IRREGULAR[n] + if n < 20 or n in RU_TENS: + return ru_word(n).removesuffix("ь") + "и" + tens, ones = divmod(n, 10) + return f"{ru_word_prep(tens * 10)} {ru_word_prep(ones)}" + + +def ru_plural(n: int, one: str, few: str, many: str) -> str: + """модуль / модуля / модулей — по последним цифрам, как в русском.""" + if 11 <= n % 100 <= 14: + return many + last = n % 10 + return one if last == 1 else few if 2 <= last <= 4 else many + + +def read(name: str) -> str: + return (ROOT / name).read_text(encoding="utf-8") + + +def parse_programme(name: str) -> tuple[list[tuple[str, list[int], float]], float]: + """Части программы: римская цифра, номера модулей, сумма недель.""" + parts: list[tuple[str, list[int], float]] = [] + inside = False + for line in read(f"docs/{name}").splitlines(): + head = re.match(r"^## (?:Часть|Part) ([IVX]+)\.", line) + if head: + parts.append((head.group(1), [], 0.0)) + inside = True + continue + if line.startswith("## "): + inside = False + row = re.match( + r"^\| (\d+) \| \[[^\]]+\]\(modules/[\w.-]+\.md\).*\| ([^|]+) \|\s*$", line + ) + if row and inside: + weeks = re.search(r"([\d.]+)", row.group(2)) + name_, numbers, total = parts[-1] + numbers.append(int(row.group(1))) + parts[-1] = (name_, numbers, total + (float(weeks.group(1)) if weeks else 0.0)) + return parts, sum(p[2] for p in parts) + + +def main() -> int: + problems: list[str] = [] + + checked: list[str] = [] + + def want(where: str, pattern: str, what: str) -> None: + """Сверить одно утверждение. Не найдено — тоже расхождение.""" + checked.append(where) + if re.search(pattern, read(where), re.M) is None: + problems.append(f"{where}: ждали {what} — не сходится или фразу переписали") + + # ---- то, из чего курс состоит ------------------------------------------- + ru_pages = sorted(p for p in (ROOT / "docs" / "modules").glob("*.md") + if not p.name.endswith(".en.md")) + en_pages = sorted((ROOT / "docs" / "modules").glob("*.en.md")) + notebooks = sorted((ROOT / "notebooks").glob("*.ipynb")) + n = len(ru_pages) + if not n: + print(" страниц модулей нет", file=sys.stderr) + return 2 + + if len(en_pages) != n: + problems.append(f"docs/modules: {n} страниц по-русски и {len(en_pages)} по-английски") + if len(notebooks) != n: + problems.append(f"notebooks: {len(notebooks)} ноутбуков на {n} модулей") + + ru_parts, ru_weeks = parse_programme("programme.md") + en_parts, en_weeks = parse_programme("programme.en.md") + parts = len(ru_parts) + numbered = [i for p in ru_parts for i in p[1]] + + if numbered != list(range(1, n + 1)): + problems.append( + f"docs/programme.md: модули пронумерованы {numbered}, а страниц {n} — " + "номер пропущен или задвоен" + ) + if [p[1] for p in en_parts] != [p[1] for p in ru_parts]: + problems.append("docs/programme.en.md: разбивка по частям не та же, что в русской") + if ru_weeks != en_weeks: + problems.append(f"недели расходятся между языками: {ru_weeks:g} и {en_weeks:g}") + + n_ru, n_en = ru_word(n), en_word(n) + p_ru, p_en = ru_word(parts), en_word(parts) + p_ru_prep = ru_word_prep(parts) + modules_ru = ru_plural(n, "модуль", "модуля", "модулей") + parts_ru = ru_plural(parts, "часть", "части", "частей") + parts_ru_prep = ru_plural(parts, "части", "частях", "частях") + + # ---- то, что курс о себе говорит ---------------------------------------- + want("README.md", rf"img\.shields\.io/badge/модулей-{n}-", f"бейдж модулей-{n}") + want("README.md", rf"(?i)все {n} {modules_ru}", f"«все {n} {modules_ru}»") + want("README.md", rf"(?i){n_ru} {modules_ru} в {p_ru_prep} {parts_ru_prep}", + f"«{n_ru} {modules_ru} в {p_ru_prep} {parts_ru_prep}»") + want("README.md", rf"(?i)все {n_ru} {modules_ru} готовы", f"«все {n_ru} {modules_ru} готовы»") + + want("README.en.md", rf"img\.shields\.io/badge/modules-{n}-", f"бейдж modules-{n}") + want("README.en.md", rf"(?i)all {n} modules", f"«all {n} modules»") + want("README.en.md", rf"(?i){n_en} modules in {p_en} parts", + f"«{n_en} modules in {p_en} parts»") + want("README.en.md", rf"(?i)all {n_en} modules are done", f"«all {n_en} modules are done»") + + want("docs/programme.md", rf"(?i)^{n_ru} {modules_ru} в {p_ru_prep} {parts_ru_prep}\.", + f"«{n_ru} {modules_ru} в {p_ru_prep} {parts_ru_prep}.»") + want("docs/programme.en.md", rf"(?i)^{n_en} modules in {p_en} parts\.", + f"«{n_en} modules in {p_en} parts.»") + + want("overrides/home.html", rf"{n} {modules_ru}", f"«{n} {modules_ru}» в плашке") + want("overrides/home.html", rf"(?i)

{p_ru} {parts_ru}, {n_ru} {modules_ru}

", + f"заголовок «{p_ru} {parts_ru}, {n_ru} {modules_ru}»") + want("overrides/home.en.html", rf"{n} modules", f"«{n} modules» в плашке") + want("overrides/home.en.html", rf"(?i)

{p_en} parts, {n_en} modules

", + f"заголовок «{p_en} parts, {n_en} modules»") + + want("CITATION.cff", rf"(?i){n_en} modules in {p_en} parts", + f"«{n_en} modules in {p_en} parts»") + want(".zenodo.json", rf"(?i){n_en} modules in {p_en} parts", + f"«{n_en} modules in {p_en} parts»") + + nav = len(re.findall(r"modules/[\w.-]+\.md", read("mkdocs.yml"))) + if nav != n: + problems.append(f"mkdocs.yml: в навигации {nav} модулей, а страниц {n}") + + # ---- недели ------------------------------------------------------------- + # «Около 50» — округление, поэтому допуск; части I–II названы точно, + # поэтому точное сравнение. + for name, source, about, exact in ( + ("docs/programme.md", ru_parts, r"\*\*Около (\d+) недель\*\*", r"I–II \((\d+) недель\)"), + ("docs/programme.en.md", en_parts, r"\*\*About (\d+) weeks\*\*", r"I–II \((\d+) weeks\)"), + ): + text = read(name) + total = sum(p[2] for p in source) + head = sum(p[2] for p in source[:2]) + + found = re.search(about, text) + if found is None: + problems.append(f"{name}: не нашлось общее число недель") + elif abs(int(found.group(1)) - total) > 2: + problems.append( + f"{name}: обещано {found.group(1)} недель, в таблицах {total:g} — " + "это уже не округление" + ) + + found = re.search(exact, text) + if found is None: + problems.append(f"{name}: не нашлось число недель на части I–II") + elif int(found.group(1)) != head: + problems.append( + f"{name}: части I–II названы как {found.group(1)} недель, " + f"а их собственные таблицы дают {head:g}" + ) + + # .zenodo.json должен ещё и разбираться. Неразбираемый файл Zenodo просто + # не прочтёт, а релиз при этом выйдет — с записью по умолчанию и без + # авторов. Сообщением, а не исключением: падать трейсбэком там, где + # остальные расхождения читаются строкой, — плохой отчёт. + try: + json.loads(read(".zenodo.json")) + except json.JSONDecodeError as broken: + problems.append(f".zenodo.json: не разбирается как JSON — {broken}") + + for problem in problems: + print(f" {problem}") + if problems: + print(f"\n расхождений: {len(problems)}", file=sys.stderr) + return 1 + + weeks_ru = ru_plural(int(ru_weeks), "неделя", "недели", "недель") + print( + f" {n} {modules_ru} в {parts} {parts_ru_prep}, {len(notebooks)} ноутбуков, " + f"{ru_weeks:g} {weeks_ru} — и все {len(checked)} мест, где курс это говорит " + f"({len(set(checked))} файлов), говорят то же." + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/check_social.py b/scripts/check_social.py new file mode 100644 index 0000000..53a6045 --- /dev/null +++ b/scripts/check_social.py @@ -0,0 +1,113 @@ +#!/usr/bin/env python3 +"""Проверки собранного сайта: каждая страница что-то показывает, когда ей делятся. + +Теги Open Graph отдаёт шаблон `overrides/main.html`, а не сама страница, и +ошибиться в нём можно тихо. `page` пуста при отрисовке 404 — jinja в этом месте +не падает, а подставляет пустую строку, и весь сайт уезжает с пустым og:title. +Переименовали картинку — og:image остаётся, но ведёт в никуда. Перенесли блок +`social_meta` так, что лендинг не заменяет теги, а дописывает, — og:title +выходит дважды, и какой из них возьмёт сборщик карточки, решает сборщик. + +Ничего из этого не заметит ни `mkdocs build --strict` (шаблон отработал), +ни проверка ссылок (теги — не ссылки), ни глаз: карточку видно только там, +куда ссылку вставили. + +Только стандартная библиотека, и работает по уже собранному каталогу: + + mkdocs build --strict + python scripts/check_social.py # по умолчанию ./site + python scripts/check_social.py <каталог> +""" + +from __future__ import annotations + +import pathlib +import re +import sys +from collections import Counter + +SITE_URL = "https://drobyshevdev.github.io/lemma/" + +# Страницы без собственного адреса: 404 отдаётся с любого пути, поэтому +# требовать от неё уникальный og:url бессмысленно. +NO_OWN_URL = {"404.html", "en/404.html"} + + +def _meta(html: str, attr: str, name: str) -> list[str]: + return re.findall(rf' int: + site = pathlib.Path(argv[1] if len(argv) > 1 else "site") + if not site.is_dir(): + print(f" каталога со сборкой нет: {site} — сначала mkdocs build", file=sys.stderr) + return 2 + + pages = sorted(site.rglob("*.html")) + if not pages: + print(f" в {site} нет ни одной страницы", file=sys.stderr) + return 2 + + problems: list[str] = [] + titles: Counter[str] = Counter() + urls: Counter[str] = Counter() + + for page in pages: + rel = page.relative_to(site).as_posix() + html = page.read_text(encoding="utf-8", errors="replace") + + for attr, name in (("property", "og:title"), ("property", "og:image"), + ("property", "og:url"), ("name", "twitter:card")): + found = _meta(html, attr, name) + if not found: + problems.append(f"{rel}: нет {name}") + elif len(found) > 1: + # Ровно то, ради чего social_meta сделан отдельным блоком. + problems.append(f"{rel}: {name} выведен {len(found)} раза — {found}") + elif not found[0].strip(): + problems.append(f"{rel}: {name} пуст") + + title = _meta(html, "property", "og:title") + if len(title) == 1 and title[0].strip(): + titles[title[0]] += 1 + + url = _meta(html, "property", "og:url") + if len(url) == 1 and url[0].strip() and rel not in NO_OWN_URL: + urls[url[0]] += 1 + + card = _meta(html, "name", "twitter:card") + if card and card[0] != "summary_large_image": + problems.append( + f"{rel}: twitter:card = {card[0]}, а карточка 1280×640 — " + "summary_large_image, иначе её покажут миниатюрой" + ) + + for image in _meta(html, "property", "og:image"): + if not image.startswith("https://"): + problems.append(f"{rel}: og:image не абсолютен ({image}) — сборщику карточки нужен адрес") + elif image.startswith(SITE_URL): + target = site / image[len(SITE_URL):] + if not target.is_file(): + problems.append(f"{rel}: og:image ведёт на {image}, а файла в сборке нет") + + # Одинаковый заголовок на всех — признак того, что шаблон не видит page и + # подставляет config.site_name. Сборка при этом проходит. + for title, count in titles.items(): + if count > 2: # два лендинга, ru и en, законно зовутся одинаково + problems.append(f"og:title «{title}» повторяется на {count} страницах — шаблон не видит page?") + for url, count in urls.items(): + if count > 1: + problems.append(f"og:url {url} повторяется на {count} страницах") + + for problem in problems: + print(f" {problem}") + if problems: + print(f"\n проблем: {len(problems)}", file=sys.stderr) + return 1 + + print(f" {len(pages)} страниц, у каждой свой заголовок и адрес, картинка одна и она на месте.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv))