From fb342e7b9d51804326392897d32eade77be2e262 Mon Sep 17 00:00:00 2001 From: emil Date: Mon, 6 Jul 2026 10:15:20 +0200 Subject: [PATCH] DEVX-118: feat: enrich lint_docs.py with single H1, max depth, line length, code block lang, orphan checks MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Add check_single_h1: each markdown file should have at most one H1 - Add check_max_heading_depth: headings should not exceed H4 (configurable) - Add check_line_length: warn on lines >120 chars (non-blocking — badge URLs) - Add check_code_block_languages: fenced code blocks must specify a language - Add check_orphan_docs: warn on docs not linked from index.md or mapping.json - Fix all code blocks in docs to specify language (text for plain blocks) - Fix duplicate H1 in .vale/styles/devx/README.md - Add 18 new tests for full coverage of new checks Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .vale/styles/devx/README.md | 3 +- .vale/styles/write-good/README.md | 2 +- AGENTS.md | 4 +- README.md | 2 +- docs/tech/architecture.md | 10 +- docs/tech/ci-cd-workflow.md | 4 +- docs/user/getting-started.md | 2 +- src/devx/ci/lint_docs.py | 175 +++++++++++++++++++++++++++++- tests/unit/test_lint_docs.py | 155 ++++++++++++++++++++++++++ 9 files changed, 342 insertions(+), 15 deletions(-) diff --git a/.vale/styles/devx/README.md b/.vale/styles/devx/README.md index 5317f02..16bf93c 100644 --- a/.vale/styles/devx/README.md +++ b/.vale/styles/devx/README.md @@ -1,2 +1,3 @@ # Custom Vale style for devx documentation -# Project-specific terminology and style rules + +Project-specific terminology and style rules diff --git a/.vale/styles/write-good/README.md b/.vale/styles/write-good/README.md index 3edcc9b..953c5a1 100644 --- a/.vale/styles/write-good/README.md +++ b/.vale/styles/write-good/README.md @@ -2,7 +2,7 @@ Based on [write-good](https://github.com/btford/write-good). > Naive linter for English prose for developers who can't write good and wanna learn to do other stuff good too. -``` +```text The MIT License (MIT) Copyright (c) 2014 Brian Ford diff --git a/AGENTS.md b/AGENTS.md index a7198e3..6840de2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -57,7 +57,7 @@ devx is a reusable Python package providing development and CI/CD tools for obla ### Package Structure -``` +```text src/devx/ ├── __init__.py # Version (single source of truth, read by setuptools) ├── cli.py # Click-based CLI entry point (devx command) @@ -158,7 +158,7 @@ git checkout -b DEVX-N-short-description ### 4. Commit (Conventional Commits) Branch commits use conventional commit format (no `DEVX-N:` prefix): -``` +```text feat: add new feature fix: resolve bug docs: update README diff --git a/README.md b/README.md index e7dd571..a07ac75 100644 --- a/README.md +++ b/README.md @@ -434,7 +434,7 @@ devx is a self-contained Python package under `src/devx/`. It never imports from scripts outside the package. All tools are invoked via `python -m devx.ci.*`, `python -m devx.tools.*`, or `python -m devx.molecule.*`. -``` +```text src/devx/ ├── __init__.py # Version (single source of truth, read by setuptools) ├── cli.py # Click-based CLI entry point (devx command) diff --git a/docs/tech/architecture.md b/docs/tech/architecture.md index f132e6c..7663692 100644 --- a/docs/tech/architecture.md +++ b/docs/tech/architecture.md @@ -6,7 +6,7 @@ from scripts outside the package. ## Package structure -``` +```text src/devx/ ├── __init__.py # Version (single source of truth, read by setuptools) ├── cli.py # Click-based CLI entry point (devx command) @@ -439,7 +439,7 @@ v2 failures. Supports loading custom platforms from a JSON file. ### PR lifecycle -``` +```text Developer creates Vikunja task (DEVX-N) │ ▼ @@ -475,7 +475,7 @@ CI workflow (ci.yml) triggers: ### Post-merge flow -``` +```text Push to master (squash-merge commit: "DEVX-N ") │ ▼ @@ -519,7 +519,7 @@ Post-merge workflow (post-merge.yml) triggers: ### Publish flow -``` +```text Tag push (vX.Y.Z) triggers publish workflow (publish.yml): │ ▼ @@ -536,7 +536,7 @@ Tag push (vX.Y.Z) triggers publish workflow (publish.yml): ### Badge generation flow -``` +```text push_badges.py: │ ├── fetch_latest_master() → git fetch + reset --hard origin/master diff --git a/docs/tech/ci-cd-workflow.md b/docs/tech/ci-cd-workflow.md index 1fe1a14..6371152 100644 --- a/docs/tech/ci-cd-workflow.md +++ b/docs/tech/ci-cd-workflow.md @@ -6,7 +6,7 @@ tag-triggered publishing. ## Workflow overview -``` +```text PR opened/synchronized ──► CI (ci.yml) │ ├── quality │ ├── detect-changes @@ -143,7 +143,7 @@ updates. ### Job dependency graph -``` +```text detect-type ──┬── validate-commit-msg (skip if release commit) ├── release (skip if release commit) │ │ diff --git a/docs/user/getting-started.md b/docs/user/getting-started.md index 8ad59da..ace22ac 100644 --- a/docs/user/getting-started.md +++ b/docs/user/getting-started.md @@ -72,7 +72,7 @@ tea CLI, etc.) and configure pre-commit hooks. devx expects a `docs/` directory with at minimum: -``` +```text docs/ ├── index.md # Documentation home page ├── mapping.json # Wiki page title mappings diff --git a/src/devx/ci/lint_docs.py b/src/devx/ci/lint_docs.py index 381a02b..6543baa 100644 --- a/src/devx/ci/lint_docs.py +++ b/src/devx/ci/lint_docs.py @@ -7,6 +7,12 @@ Checks performed (all configurable via pyproject.toml ``[tool.devx.docs]``): - **Broken internal links**: relative paths and anchors in markdown files must resolve to actual files and headings. - **Heading hierarchy**: no skipping heading levels (e.g., ``#`` → ``###``). +- **Single H1**: each markdown file should have at most one H1 heading. +- **Max heading depth**: headings should not exceed H4 (configurable). +- **Max line length**: lines should not exceed 120 characters (configurable). +- **Code block language**: fenced code blocks should specify a language. +- **Orphan docs**: docs not linked from index.md or mapping.json (warning). +- **Mapping completeness**: all docs/*.md should be in mapping.json (warning). - **TODO/FIXME**: flags leftover TODO/FIXME markers in documentation. - **Stale docs**: files not modified in >180 days (warning only). - **Trailing whitespace**: lines should not end with whitespace. @@ -49,6 +55,15 @@ REQUIRED_DOC_FILES = ["index.md"] # Maximum age for docs before they're considered stale (days) STALE_THRESHOLD_DAYS = 180 +# Maximum heading depth (H4 by default) +MAX_HEADING_DEPTH = 4 + +# Maximum line length +MAX_LINE_LENGTH = 120 + +# Code block without language: ``` followed by optional whitespace only +_CODE_BLOCK_NO_LANG_RE = re.compile(r"^```[ \t]*$", re.MULTILINE) + # Files excluded from duplicate heading checks (auto-generated or structured # with repeated subsections under different parent sections) DUPLICATE_HEADING_EXCLUDES = { @@ -318,6 +333,120 @@ def check_duplicate_headings(root: Path) -> list[str]: return issues +def check_single_h1(root: Path) -> list[str]: + """Check that each markdown file has at most one H1 heading.""" + issues: list[str] = [] + md_files = [f for f in root.rglob("*.md") if not any(part in _EXCLUDE_DIRS for part in f.parts)] + + for md_file in md_files: + rel_path = md_file.relative_to(root) + if md_file.name in DUPLICATE_HEADING_EXCLUDES: + continue + content = strip_code_blocks(md_file.read_text(encoding="utf-8")) + h1_count = len(re.findall(r"^#\s+", content, re.MULTILINE)) + if h1_count > 1: + issues.append(f"{rel_path}: {h1_count} H1 headings — should have at most 1") + + return issues + + +def check_max_heading_depth(root: Path) -> list[str]: + """Check that headings don't exceed MAX_HEADING_DEPTH.""" + issues: list[str] = [] + md_files = [f for f in root.rglob("*.md") if not any(part in _EXCLUDE_DIRS for part in f.parts)] + + for md_file in md_files: + rel_path = md_file.relative_to(root) + content = strip_code_blocks(md_file.read_text(encoding="utf-8")) + for match in re.finditer(r"^(#{1,6})\s+", content, re.MULTILINE): + level = len(match.group(1)) + if level > MAX_HEADING_DEPTH: + line_num = content[: match.start()].count("\n") + 1 + issues.append(f"{rel_path}:{line_num}: heading depth H{level} exceeds max H{MAX_HEADING_DEPTH}") + + return issues + + +def check_line_length(root: Path) -> list[str]: + """Check that no lines exceed MAX_LINE_LENGTH characters.""" + issues: list[str] = [] + md_files = [f for f in root.rglob("*.md") if not any(part in _EXCLUDE_DIRS for part in f.parts)] + + for md_file in md_files: + rel_path = md_file.relative_to(root) + content = md_file.read_text(encoding="utf-8") + for i, line in enumerate(content.splitlines(), 1): + if len(line) > MAX_LINE_LENGTH: + issues.append(f"{rel_path}:{i}: line too long ({len(line)} > {MAX_LINE_LENGTH} chars)") + + return issues + + +def check_code_block_languages(root: Path) -> list[str]: + """Check that fenced code blocks specify a language.""" + issues: list[str] = [] + md_files = [f for f in root.rglob("*.md") if not any(part in _EXCLUDE_DIRS for part in f.parts)] + + for md_file in md_files: + rel_path = md_file.relative_to(root) + content = md_file.read_text(encoding="utf-8") + in_code_block = False + for i, line in enumerate(content.splitlines(), 1): + stripped = line.strip() + if stripped.startswith("```"): + if not in_code_block: + # Opening fence — check for language + if _CODE_BLOCK_NO_LANG_RE.match(line): + issues.append(f"{rel_path}:{i}: code block without language specifier") + in_code_block = True + else: + # Closing fence + in_code_block = False + + return issues + + +def check_orphan_docs(root: Path, docs_dir: Path) -> list[str]: + """Check for docs not linked from index.md or mapping.json (warnings).""" + issues: list[str] = [] + if not docs_dir.is_dir(): + return issues + + # Collect all referenced files from index.md and mapping.json + referenced: set[str] = set() + index_file = docs_dir / "index.md" + if index_file.exists(): + content = index_file.read_text(encoding="utf-8") + for match in _LINK_RE.finditer(content): + url = match.group(2).strip() + if not url.startswith(("http://", "https://", "mailto:")): + referenced.add(url.split("#")[0]) + + mapping_file = docs_dir / "mapping.json" + if mapping_file.exists(): + try: + mapping = json.loads(mapping_file.read_text(encoding="utf-8")) + if isinstance(mapping, dict): + # Add both keys (filenames) and values (wiki page names) + for k, v in mapping.items(): + if isinstance(k, str): + referenced.add(k) + if isinstance(v, str): + referenced.add(v) + except (json.JSONDecodeError, AttributeError): + pass + + # Check each doc file + for md_file in sorted(docs_dir.rglob("*.md")): + if md_file.name == "index.md": + continue + rel_path = md_file.relative_to(docs_dir).as_posix() + if rel_path not in referenced and md_file.name not in referenced: + issues.append(f"docs/{rel_path}: orphan doc — not linked from index.md or mapping.json") + + return issues + + @click.command() @click.option("--root", default=".", help="Repository root directory.") @click.option("--docs-dir", default=None, help="Docs directory (default: /docs).") @@ -327,6 +456,11 @@ def check_duplicate_headings(root: Path) -> list[str]: @click.option("--check-stale/--no-check-stale", default=False, help="Check for stale docs.") @click.option("--check-trailing/--no-check-trailing", default=True, help="Check trailing whitespace.") @click.option("--check-duplicates/--no-check-duplicates", default=True, help="Check duplicate headings.") +@click.option("--check-single-h1/--no-check-single-h1", "single_h1", default=True, help="Check single H1 per file.") +@click.option("--check-depth/--no-check-depth", "depth", default=True, help="Check max heading depth.") +@click.option("--check-line-length/--no-check-line-length", "line_length", default=True, help="Check line length.") +@click.option("--check-code-lang/--no-check-code-lang", "code_lang", default=True, help="Check code block languages.") +@click.option("--check-orphans/--no-check-orphans", "orphans", default=False, help="Check for orphan docs (warnings).") @click.option("--fix", is_flag=True, default=False, help="Auto-fix trailing whitespace.") def main( root: str, @@ -337,6 +471,11 @@ def main( check_stale: bool, check_trailing: bool, check_duplicates: bool, + single_h1: bool, + depth: bool, + line_length: bool, + code_lang: bool, + orphans: bool, fix: bool, ) -> None: """Lint documentation files for structure, links, and quality.""" @@ -369,6 +508,31 @@ def main( click.echo(_("Checking duplicate headings...")) all_issues.extend(check_duplicate_headings(root_path)) + # Single H1 + if single_h1: + click.echo(_("Checking single H1 per file...")) + all_issues.extend(check_single_h1(root_path)) + + # Max heading depth + if depth: + click.echo(_("Checking max heading depth...")) + all_issues.extend(check_max_heading_depth(root_path)) + + # Line length (warnings — badge URLs and tables can exceed 120) + if line_length: + click.echo(_("Checking line length...")) + ll_issues = check_line_length(root_path) + for issue in ll_issues[:10]: # Show first 10 only + click.echo(f" WARN: {issue}") + if len(ll_issues) > 10: + click.echo(_(" ... and {n} more", n=len(ll_issues) - 10)) + click.echo(_(" {n} long lines found (warnings only)", n=len(ll_issues))) + + # Code block languages + if code_lang: + click.echo(_("Checking code block languages...")) + all_issues.extend(check_code_block_languages(root_path)) + # TODO/FIXME if check_todo: click.echo(_("Checking for TODO/FIXME markers...")) @@ -391,15 +555,22 @@ def main( else: all_issues.extend(ws_issues) - # Stale docs + # Stale docs (warnings) if check_stale: click.echo(_("Checking for stale docs...")) stale = check_stale_docs(root_path) for issue in stale: click.echo(f" WARN: {issue}") - # Stale docs are warnings, not errors click.echo(_(" {n} stale docs found (warnings only)", n=len(stale))) + # Orphan docs (warnings) + if orphans: + click.echo(_("Checking for orphan docs...")) + orphan_issues = check_orphan_docs(root_path, docs_path) + for issue in orphan_issues: + click.echo(f" WARN: {issue}") + click.echo(_(" {n} orphan docs found (warnings only)", n=len(orphan_issues))) + # Report click.echo(f"\n{'=' * 60}") if all_issues: diff --git a/tests/unit/test_lint_docs.py b/tests/unit/test_lint_docs.py index f350bc5..bb6c52a 100644 --- a/tests/unit/test_lint_docs.py +++ b/tests/unit/test_lint_docs.py @@ -9,11 +9,16 @@ from pathlib import Path from click.testing import CliRunner from devx.ci.lint_docs import ( + check_code_block_languages, check_docs_structure, check_duplicate_headings, check_heading_hierarchy, check_internal_links, + check_line_length, + check_max_heading_depth, + check_orphan_docs, check_required_files, + check_single_h1, check_stale_docs, check_todo_fixme, check_trailing_whitespace, @@ -362,6 +367,100 @@ class TestCheckDuplicateHeadings: assert issues == [] +class TestCheckSingleH1: + def test_single_h1_ok(self, tmp_path: Path) -> None: + (tmp_path / "README.md").write_text("# Title\n## Section\n") + issues = check_single_h1(tmp_path) + assert issues == [] + + def test_multiple_h1_fails(self, tmp_path: Path) -> None: + (tmp_path / "README.md").write_text("# Title 1\n# Title 2\n") + issues = check_single_h1(tmp_path) + assert len(issues) == 1 + assert "2 H1" in issues[0] + + def test_no_h1_ok(self, tmp_path: Path) -> None: + (tmp_path / "README.md").write_text("## Section\n") + issues = check_single_h1(tmp_path) + assert issues == [] + + +class TestCheckMaxHeadingDepth: + def test_ok(self, tmp_path: Path) -> None: + (tmp_path / "README.md").write_text("# H1\n## H2\n### H3\n#### H4\n") + issues = check_max_heading_depth(tmp_path) + assert issues == [] + + def test_too_deep(self, tmp_path: Path) -> None: + (tmp_path / "README.md").write_text("# H1\n##### H5\n") + issues = check_max_heading_depth(tmp_path) + assert len(issues) == 1 + assert "H5" in issues[0] + + +class TestCheckLineLength: + def test_ok(self, tmp_path: Path) -> None: + (tmp_path / "README.md").write_text("# Short line\n") + issues = check_line_length(tmp_path) + assert issues == [] + + def test_too_long(self, tmp_path: Path) -> None: + (tmp_path / "README.md").write_text("# " + "x" * 200 + "\n") + issues = check_line_length(tmp_path) + assert len(issues) == 1 + assert "202" in issues[0] + + +class TestCheckCodeBlockLanguages: + def test_with_language(self, tmp_path: Path) -> None: + (tmp_path / "README.md").write_text("```python\nprint('hi')\n```\n") + issues = check_code_block_languages(tmp_path) + assert issues == [] + + def test_without_language(self, tmp_path: Path) -> None: + (tmp_path / "README.md").write_text("```\nplain text\n```\n") + issues = check_code_block_languages(tmp_path) + assert len(issues) == 1 + assert "without language" in issues[0] + + def test_closing_fence_not_flagged(self, tmp_path: Path) -> None: + (tmp_path / "README.md").write_text("```python\nprint('hi')\n```\n") + issues = check_code_block_languages(tmp_path) + assert issues == [] + + +class TestCheckOrphanDocs: + def test_no_orphans(self, tmp_path: Path) -> None: + docs = tmp_path / "docs" + docs.mkdir() + (docs / "index.md").write_text("# Home\n[link](page.md)\n") + (docs / "page.md").write_text("# Page\n") + issues = check_orphan_docs(tmp_path, docs) + assert issues == [] + + def test_orphan_found(self, tmp_path: Path) -> None: + docs = tmp_path / "docs" + docs.mkdir() + (docs / "index.md").write_text("# Home\n") + (docs / "page.md").write_text("# Page\n") + issues = check_orphan_docs(tmp_path, docs) + assert len(issues) == 1 + assert "orphan" in issues[0] + + def test_no_docs_dir(self, tmp_path: Path) -> None: + issues = check_orphan_docs(tmp_path, tmp_path / "docs") + assert issues == [] + + def test_referenced_in_mapping(self, tmp_path: Path) -> None: + docs = tmp_path / "docs" + docs.mkdir() + (docs / "index.md").write_text("# Home\n") + (docs / "mapping.json").write_text(json.dumps({"page.md": "Page"})) + (docs / "page.md").write_text("# Page\n") + issues = check_orphan_docs(tmp_path, docs) + assert issues == [] + + class TestMain: def test_passes_clean_repo(self, tmp_path: Path) -> None: """A clean repo with all files should pass.""" @@ -434,3 +533,59 @@ class TestMain: # Stale docs are warnings, not errors assert result.exit_code == 0 assert "stale" in result.output + + def test_line_length_warning(self, tmp_path: Path) -> None: + """--check-line-length should warn but not fail.""" + (tmp_path / "README.md").write_text("# " + "x" * 200 + "\n") + (tmp_path / "AGENTS.md").write_text("# AGENTS\n") + (tmp_path / "CHANGELOG.md").write_text("# Changelog\n") + docs = tmp_path / "docs" + docs.mkdir() + (docs / "index.md").write_text("# Home\n") + (docs / "mapping.json").write_text(json.dumps({"index.md": "Home"})) + runner = CliRunner() + result = runner.invoke(main, ["--root", str(tmp_path), "--check-line-length"]) + assert result.exit_code == 0 + assert "long lines" in result.output + + def test_line_length_many_warnings(self, tmp_path: Path) -> None: + """More than 10 long lines should show '... and N more'.""" + long_line = "x" * 200 + "\n" + (tmp_path / "README.md").write_text(long_line * 15) + (tmp_path / "AGENTS.md").write_text("# AGENTS\n") + (tmp_path / "CHANGELOG.md").write_text("# Changelog\n") + docs = tmp_path / "docs" + docs.mkdir() + (docs / "index.md").write_text("# Home\n") + (docs / "mapping.json").write_text(json.dumps({"index.md": "Home"})) + runner = CliRunner() + result = runner.invoke(main, ["--root", str(tmp_path), "--check-line-length"]) + assert result.exit_code == 0 + assert "more" in result.output + + def test_orphan_docs_warning(self, tmp_path: Path) -> None: + """--check-orphans should warn but not fail.""" + (tmp_path / "README.md").write_text("# Title\n") + (tmp_path / "AGENTS.md").write_text("# AGENTS\n") + (tmp_path / "CHANGELOG.md").write_text("# Changelog\n") + docs = tmp_path / "docs" + docs.mkdir() + (docs / "index.md").write_text("# Home\n") + (docs / "mapping.json").write_text(json.dumps({"index.md": "Home"})) + (docs / "orphan.md").write_text("# Orphan\n") + runner = CliRunner() + result = runner.invoke(main, ["--root", str(tmp_path), "--check-orphans"]) + assert result.exit_code == 0 + assert "orphan" in result.output + + def test_orphan_docs_invalid_mapping(self, tmp_path: Path) -> None: + """Invalid mapping.json should not crash orphan check.""" + docs = tmp_path / "docs" + docs.mkdir() + (docs / "index.md").write_text("# Home\n") + (docs / "mapping.json").write_text("invalid json{") + (docs / "page.md").write_text("# Page\n") + # Should not raise — just returns issues + issues = check_orphan_docs(tmp_path, docs) + assert len(issues) == 1 + assert "orphan" in issues[0]