diff --git a/src/codealmanac/services/wiki/sections.py b/src/codealmanac/services/wiki/sections.py index 4f2b0b4b..4c186418 100644 --- a/src/codealmanac/services/wiki/sections.py +++ b/src/codealmanac/services/wiki/sections.py @@ -1,8 +1,13 @@ +import re + from markdown_it import MarkdownIt from codealmanac.core.models import CodeAlmanacModel MARKDOWN = MarkdownIt("commonmark", {"html": False, "linkify": False}) +# Token.map line numbers only count these breaks. str.splitlines() also splits +# on form feeds, U+2028 and friends, which would shift every slice after them. +SOURCE_LINE = re.compile(r"[^\r\n]*(?:\r\n?|\n)|[^\r\n]+$") class WikiSection(CodeAlmanacModel): @@ -29,7 +34,7 @@ def project_sections(body: str, page_title: str) -> tuple[WikiSection, ...]: if not boundaries: return (section(0, (page_title,), body),) - lines = body.splitlines(keepends=True) + lines = SOURCE_LINE.findall(body) sections: list[WikiSection] = [] headings: dict[int, str] = {} diff --git a/tests/test_wiki_sections.py b/tests/test_wiki_sections.py index 40da32d2..e68707b0 100644 --- a/tests/test_wiki_sections.py +++ b/tests/test_wiki_sections.py @@ -67,3 +67,24 @@ def test_heading_free_page_still_has_one_searchable_section(): assert len(sections) == 1 assert sections[0].heading_path == ("Note",) assert sections[0].body == "Only prose.\n" + + +def test_section_slices_count_lines_the_way_the_markdown_parser_does(): + body = ( + "Pasted note
with a line separator.\n\n" + "## Setup\n\n" + "Install it.\x0cNext page.\n\n" + "## Usage\n\n" + "Run it.\n" + ) + + sections = project_sections(body, page_title="Tool") + + assert [section.heading_path for section in sections] == [ + ("Tool",), + ("Setup",), + ("Usage",), + ] + assert sections[0].body == "Pasted note
with a line separator.\n\n" + assert sections[1].body == "\nInstall it.\x0cNext page.\n\n" + assert sections[2].body == "\nRun it.\n"