From 3ee66fed133a08dcab18953e595d66e3720d5baa Mon Sep 17 00:00:00 2001 From: Abhishek B R Date: Thu, 1 Oct 2026 00:36:30 +0530 Subject: [PATCH] fix: slice search sections on the same line breaks markdown-it counts project_sections cut the body with str.splitlines(), but the heading boundaries come from markdown-it token maps, which only count \n, \r\n and \r as line breaks. splitlines() also breaks on form feeds, U+2028, U+2029, U+0085 and a few control characters, so one of those in a page shifted every later slice: section bodies picked up their own heading line and the text before it moved into the wrong section. Split on the same three breaks markdown-it uses. --- src/codealmanac/services/wiki/sections.py | 7 ++++++- tests/test_wiki_sections.py | 21 +++++++++++++++++++++ 2 files changed, 27 insertions(+), 1 deletion(-) diff --git a/src/codealmanac/services/wiki/sections.py b/src/codealmanac/services/wiki/sections.py index 4f2b0b4b..4c186418 100644 --- a/src/codealmanac/services/wiki/sections.py +++ b/src/codealmanac/services/wiki/sections.py @@ -1,8 +1,13 @@ +import re + from markdown_it import MarkdownIt from codealmanac.core.models import CodeAlmanacModel MARKDOWN = MarkdownIt("commonmark", {"html": False, "linkify": False}) +# Token.map line numbers only count these breaks. str.splitlines() also splits +# on form feeds, U+2028 and friends, which would shift every slice after them. +SOURCE_LINE = re.compile(r"[^\r\n]*(?:\r\n?|\n)|[^\r\n]+$") class WikiSection(CodeAlmanacModel): @@ -29,7 +34,7 @@ def project_sections(body: str, page_title: str) -> tuple[WikiSection, ...]: if not boundaries: return (section(0, (page_title,), body),) - lines = body.splitlines(keepends=True) + lines = SOURCE_LINE.findall(body) sections: list[WikiSection] = [] headings: dict[int, str] = {} diff --git a/tests/test_wiki_sections.py b/tests/test_wiki_sections.py index 40da32d2..e68707b0 100644 --- a/tests/test_wiki_sections.py +++ b/tests/test_wiki_sections.py @@ -67,3 +67,24 @@ def test_heading_free_page_still_has_one_searchable_section(): assert len(sections) == 1 assert sections[0].heading_path == ("Note",) assert sections[0].body == "Only prose.\n" + + +def test_section_slices_count_lines_the_way_the_markdown_parser_does(): + body = ( + "Pasted note
with a line separator.\n\n" + "## Setup\n\n" + "Install it.\x0cNext page.\n\n" + "## Usage\n\n" + "Run it.\n" + ) + + sections = project_sections(body, page_title="Tool") + + assert [section.heading_path for section in sections] == [ + ("Tool",), + ("Setup",), + ("Usage",), + ] + assert sections[0].body == "Pasted note
with a line separator.\n\n" + assert sections[1].body == "\nInstall it.\x0cNext page.\n\n" + assert sections[2].body == "\nRun it.\n"