mirror of
https://github.com/VectifyAI/PageIndex.git
synced 2026-07-24 21:41:04 +02:00
Merge pull request #250 from VectifyAI/feat/md-bold-heading-recognition
Recognize whole-line bold as level-1 heading in markdown parser
This commit is contained in:
commit
c58cd62b50
2 changed files with 35 additions and 10 deletions
|
|
@ -31,6 +31,7 @@ async def generate_summaries_for_structure_md(structure, summary_token_threshold
|
|||
|
||||
def extract_nodes_from_markdown(markdown_content):
|
||||
header_pattern = r'^(#{1,6})\s+(.+)$'
|
||||
bold_heading_pattern = r'^\*\*(.+?)\*\*\s*$'
|
||||
code_block_pattern = r'^```'
|
||||
node_list = []
|
||||
|
||||
|
|
@ -54,7 +55,15 @@ def extract_nodes_from_markdown(markdown_content):
|
|||
match = re.match(header_pattern, stripped_line)
|
||||
if match:
|
||||
title = match.group(2).strip()
|
||||
node_list.append({'node_title': title, 'line_num': line_num})
|
||||
level = len(match.group(1))
|
||||
node_list.append({'node_title': title, 'line_num': line_num, 'level': level})
|
||||
continue
|
||||
|
||||
bold_match = re.match(bold_heading_pattern, stripped_line)
|
||||
if bold_match:
|
||||
title = bold_match.group(1).strip()
|
||||
if title:
|
||||
node_list.append({'node_title': title, 'line_num': line_num, 'level': 1})
|
||||
|
||||
return node_list, lines
|
||||
|
||||
|
|
@ -62,17 +71,10 @@ def extract_nodes_from_markdown(markdown_content):
|
|||
def extract_node_text_content(node_list, markdown_lines):
|
||||
all_nodes = []
|
||||
for node in node_list:
|
||||
line_content = markdown_lines[node['line_num'] - 1]
|
||||
header_match = re.match(r'^(#{1,6})', line_content)
|
||||
|
||||
if header_match is None:
|
||||
print(f"Warning: Line {node['line_num']} does not contain a valid header: '{line_content}'")
|
||||
continue
|
||||
|
||||
processed_node = {
|
||||
'title': node['node_title'],
|
||||
'line_num': node['line_num'],
|
||||
'level': len(header_match.group(1))
|
||||
'level': node['level']
|
||||
}
|
||||
all_nodes.append(processed_node)
|
||||
|
||||
|
|
@ -339,4 +341,4 @@ if __name__ == "__main__":
|
|||
with open(output_path, 'w', encoding='utf-8') as f:
|
||||
json.dump(tree_structure, f, indent=2, ensure_ascii=False)
|
||||
|
||||
print(f"\nTree structure saved to: {output_path}")
|
||||
print(f"\nTree structure saved to: {output_path}")
|
||||
|
|
|
|||
23
tests/test_page_index_md.py
Normal file
23
tests/test_page_index_md.py
Normal file
|
|
@ -0,0 +1,23 @@
|
|||
import unittest
|
||||
|
||||
from pageindex.page_index_md import extract_nodes_from_markdown
|
||||
|
||||
|
||||
class ExtractNodesFromMarkdownTest(unittest.TestCase):
|
||||
def test_skips_bold_heading_with_only_whitespace(self):
|
||||
nodes, _ = extract_nodes_from_markdown("** **\n**Valid heading**")
|
||||
|
||||
self.assertEqual(
|
||||
nodes,
|
||||
[
|
||||
{
|
||||
"node_title": "Valid heading",
|
||||
"line_num": 2,
|
||||
"level": 1,
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue