Files
hbm-foundry-system/scripts/parsers/book-walker.ts
T

186 lines
6.3 KiB
TypeScript

/**
* Lightweight markdown walker tuned to the HbM book layout.
*
* Supports two formats:
*
* 1. Legacy Google-Docs export (plain text, `_books/`):
* ## Chapter heading
* Section heading (plain text / bold)
* Item Name
* * Field: value
*
* 2. Native Obsidian/repo files (`rules/`, `disciplines/`, etc.):
* --- ← YAML frontmatter (skipped)
* ...
* ---
* ## Rozdział I - ... ← chapter (any heading level)
* ### Sub-section ← section / sub-section
* #### Spell Name ← item name
* - [[#Anchor|Label]] ← Obsidian TOC bullet (skipped)
* * Field: value
*
* The walker emits one `BookBlock` per item it can find. Section context
* is carried through so parsers know e.g. which discipline a spell lives
* under.
*/
import { readFileSync } from 'node:fs';
export interface BookBlock {
/** The line directly preceding the bullet block - usually the item name. */
name: string;
/** Bullet keys/values, preserved in order. */
fields: Array<{ key: string; value: string }>;
/** Free-form paragraph text following the bullet block (description). */
description: string;
/** Most recent chapter (any `#+ ...`) heading. */
chapter: string;
/** Most recent non-bullet, non-empty line above the item name (sub-section). */
section: string;
/** 1-based line number of the item name in the source file. */
line: number;
}
const SEPARATOR = /^_{3,}$/;
const BULLET = /^\*\s+([^:]+):\s*(.+)$/;
// Match any markdown heading level (1-6 `#` chars).
const HEADING = /^#{1,6}\s+(.+)$/;
// Chapter headings are the two highest heading levels we encounter - heuristic:
// treat ## headings as "chapter" and ### headings as "section".
const CHAPTER_HEADING = /^#{1,2}\s+(.+)$/;
const SECTION_HEADING = /^#{3,4}\s+(.+)$/;
// Obsidian TOC bullets: `- [[#Anchor|Label]]` or `* [[#Anchor|Label]]`
const OBSIDIAN_TOC_BULLET = /^[-*]\s+\[\[#/;
// Frontmatter fence
const FRONTMATTER_FENCE = /^---\s*$/;
/** Strip leading markdown heading markers and bold `**` from a raw line. */
function stripHeadingMarkers(line: string): string {
return line.replace(/^#{1,6}\s+/, '').replace(/^\*\*(.+)\*\*$/, '$1').trim();
}
/** Skip YAML frontmatter block at start of file. Returns the index of the first non-frontmatter line. */
function skipFrontmatter(lines: string[]): number {
if (lines.length === 0 || !FRONTMATTER_FENCE.test(lines[0].trim())) return 0;
for (let i = 1; i < lines.length; i++) {
if (FRONTMATTER_FENCE.test(lines[i].trim())) return i + 1;
}
return 0; // malformed - start from 0
}
export interface WalkOptions {
/**
* Predicate that returns `true` when the line (after stripping heading markers)
* is a known section header (e.g. matches a discipline / school / deity label).
* Used by the walker to track the current section across separators and prose.
*/
isSectionHeader?: (line: string) => boolean;
}
export function walkBook(path: string, opts: WalkOptions = {}): BookBlock[] {
const text = readFileSync(path, 'utf8');
const lines = text.split(/\r?\n/);
const blocks: BookBlock[] = [];
let chapter = '';
let section = '';
const isSectionHeader = opts.isSectionHeader ?? (() => false);
const startIdx = skipFrontmatter(lines);
// Look ahead for bullet-block starts; track the line above as the name.
for (let i = startIdx; i < lines.length; i++) {
const rawLine = lines[i].trimEnd();
// Skip Obsidian TOC bullets - they look like `- [[#Section|Label]]`.
if (OBSIDIAN_TOC_BULLET.test(rawLine.trim())) continue;
// Chapter headings: ## or # level (the two highest we honour).
const chMatch = rawLine.match(CHAPTER_HEADING);
if (chMatch) {
chapter = stripHeadingMarkers(rawLine);
// Reset section only for top-level (##) headings.
if (rawLine.startsWith('## ') || rawLine.startsWith('# ')) section = '';
continue;
}
// Section sub-headings: ### or #### level become section markers.
const secMatch = rawLine.match(SECTION_HEADING);
if (secMatch) {
const bare = stripHeadingMarkers(rawLine);
const isL3 = rawLine.startsWith('### ');
if (isL3 || isSectionHeader(bare)) {
section = bare;
continue;
}
}
// Track plain-text section headers (legacy format).
const trimmed = rawLine.trim();
if (trimmed && !SEPARATOR.test(trimmed) && !BULLET.test(trimmed) && isSectionHeader(trimmed)) {
section = trimmed;
continue;
}
if (BULLET.test(rawLine)) {
// Walk up to find the name line (first non-empty, non-separator, non-heading line above).
let nameLine = '';
let nameIdx = i - 1;
while (nameIdx >= startIdx) {
const candidate = lines[nameIdx].trim();
// Skip empty lines, separators, TOC links, list items, and requirements lines
if (
candidate &&
!SEPARATOR.test(candidate) &&
!OBSIDIAN_TOC_BULLET.test(candidate) &&
!/^[*-]\s+/.test(candidate) &&
!/^(?:[*-]\s+)?Wymagania?:/i.test(candidate) &&
!/^>/.test(candidate)
) {
// Strip any heading markers; skip pure chapter/section headings as the name
// (they are already captured in `chapter`/`section`).
nameLine = stripHeadingMarkers(candidate);
break;
}
nameIdx--;
}
// Collect the bullet block.
const fields: Array<{ key: string; value: string }> = [];
let j = i;
while (j < lines.length) {
const m = lines[j].trimEnd().match(BULLET);
if (!m) break;
fields.push({ key: m[1].trim(), value: m[2].trim() });
j++;
}
// Description = following lines until next bullet block, separator, or heading.
const descLines: string[] = [];
let k = j;
while (k < lines.length) {
const cur = lines[k].trimEnd();
if (SEPARATOR.test(cur)) break;
if (HEADING.test(cur)) break;
if (BULLET.test(cur)) break;
if (OBSIDIAN_TOC_BULLET.test(cur.trim())) break;
descLines.push(cur);
k++;
}
blocks.push({
name: nameLine,
fields,
description: descLines.join('\n').trim(),
chapter,
section,
line: nameIdx + 1,
});
i = k - 1; // advance past description
}
}
return blocks;
}