From 8515f5dff9c10ac5b7f685f51e1d0cbc3ff60b10 Mon Sep 17 00:00:00 2001 From: SantamRC Date: Fri, 28 Aug 2026 14:44:45 +0530 Subject: [PATCH 1/9] feat: flutter IB JSON API generator --- utils/md2json/__init__.py | 36 ++++ utils/md2json/__main__.py | 19 ++ utils/md2json/bibliography.py | 186 +++++++++++++++++ utils/md2json/blocks.py | 371 ++++++++++++++++++++++++++++++++++ utils/md2json/book.py | 80 ++++++++ utils/md2json/cli.py | 68 +++++++ utils/md2json/config.py | 76 +++++++ utils/md2json/frontmatter.py | 22 ++ utils/md2json/inline.py | 52 +++++ utils/md2json/model.py | 32 +++ utils/md2json/output.py | 19 ++ 11 files changed, 961 insertions(+) create mode 100644 utils/md2json/__init__.py create mode 100644 utils/md2json/__main__.py create mode 100644 utils/md2json/bibliography.py create mode 100644 utils/md2json/blocks.py create mode 100644 utils/md2json/book.py create mode 100644 utils/md2json/cli.py create mode 100644 utils/md2json/config.py create mode 100644 utils/md2json/frontmatter.py create mode 100644 utils/md2json/inline.py create mode 100644 utils/md2json/model.py create mode 100644 utils/md2json/output.py diff --git a/utils/md2json/__init__.py b/utils/md2json/__init__.py new file mode 100644 index 00000000..e4f628c9 --- /dev/null +++ b/utils/md2json/__init__.py @@ -0,0 +1,36 @@ +"""Translate the Jekyll book in ``docs/`` into the view JSON the app consumes. + +Run it with ``python3 utils/md2json``. There are no options: every run reads +``docs/`` and rewrites the whole output tree. + +Module map: + config constants, paths, widget/heading tables + frontmatter Jekyll front matter parsing + inline inline markdown/HTML -> plain text + blocks block-level markdown -> view documents (the Parser) + model Page / Section / Chapter dataclasses + book discovery and assembly from the docs tree + output serialisation and writing + cli the run loop +""" + +from .blocks import Parser +from .book import build_navbar, build_page, chapter_contents, discover_chapters +from .cli import main +from .frontmatter import parse_front_matter +from .inline import inline_text +from .model import Chapter, Page, Section + +__all__ = [ + "Chapter", + "Page", + "Parser", + "Section", + "build_navbar", + "build_page", + "chapter_contents", + "discover_chapters", + "inline_text", + "main", + "parse_front_matter", +] diff --git a/utils/md2json/__main__.py b/utils/md2json/__main__.py new file mode 100644 index 00000000..fcfab301 --- /dev/null +++ b/utils/md2json/__main__.py @@ -0,0 +1,19 @@ +"""Entry point. + +Supports both `python3 -m md2json` (run as a package, from utils/ or with +utils/ on PYTHONPATH) and `python3 utils/md2json` from anywhere. In the latter +case Python puts *this* directory on sys.path rather than its parent, so the +package is not importable by name and relative imports fail; add the parent +directory and import absolutely instead. +""" + +import sys +from pathlib import Path + +if __package__: + from .cli import main +else: # python3 utils/md2json + sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + from md2json.cli import main + +raise SystemExit(main()) diff --git a/utils/md2json/bibliography.py b/utils/md2json/bibliography.py new file mode 100644 index 00000000..4b382339 --- /dev/null +++ b/utils/md2json/bibliography.py @@ -0,0 +1,186 @@ +"""Resolve {% cite %} / {% bibliography %} against the BibTeX sources. + +Jekyll renders these with jekyll-scholar (``style: ieee-with-url``). Without +this module the citations are stripped and the References sections dropped, +which leaves dangling sentences like "described in Section 1.9 in and in ...". + +Only what the book actually uses is supported: @book, @article, @techreport and +@misc entries, and the ``{% bibliography --cited --file X %}`` tag form. +""" + +from __future__ import annotations + +import re + +from .config import BIBLIOGRAPHY_PATH + +ENTRY_RE = re.compile(r"@(\w+)\s*\{\s*([^,\s]+)\s*,", re.I) +FIELD_RE = re.compile(r"(\w+)\s*=\s*", re.I) +# BibTeX convention: OPT-prefixed fields are commented out and must be ignored. +OPT_PREFIX_RE = re.compile(r"^opt", re.I) +LATEX_ESCAPES = {r"\&": "&", r"\_": "_", r"\%": "%", r"\$": "$", r"\#": "#"} + + +def _read_braced(text: str, start: int) -> tuple[str, int]: + """Read a {...} group with balanced braces, returning its body and end index.""" + depth, i = 0, start + while i < len(text): + if text[i] == "{": + depth += 1 + elif text[i] == "}": + depth -= 1 + if depth == 0: + return text[start + 1:i], i + 1 + i += 1 + return text[start + 1:], len(text) + + +def _clean(value: str) -> str: + for escape, plain in LATEX_ESCAPES.items(): + value = value.replace(escape, plain) + value = value.replace("{", "").replace("}", "") + return re.sub(r"\s+", " ", value).strip().rstrip(",") + + +def _read_value(text: str, i: int) -> tuple[str, int]: + while i < len(text) and text[i].isspace(): + i += 1 + if i < len(text) and text[i] == "{": + raw, i = _read_braced(text, i) + return _clean(raw), i + if i < len(text) and text[i] == '"': + end = text.find('"', i + 1) + end = len(text) if end == -1 else end + return _clean(text[i + 1:end]), end + 1 + match = re.compile(r"[^,}]*").match(text, i) + return _clean(match.group(0)), match.end() + + +def parse_bibtex(text: str) -> dict[str, dict[str, str]]: + """Parse a .bib file into {key: {field: value}}, plus a "_kind" entry type. + + The entry type is stored under "_kind" rather than "type" because BibTeX has + a real `type` field (@techreport uses it for "Standard", "Tech. report", ...). + """ + entries: dict[str, dict[str, str]] = {} + for match in ENTRY_RE.finditer(text): + kind, key = match.group(1).lower(), match.group(2) + brace = text.index("{", match.start()) + body, _ = _read_braced(text, brace) + fields: dict[str, str] = {"_kind": kind} + i = 0 + while i < len(body): + field = FIELD_RE.search(body, i) + if field is None: + break + value, i = _read_value(body, field.end()) + name = field.group(1).lower() + if OPT_PREFIX_RE.match(name) and name != "options": + continue # OPTauthor / OPTmonth: commented out in BibTeX + if value: + fields[name] = value + entries[key] = fields + return entries + + +def load_bibliography() -> dict[str, dict[str, str]]: + """Load every .bib file in _bibliography/ into one lookup table.""" + entries: dict[str, dict[str, str]] = {} + if not BIBLIOGRAPHY_PATH.is_dir(): + return entries + for path in sorted(BIBLIOGRAPHY_PATH.glob("*.bib")): + entries.update(parse_bibtex(path.read_text(encoding="utf-8"))) + return entries + + +def _format_author(author: str) -> str: + """"Donzellini, G. and Oneto, L." -> "G. Donzellini, L. Oneto".""" + names = [n.strip() for n in re.split(r"\s+and\s+", author) if n.strip()] + formatted = [] + for name in names: + if "," in name: + last, first = (part.strip() for part in name.split(",", 1)) + initials = " ".join( + part[0] + "." if not part.endswith(".") else part + for part in first.replace(".", ". ").split() + if part + ) + formatted.append(f"{initials} {last}".strip()) + else: + formatted.append(name) + if len(formatted) > 2: + return ", ".join(formatted[:-1]) + ", and " + formatted[-1] + if len(formatted) == 2: + return f"{formatted[0]} and {formatted[1]}" + return formatted[0] if formatted else "" + + +def format_entry(entry: dict[str, str]) -> str: + """Render one entry roughly in IEEE "ieee-with-url" style, as plain text.""" + parts: list[str] = [] + author = _format_author(entry.get("author", "")) + if author: + parts.append(author + ",") + + title = entry.get("title", "") + kind = entry.get("_kind", "misc") + if kind in {"article", "techreport", "misc"}: + parts.append(f'"{title},"' if title else "") + else: + parts.append(f"{title}." if title else "") + + if kind == "article": + if entry.get("journal"): + parts.append(entry["journal"] + ",") + if entry.get("volume"): + parts.append(f"vol. {entry['volume']},") + if entry.get("number"): + parts.append(f"no. {entry['number']},") + if entry.get("pages"): + parts.append(f"pp. {entry['pages']},") + else: + if entry.get("institution"): + parts.append(entry["institution"] + ",") + if entry.get("number"): + parts.append(entry["number"] + ",") + if entry.get("publisher"): + parts.append(entry["publisher"] + ",") + + if entry.get("year"): + parts.append(f"{entry['year']}.") + + link = entry.get("url") or entry.get("howpublished") or "" + if link.startswith("http"): + parts.append(f"Available: {link}") + elif entry.get("doi"): + parts.append(f"doi: {entry['doi']}") + + return re.sub(r"\s+", " ", " ".join(p for p in parts if p)).strip() + + +class CitationRegistry: + """Numbers the citations on one page, in order of first appearance.""" + + def __init__(self, entries: dict[str, dict[str, str]]): + self.entries = entries + self.order: list[str] = [] + self.missing: list[str] = [] + + def mark(self, keys: list[str]) -> str: + """Register cited keys and return their inline marker, e.g. "[1], [2]".""" + numbers = [] + for key in keys: + if key not in self.entries and key not in self.missing: + self.missing.append(key) + if key not in self.order: + self.order.append(key) + numbers.append(self.order.index(key) + 1) + return ", ".join(f"[{n}]" for n in numbers) + + def rendered(self) -> list[str]: + """The reference list, in citation order.""" + return [ + format_entry(self.entries[key]) + for key in self.order + if key in self.entries + ] diff --git a/utils/md2json/blocks.py b/utils/md2json/blocks.py new file mode 100644 index 00000000..413a9dc9 --- /dev/null +++ b/utils/md2json/blocks.py @@ -0,0 +1,371 @@ +"""Block-level markdown parsing: markdown lines -> a list of view documents.""" + +from __future__ import annotations + +import html +import re + +from .bibliography import CitationRegistry +from .config import ( + DROPPED_HEADINGS, + DROPPED_SECTIONS, + INCLUDE_WIDGETS, + SIZE_BODY, + SIZE_HEADING, + SIZE_SUBHEADING, + STRUCTURAL_INCLUDES, + UNLABELLED_FENCES, +) +from .inline import inline_text + +HEADING_RE = re.compile(r"^(#{1,6})\s+(.*?)\s*#*$") +ATTR_RE = re.compile(r"^\{:.*\}\s*$") +FENCE_RE = re.compile(r"^\s*(?:```|~~~)\s*(\w*)\s*$") +HR_RE = re.compile(r"^\s*(?:-{3,}|\*{3,}|_{3,})\s*$") +TABLE_ROW_RE = re.compile(r"^\s*\|.*") +TABLE_SEP_RE = re.compile(r"^\s*\|?[\s:\-|]+\|[\s:\-|]*$") +LIST_ITEM_RE = re.compile(r"^(\s*)(?:([-*+])|(\d+)[.)])\s+(.*)$") +LIQUID_RE = re.compile(r"^\s*\{%\s*(.*?)\s*%\}\s*$") +INCLUDE_RE = re.compile(r"^include\s+(\S+)(.*)$") +INCLUDE_ARG_RE = re.compile(r"(\w+)\s*=\s*[\"']([^\"']*)[\"']") +IFRAME_SRC_RE = re.compile(r"]*\ssrc=[\"']([^\"']+)[\"']", re.I) + + +class Parser: + def __init__(self, body: str, *, keep_title: bool = False, bibliography=None): + self.lines = body.split("\n") + self.i = 0 + self.keep_title = keep_title + self.citations = CitationRegistry(bibliography or {}) + self.pending_references: str | None = None + self.views: list[dict] = [] + self.toc: list[str] = [] + self.warnings: list[str] = [] + self.wants_toc = False + self.skipping_section = False # inside "Table of contents" / "References" + self.seen_title = False + self.next_is_quiz = False + self.scroll_id = 0 + + # -- helpers ---------------------------------------------------------- # + + def text(self, raw: str) -> str: + """Flatten inline markup, numbering any citations it contains.""" + return inline_text(raw, self.citations) + + def peek(self, offset: int = 0) -> str | None: + index = self.i + offset + return self.lines[index] if index < len(self.lines) else None + + def add_text(self, size: str, content: str, scroll_to: int | None = None) -> None: + if not content: + return + view: dict = {"type": "text", "size": size, "content": content} + if scroll_to is not None: + view["scrollToId"] = scroll_to + self.views.append(view) + + def add_widget(self, sub_type: str, **payload) -> None: + self.views.append({"type": "widget", "sub_type": sub_type, **payload}) + + def add_section_heading(self, text: str, *, in_toc: bool) -> None: + """A top-level section heading, optionally registered in the page TOC.""" + if not in_toc: + self.add_text(SIZE_HEADING, text) + return + self.toc.append(text) + self.add_text(SIZE_HEADING, text, scroll_to=self.scroll_id) + self.scroll_id += 1 + + # -- entry point ------------------------------------------------------ # + + def parse(self) -> tuple[list[dict], list[str], list[str]]: + while self.i < len(self.lines): + line = self.lines[self.i] + + if not line.strip(): + self.i += 1 + elif ATTR_RE.match(line): + self.handle_attribute(line) + elif HEADING_RE.match(line): + self.handle_heading(HEADING_RE.match(line)) + elif HR_RE.match(line) and not TABLE_ROW_RE.match(line): + self.i += 1 + elif FENCE_RE.match(line): + self.handle_fence(FENCE_RE.match(line).group(1)) + elif TABLE_ROW_RE.match(line): + self.handle_table() + elif LIST_ITEM_RE.match(line): + self.handle_list() + elif LIQUID_RE.match(line): + self.handle_liquid(LIQUID_RE.match(line).group(1)) + elif " None: + """kramdown block attributes: `{: .no_toc}`, `{:toc}`, `{:.quiz}`.""" + if "quiz" in line: + self.next_is_quiz = True + # Several pages park the quiz below "## References". The quiz is real + # content, so it has to survive that dropped section. + self.skipping_section = False + if ":toc}" in line.replace(" ", ""): + self.wants_toc = True + self.i += 1 + + def handle_heading(self, match: re.Match) -> None: + level, raw = len(match.group(1)), match.group(2) + self.i += 1 + + no_toc = False + while ATTR_RE.match(self.peek() or ""): + no_toc = no_toc or "no_toc" in self.lines[self.i] + self.handle_attribute(self.lines[self.i]) + + text = self.text(raw) + key = text.lower().rstrip(":") + + if level == 1 and not self.seen_title: + # The page's own title. Section pages carry it in "name" only; chapter + # index pages and the standalone pages render the long form as well. + self.seen_title = True + self.skipping_section = False + if self.keep_title: + self.add_text(SIZE_HEADING, text) + return + + if key == "references": + # Emitted only once {% bibliography %} yields entries, so a page that + # cites nothing does not get an empty References heading. + self.pending_references = text + self.skipping_section = True + return + + if key in DROPPED_SECTIONS: + self.skipping_section = True + return + + if key in DROPPED_HEADINGS: + self.skipping_section = False + return + + self.skipping_section = False + + # A repeated level-1 heading is a second top-level section, not a + # sub-heading: treat it exactly like "##". + if level <= 2: + self.add_section_heading(text, in_toc=not no_toc) + else: + self.add_text(SIZE_SUBHEADING, text) + + def handle_paragraph(self) -> None: + chunk: list[str] = [] + while self.i < len(self.lines): + line = self.lines[self.i] + if ( + not line.strip() + or HEADING_RE.match(line) + or ATTR_RE.match(line) + or FENCE_RE.match(line) + or TABLE_ROW_RE.match(line) + or LIST_ITEM_RE.match(line) + or LIQUID_RE.match(line) + or HR_RE.match(line) + ): + break + chunk.append(line.strip()) + self.i += 1 + + if self.skipping_section: + return + self.add_text(SIZE_BODY, self.text(" ".join(chunk))) + + def handle_fence(self, language: str) -> None: + self.i += 1 + code: list[str] = [] + while self.i < len(self.lines) and not FENCE_RE.match(self.lines[self.i]): + code.append(self.lines[self.i]) + self.i += 1 + self.i += 1 # closing fence + + if self.skipping_section: + return + body = html.unescape("\n".join(code).rstrip()) + if not body: + return + label = "" if language.lower() in UNLABELLED_FENCES else f"{language.title()}:\n" + self.add_widget("clipboard", content=f"{label}{body}") + + def handle_table(self) -> None: + rows: list[list[str]] = [] + while self.i < len(self.lines) and TABLE_ROW_RE.match(self.lines[self.i]): + line = self.lines[self.i] + self.i += 1 + if TABLE_SEP_RE.match(line) and set(line) <= set(" |:-"): + continue + cells = [self.text(c) for c in line.strip().strip("|").split("|")] + rows.append(cells) + + if self.skipping_section or not rows: + return + heading, *body = rows + self.add_widget("table", content={"heading": heading, "rows": body}) + + def handle_list(self) -> None: + items = self.collect_list() + if self.skipping_section: + return + if not items: + return + + if self.next_is_quiz: + self.next_is_quiz = False + self.add_widget("pop-quiz", content=self.build_quiz(items)) + return + + # A bare "1. TOC" placeholder belongs to the dropped TOC section. + if len(items) == 1 and items[0]["text"].strip().upper() == "TOC": + return + + if items[0]["ordered"]: + self.add_widget("numbered_list", items=[self.build_item(i) for i in items]) + else: + self.add_widget("bullet_list", items=[i["text"] for i in items]) + + def collect_list(self) -> list[dict]: + """Collect one list, nesting children by indentation.""" + stack: list[tuple[int, list[dict]]] = [] + root: list[dict] = [] + current: list[dict] = root + base_indent: int | None = None + + while self.i < len(self.lines): + line = self.lines[self.i] + match = LIST_ITEM_RE.match(line) + + if match is None: + if not line.strip(): + # A blank line only ends the list if no list item follows. + lookahead = self.i + 1 + while lookahead < len(self.lines) and not self.lines[lookahead].strip(): + lookahead += 1 + if lookahead < len(self.lines) and LIST_ITEM_RE.match(self.lines[lookahead]): + self.i = lookahead + continue + break + + indent = len(match.group(1).expandtabs(4)) + ordered = match.group(3) is not None + text = self.text(match.group(4)) + self.i += 1 + + if base_indent is None: + base_indent = indent + + if indent > base_indent: + parent = current[-1] if current else None + if parent is not None: + stack.append((base_indent, current)) + current = parent["children"] + base_indent = indent + else: + while indent < base_indent and stack: + base_indent, current = stack.pop() + + current.append({"text": text, "ordered": ordered, "children": []}) + + return root + + def build_item(self, item: dict) -> dict: + entry: dict = {"content": item["text"]} + if item["children"]: + entry["children"] = [c["text"] for c in item["children"]] + return entry + + def build_quiz(self, items: list[dict]) -> list[dict]: + """`{:.quiz}` lists: numbered children are answers, bulleted ones are not.""" + questions = [] + for item in items: + options = [ + {"option": child["text"], "isAnswer": child["ordered"]} + for child in item["children"] + ] + if not options: + self.warnings.append(f"quiz question without options: {item['text']!r}") + continue + if not any(o["isAnswer"] for o in options): + self.warnings.append(f"quiz question without an answer: {item['text']!r}") + questions.append({"question": item["text"], "options": options}) + return questions + + def render_bibliography(self) -> None: + """{% bibliography --cited %} -> the References heading and its list.""" + entries = self.citations.rendered() + for key in self.citations.missing: + self.warnings.append(f"cited key not found in _bibliography: {key!r}") + if not entries: + return + self.skipping_section = False + if self.pending_references is not None: + self.add_section_heading(self.pending_references, in_toc=True) + self.pending_references = None + self.add_widget("numbered_list", items=[{"content": e} for e in entries]) + + def handle_liquid(self, tag: str) -> None: + self.i += 1 + + if tag.startswith("bibliography"): + self.render_bibliography() + return + + include = INCLUDE_RE.match(tag) + if include is None: + if self.skipping_section: + return + self.warnings.append(f"unhandled liquid tag: {{% {tag} %}}") + return + + name, rest = include.group(1), include.group(2) + # Structural includes describe the document's shape. One chapter files its + # chapter_toc under "## Table of contents" instead of "## Chapter contents", + # so it must not be dropped along with that section. + if self.skipping_section and name not in STRUCTURAL_INCLUDES: + return + + args = dict(INCLUDE_ARG_RE.findall(rest)) + + if name == "image.html": + content = {"link": args.get("url", "")} + if args.get("description"): + content["description"] = args["description"] + self.add_widget("image", content=content) + elif name == "chapter_toc.html": + self.add_widget("chapter_contents", items=[]) # filled in by the caller + elif name in INCLUDE_WIDGETS: + self.add_widget(INCLUDE_WIDGETS[name]) + else: + self.warnings.append(f"unhandled include: {name}") + + def handle_iframe(self) -> None: + block: list[str] = [] + while self.i < len(self.lines): + block.append(self.lines[self.i]) + self.i += 1 + if "" in block[-1].lower() or "/>" in block[-1]: + break + + if self.skipping_section: + return + match = IFRAME_SRC_RE.search("\n".join(block)) + if match is None: + self.warnings.append("iframe without a src attribute") + return + self.add_widget("image", content={"link": match.group(1)}) diff --git a/utils/md2json/book.py b/utils/md2json/book.py new file mode 100644 index 00000000..b665ef88 --- /dev/null +++ b/utils/md2json/book.py @@ -0,0 +1,80 @@ +"""Discovery and assembly: docs/ on disk -> Page / Chapter objects.""" + +from __future__ import annotations + +from pathlib import Path + +from .blocks import Parser +from .config import LEVEL_LABELS, LEVEL_ORDER +from .frontmatter import parse_front_matter +from .model import Chapter, Page, Section + + +def build_page(path: Path, *, keep_title: bool = False, bibliography=None) -> Page: + front, body = parse_front_matter(path.read_text(encoding="utf-8")) + parser = Parser(body, keep_title=keep_title, bibliography=bibliography) + views, _toc, warnings = parser.parse() + return Page(name=front.get("title", path.stem), views=views, warnings=warnings) + + +def chapter_contents(chapter: Chapter) -> list[dict]: + """Group a chapter's sections by difficulty level for the index page.""" + grouped: dict[str, list[dict]] = {level: [] for level in LEVEL_ORDER} + for number, section in enumerate(chapter.sections, start=1): + grouped.setdefault(section.level, []).append( + {"id": number, "name": section.title} + ) + return [ + {"category": LEVEL_LABELS.get(level, level.title()), "content": entries} + for level in LEVEL_ORDER + if (entries := grouped.get(level)) + ] + + +def discover_chapters(docs: Path) -> list[Chapter]: + chapters: list[Chapter] = [] + for index_path in sorted(docs.glob("*/index.md")): + front, _ = parse_front_matter(index_path.read_text(encoding="utf-8")) + sections: list[Section] = [] + for md in index_path.parent.glob("*.md"): + if md.name == "index.md": + continue + meta, _ = parse_front_matter(md.read_text(encoding="utf-8")) + if meta.get("published", "true").lower() == "false": + continue # placeholder page, not part of the book yet + sections.append( + Section( + path=md, + title=meta.get("title", md.stem), + level=meta.get("cvib_level", "basic").lower(), + nav_order=meta.get("nav_order", "l9s999"), + ) + ) + sections.sort(key=lambda s: (s.nav_order, s.title)) + chapters.append( + Chapter( + directory=index_path.parent.name, + index_path=index_path, + title=front.get("title", index_path.parent.name), + nav_order=int(front.get("nav_order", "0") or 0), + sections=sections, + ) + ) + chapters.sort(key=lambda c: (c.nav_order, c.directory)) + return chapters + + +def build_navbar(chapters: list[Chapter]) -> dict: + return { + "chapters": [ + { + "id": chapter_id, + "name": chapter.title, + "sub-chapters": [ + {"id": number, "name": section.title} + for number, section in enumerate(chapter.sections, start=1) + ], + } + for chapter_id, chapter in enumerate(chapters, start=1) + ] + } diff --git a/utils/md2json/cli.py b/utils/md2json/cli.py new file mode 100644 index 00000000..8bc2bfbb --- /dev/null +++ b/utils/md2json/cli.py @@ -0,0 +1,68 @@ +"""Entry point: regenerate the whole output directory in one shot.""" + +from __future__ import annotations + +from pathlib import Path + +from .bibliography import load_bibliography +from .book import build_navbar, build_page, chapter_contents, discover_chapters +from .config import DOCS_PATH, OUTPUT_PATH, REPO_ROOT, ROOT_PAGES, resolve_root_page +from .output import write_document + + +def main() -> int: + if not DOCS_PATH.is_dir(): + print(f"docs directory not found: {DOCS_PATH}") + return 1 + + bibliography = load_bibliography() + chapters = discover_chapters(DOCS_PATH) + written = 0 + warnings: list[tuple[Path, str]] = [] + + # Standalone pages that live at the root of the output directory. + for source, target in ROOT_PAGES.items(): + path = resolve_root_page(DOCS_PATH, source) + if path is None: + print(f"[miss] {source} not found; {target} not generated") + continue + page = build_page(path, keep_title=True, bibliography=bibliography) + warnings += [(path, w) for w in page.warnings] + write_document(OUTPUT_PATH / target, {"name": page.name, "views": page.views}) + written += 1 + + write_document(OUTPUT_PATH / "navbar.json", build_navbar(chapters)) + written += 1 + + for chapter in chapters: + index = build_page(chapter.index_path, keep_title=True, bibliography=bibliography) + warnings += [(chapter.index_path, w) for w in index.warnings] + for view in index.views: + if view.get("sub_type") == "chapter_contents": + view["items"] = chapter_contents(chapter) + write_document( + OUTPUT_PATH / chapter.directory / "0.json", + {"name": index.name, "views": index.views}, + ) + written += 1 + + for number, section in enumerate(chapter.sections, start=1): + page = build_page(section.path, bibliography=bibliography) + warnings += [(section.path, w) for w in page.warnings] + write_document( + OUTPUT_PATH / chapter.directory / f"{number}.json", + {"name": page.name, "views": page.views}, + ) + written += 1 + + if warnings: + print("\nwarnings:") + for path, warning in warnings: + try: + shown = path.relative_to(REPO_ROOT) + except ValueError: + shown = path + print(f" {shown}: {warning}") + + print(f"\n{written} files written to {OUTPUT_PATH.relative_to(REPO_ROOT)}/") + return 0 diff --git a/utils/md2json/config.py b/utils/md2json/config.py new file mode 100644 index 00000000..d63e3ad0 --- /dev/null +++ b/utils/md2json/config.py @@ -0,0 +1,76 @@ +"""Tunables shared by the docs -> chapters translation. + +Everything in here is data: which files to read, which Jekyll constructs map to +which frontend widget, and which headings are plumbing rather than content. +""" + +from __future__ import annotations + +from pathlib import Path + +# utils/md2json/config.py -> utils/md2json -> utils -> repo root +REPO_ROOT = Path(__file__).resolve().parents[2] +DOCS_PATH = REPO_ROOT / "docs" +# The generated tree. Regenerated in full on every run. +OUTPUT_PATH = REPO_ROOT / "TestJSON" +# BibTeX sources behind {% cite %} / {% bibliography %}. +BIBLIOGRAPHY_PATH = REPO_ROOT / "_bibliography" + +# Standalone Jekyll pages that become root-level documents. These live at the +# repo root rather than under docs/: `about.md` and `CONTRIBUTING.md` both carry +# Jekyll front matter (`title: About` / `title: Guidelines`) and neither is in +# the `_config.yml` exclude list, so Jekyll serves them alongside the book. +ROOT_PAGES = { + "about.md": "about.json", + "CONTRIBUTING.md": "guidelines.json", +} + +# cvib_level -> category label used by the chapter_contents widget. +LEVEL_LABELS = { + "basic": "Basic Level", + "medium": "Medium Level", + "advanced": "Advanced Level", +} +LEVEL_ORDER = ["basic", "medium", "advanced"] + +# {% include .html %} -> interactive widget sub_type. Includes that are not +# listed here are dropped with a warning; add them as the frontend gains widgets. +INCLUDE_WIDGETS = { + "binary.html": "binary-simulator", + "bool.html": "boolean-simulator", + "fsm.html": "fsm-simulator", + "gates.html": "gates-simulator", + "kmap.html": "kmap-simulator", + "truth_table.html": "truth-table-simulator", + "application2.html": "character_representation", + "application1.html": "subject_encoder", + "binary2.html": "bitwise-simulator", + "flipflop2.html": "flipflop-simulator", +} + +# Structural includes that carry document shape rather than page content, so they +# survive a dropped section (see DROPPED_SECTIONS). +STRUCTURAL_INCLUDES = {"chapter_toc.html"} + +# Headings that carry Jekyll plumbing rather than content. The heading and +# everything under it is dropped (the "1. TOC" placeholder). +DROPPED_SECTIONS = {"table of contents"} +# Headings whose body is real content but whose title is plumbing. +DROPPED_HEADINGS = {"chapter contents"} + +# Code fence languages that are content rather than source code: no label line. +UNLABELLED_FENCES = {"", "text", "txt", "yaml", "yml", "markdown", "md"} + +# Text sizes. The frontend renders H1 as a section heading, H2 as a sub-heading +# and H3 as body copy. +SIZE_HEADING = "H1" +SIZE_SUBHEADING = "H2" +SIZE_BODY = "H3" + + +def resolve_root_page(docs: Path, source: str) -> Path | None: + """Find a root page, preferring the docs tree and falling back to the repo.""" + for candidate in (docs / source, REPO_ROOT / source): + if candidate.is_file(): + return candidate + return None diff --git a/utils/md2json/frontmatter.py b/utils/md2json/frontmatter.py new file mode 100644 index 00000000..8619d9d1 --- /dev/null +++ b/utils/md2json/frontmatter.py @@ -0,0 +1,22 @@ +"""Jekyll front matter parsing.""" + +from __future__ import annotations + +import re + +FRONT_MATTER_RE = re.compile(r"\A---\n(.*?)\n---\n?", re.S) + + +def parse_front_matter(text: str) -> tuple[dict[str, str], str]: + """Split a page into its (flat) front matter mapping and its body.""" + match = FRONT_MATTER_RE.match(text) + if match is None: + return {}, text + + data: dict[str, str] = {} + for line in match.group(1).split("\n"): + if ":" not in line or line.lstrip().startswith("#"): + continue + key, value = line.split(":", 1) + data[key.strip()] = value.strip().strip("\"'") + return data, text[match.end():] diff --git a/utils/md2json/inline.py b/utils/md2json/inline.py new file mode 100644 index 00000000..80e48007 --- /dev/null +++ b/utils/md2json/inline.py @@ -0,0 +1,52 @@ +"""Inline markdown/HTML -> the plain text the frontend widgets expect.""" + +from __future__ import annotations + +import html +import re + +CITE_RE = re.compile(r"\{%\s*cite\s+(.*?)\s*%\}") +IMAGE_LINK_RE = re.compile(r"!\[([^\]]*)\]\(([^)]+)\)") +LINK_RE = re.compile(r"\[([^\]]+)\]\(([^)]+)\)") +AUTOLINK_RE = re.compile(r"<((?:https?|mailto):[^>]+)>") +SUP_RE = re.compile(r"(.*?)", re.S | re.I) +SUB_RE = re.compile(r"(.*?)", re.S | re.I) +TAG_RE = re.compile(r"]*>") +BOLD_ITALIC_RE = re.compile(r"(\*{1,3}|_{1,3})(\S.*?\S|\S)\1", re.S) +CODE_RE = re.compile(r"`([^`]+)`") + + +def cite_keys(argument: str) -> list[str]: + """The keys in a {% cite a b --file books %} tag, ignoring its flags.""" + keys = [] + for token in argument.split(): + if token.startswith("--"): + break + keys.append(token) + return keys + + +def inline_text(text: str, citations=None) -> str: + """Flatten inline markdown/HTML to plain text. + + With a CitationRegistry, {% cite %} becomes an IEEE marker such as "[1]"; + without one it is dropped, which is what non-prose contexts want. + """ + if citations is None: + text = CITE_RE.sub("", text) + else: + text = CITE_RE.sub(lambda m: citations.mark(cite_keys(m.group(1))), text) + text = IMAGE_LINK_RE.sub(lambda m: m.group(1) or "", text) + # In-page anchors have no meaning outside the Jekyll build: keep the label only. + text = LINK_RE.sub( + lambda m: m.group(1) if m.group(2).startswith("#") else f"{m.group(1)} ({m.group(2)})", + text, + ) + text = AUTOLINK_RE.sub(lambda m: m.group(1), text) + text = SUP_RE.sub(lambda m: f"^{m.group(1)}", text) + text = SUB_RE.sub(lambda m: f"_{m.group(1)}", text) + text = TAG_RE.sub("", text) + text = CODE_RE.sub(lambda m: m.group(1), text) + text = BOLD_ITALIC_RE.sub(lambda m: m.group(2), text) + text = html.unescape(text) + return re.sub(r"\s+", " ", text).strip() diff --git a/utils/md2json/model.py b/utils/md2json/model.py new file mode 100644 index 00000000..39966b6b --- /dev/null +++ b/utils/md2json/model.py @@ -0,0 +1,32 @@ +"""Dataclasses describing a parsed page and the book's structure.""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from pathlib import Path + + +@dataclass +class Page: + """A parsed markdown page, ready to be rendered as a view document.""" + + name: str + views: list[dict] = field(default_factory=list) + warnings: list[str] = field(default_factory=list) + + +@dataclass +class Section: + path: Path + title: str + level: str + nav_order: str + + +@dataclass +class Chapter: + directory: str + index_path: Path + title: str + nav_order: int + sections: list[Section] diff --git a/utils/md2json/output.py b/utils/md2json/output.py new file mode 100644 index 00000000..ad0c2977 --- /dev/null +++ b/utils/md2json/output.py @@ -0,0 +1,19 @@ +"""Writing view documents to disk.""" + +from __future__ import annotations + +import json +from pathlib import Path + +from .config import OUTPUT_PATH + + +def dumps(document: dict) -> str: + return json.dumps(document, indent=4, ensure_ascii=False) + "\n" + + +def write_document(path: Path, document: dict) -> None: + """Write one document, overwriting whatever was there.""" + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(dumps(document), encoding="utf-8") + print(f"[write] {path.relative_to(OUTPUT_PATH)}") From 1c2430703a3d2c6caaabc4389a600c6c1354cbc7 Mon Sep 17 00:00:00 2001 From: SantamRC Date: Sat, 29 Aug 2026 00:38:43 +0530 Subject: [PATCH 2/9] refactor: centralise widget names and harden repo-root lookup The generator's widget vocabulary was split across two files: INCLUDE_WIDGETS lived in config.py while nine sub_type strings were literals in blocks.py, so "what does the app render?" could not be answered from one place and renaming a widget meant grepping the parser. - move the nine generator-named sub_types into config.py as WIDGET_* constants - name image.html / chapter_toc.html once in config; both were duplicated as literals in blocks.py, with chapter_toc.html already in STRUCTURAL_INCLUDES - move the "references" heading into config alongside the other heading rules - locate the repo by walking up to _config.yml instead of parents[2], which silently resolved to the wrong tree if the package were moved Also fixes a quiz in docs/logic-design/kmaps.md that mixed tab and 8-space indentation in one list. A tab expands to 4 columns and the spaces to 8, so three options parsed as shallower than their siblings and were lifted to the top level, producing a question with no correct answer plus a phantom question titled "3-variable", and dropping an option from an earlier question. Whitespace only, no wording changed; kramdown resolves indentation the same way, so this should correct the website rendering too. Generated output is byte-for-byte unchanged, verified against a snapshot. --- .gitignore | 6 ++++++ docs/logic-design/kmaps.md | 4 ++-- utils/md2json/blocks.py | 37 ++++++++++++++++++++++++------------- utils/md2json/cli.py | 11 +++++++++-- utils/md2json/config.py | 35 +++++++++++++++++++++++++++++++---- 5 files changed, 72 insertions(+), 21 deletions(-) diff --git a/.gitignore b/.gitignore index 4719f95c..fcddb672 100644 --- a/.gitignore +++ b/.gitignore @@ -7,3 +7,9 @@ out *~ /.idea + +/chapters/ +/chapters2/ +/TestJSON/ +__pycache__/ +*.py[co] diff --git a/docs/logic-design/kmaps.md b/docs/logic-design/kmaps.md index df401126..9f1c5d9a 100644 --- a/docs/logic-design/kmaps.md +++ b/docs/logic-design/kmaps.md @@ -193,7 +193,7 @@ This illustrates the idea that this is a greedy algorithm, and does not always r * POS * SOP 1. Entries - * Latches + * Latches 2. K-map can be used to minimize functions of up to ___ variables ? * 5 @@ -202,7 +202,7 @@ This illustrates the idea that this is a greedy algorithm, and does not always r * 3 3. In which K-map 16 cells are there ? - * 2-variable + * 2-variable * 3-variable 1. 4-variable * 5-variable diff --git a/utils/md2json/blocks.py b/utils/md2json/blocks.py index 413a9dc9..e40f5235 100644 --- a/utils/md2json/blocks.py +++ b/utils/md2json/blocks.py @@ -7,14 +7,25 @@ from .bibliography import CitationRegistry from .config import ( + CHAPTER_TOC_INCLUDE, DROPPED_HEADINGS, DROPPED_SECTIONS, + IMAGE_INCLUDE, INCLUDE_WIDGETS, + REFERENCES_HEADING, SIZE_BODY, SIZE_HEADING, SIZE_SUBHEADING, STRUCTURAL_INCLUDES, UNLABELLED_FENCES, + WIDGET_BULLET_LIST, + WIDGET_CHAPTER_CONTENTS, + WIDGET_CLIPBOARD, + WIDGET_IMAGE, + WIDGET_NUMBERED_LIST, + WIDGET_QUIZ, + WIDGET_TABLE, + WIDGET_TOC, ) from .inline import inline_text @@ -105,7 +116,7 @@ def parse(self) -> tuple[list[dict], list[str], list[str]]: self.handle_paragraph() if self.wants_toc and self.toc: - self.views.insert(0, {"type": "widget", "sub_type": "toc", "items": self.toc}) + self.views.insert(0, {"type": "widget", "sub_type": WIDGET_TOC, "items": self.toc}) return self.views, self.toc, self.warnings # -- block handlers --------------------------------------------------- # @@ -142,7 +153,7 @@ def handle_heading(self, match: re.Match) -> None: self.add_text(SIZE_HEADING, text) return - if key == "references": + if key == REFERENCES_HEADING: # Emitted only once {% bibliography %} yields entries, so a page that # cites nothing does not get an empty References heading. self.pending_references = text @@ -202,7 +213,7 @@ def handle_fence(self, language: str) -> None: if not body: return label = "" if language.lower() in UNLABELLED_FENCES else f"{language.title()}:\n" - self.add_widget("clipboard", content=f"{label}{body}") + self.add_widget(WIDGET_CLIPBOARD, content=f"{label}{body}") def handle_table(self) -> None: rows: list[list[str]] = [] @@ -217,7 +228,7 @@ def handle_table(self) -> None: if self.skipping_section or not rows: return heading, *body = rows - self.add_widget("table", content={"heading": heading, "rows": body}) + self.add_widget(WIDGET_TABLE, content={"heading": heading, "rows": body}) def handle_list(self) -> None: items = self.collect_list() @@ -228,7 +239,7 @@ def handle_list(self) -> None: if self.next_is_quiz: self.next_is_quiz = False - self.add_widget("pop-quiz", content=self.build_quiz(items)) + self.add_widget(WIDGET_QUIZ, content=self.build_quiz(items)) return # A bare "1. TOC" placeholder belongs to the dropped TOC section. @@ -236,9 +247,9 @@ def handle_list(self) -> None: return if items[0]["ordered"]: - self.add_widget("numbered_list", items=[self.build_item(i) for i in items]) + self.add_widget(WIDGET_NUMBERED_LIST, items=[self.build_item(i) for i in items]) else: - self.add_widget("bullet_list", items=[i["text"] for i in items]) + self.add_widget(WIDGET_BULLET_LIST, items=[i["text"] for i in items]) def collect_list(self) -> list[dict]: """Collect one list, nesting children by indentation.""" @@ -317,7 +328,7 @@ def render_bibliography(self) -> None: if self.pending_references is not None: self.add_section_heading(self.pending_references, in_toc=True) self.pending_references = None - self.add_widget("numbered_list", items=[{"content": e} for e in entries]) + self.add_widget(WIDGET_NUMBERED_LIST, items=[{"content": e} for e in entries]) def handle_liquid(self, tag: str) -> None: self.i += 1 @@ -342,13 +353,13 @@ def handle_liquid(self, tag: str) -> None: args = dict(INCLUDE_ARG_RE.findall(rest)) - if name == "image.html": + if name == IMAGE_INCLUDE: content = {"link": args.get("url", "")} if args.get("description"): content["description"] = args["description"] - self.add_widget("image", content=content) - elif name == "chapter_toc.html": - self.add_widget("chapter_contents", items=[]) # filled in by the caller + self.add_widget(WIDGET_IMAGE, content=content) + elif name == CHAPTER_TOC_INCLUDE: + self.add_widget(WIDGET_CHAPTER_CONTENTS, items=[]) # filled in by the caller elif name in INCLUDE_WIDGETS: self.add_widget(INCLUDE_WIDGETS[name]) else: @@ -368,4 +379,4 @@ def handle_iframe(self) -> None: if match is None: self.warnings.append("iframe without a src attribute") return - self.add_widget("image", content={"link": match.group(1)}) + self.add_widget(WIDGET_IMAGE, content={"link": match.group(1)}) diff --git a/utils/md2json/cli.py b/utils/md2json/cli.py index 8bc2bfbb..54d12d76 100644 --- a/utils/md2json/cli.py +++ b/utils/md2json/cli.py @@ -6,7 +6,14 @@ from .bibliography import load_bibliography from .book import build_navbar, build_page, chapter_contents, discover_chapters -from .config import DOCS_PATH, OUTPUT_PATH, REPO_ROOT, ROOT_PAGES, resolve_root_page +from .config import ( + DOCS_PATH, + OUTPUT_PATH, + REPO_ROOT, + ROOT_PAGES, + WIDGET_CHAPTER_CONTENTS, + resolve_root_page, +) from .output import write_document @@ -38,7 +45,7 @@ def main() -> int: index = build_page(chapter.index_path, keep_title=True, bibliography=bibliography) warnings += [(chapter.index_path, w) for w in index.warnings] for view in index.views: - if view.get("sub_type") == "chapter_contents": + if view.get("sub_type") == WIDGET_CHAPTER_CONTENTS: view["items"] = chapter_contents(chapter) write_document( OUTPUT_PATH / chapter.directory / "0.json", diff --git a/utils/md2json/config.py b/utils/md2json/config.py index d63e3ad0..fd7292b2 100644 --- a/utils/md2json/config.py +++ b/utils/md2json/config.py @@ -8,8 +8,15 @@ from pathlib import Path -# utils/md2json/config.py -> utils/md2json -> utils -> repo root -REPO_ROOT = Path(__file__).resolve().parents[2] +def _find_repo_root(start: Path) -> Path: + """Walk up to the directory holding _config.yml, so the package can move.""" + for candidate in (start, *start.parents): + if (candidate / "_config.yml").is_file(): + return candidate + return start.parents[2] # utils/md2json/config.py -> repo root + + +REPO_ROOT = _find_repo_root(Path(__file__).resolve().parent) DOCS_PATH = REPO_ROOT / "docs" # The generated tree. Regenerated in full on every run. OUTPUT_PATH = REPO_ROOT / "TestJSON" @@ -48,9 +55,13 @@ "flipflop2.html": "flipflop-simulator", } -# Structural includes that carry document shape rather than page content, so they +# Includes handled specially rather than as a plain widget: they carry arguments +# or document structure. Named here so no module repeats the filenames. +IMAGE_INCLUDE = "image.html" +CHAPTER_TOC_INCLUDE = "chapter_toc.html" +# Structural includes carry document shape rather than page content, so they # survive a dropped section (see DROPPED_SECTIONS). -STRUCTURAL_INCLUDES = {"chapter_toc.html"} +STRUCTURAL_INCLUDES = {CHAPTER_TOC_INCLUDE} # Headings that carry Jekyll plumbing rather than content. The heading and # everything under it is dropped (the "1. TOC" placeholder). @@ -61,6 +72,22 @@ # Code fence languages that are content rather than source code: no label line. UNLABELLED_FENCES = {"", "text", "txt", "yaml", "yml", "markdown", "md"} +# Widget sub_types the generator names itself. Unlike INCLUDE_WIDGETS these have +# no token in the markdown to derive from -- a table is just "| a | b |" -- so +# they are the translator's own vocabulary. This is the frontend contract: +# renaming one here renames it in the JSON the app consumes. +WIDGET_TOC = "toc" +WIDGET_TABLE = "table" +WIDGET_BULLET_LIST = "bullet_list" +WIDGET_NUMBERED_LIST = "numbered_list" +WIDGET_CLIPBOARD = "clipboard" +WIDGET_IMAGE = "image" +WIDGET_QUIZ = "pop-quiz" +WIDGET_CHAPTER_CONTENTS = "chapter_contents" + +# Heading whose body is a bibliography rather than prose. +REFERENCES_HEADING = "references" + # Text sizes. The frontend renders H1 as a section heading, H2 as a sub-heading # and H3 as body copy. SIZE_HEADING = "H1" From a4aa66e56feff8f919930f76cf6a6cac0441c133 Mon Sep 17 00:00:00 2001 From: SantamRC Date: Sat, 29 Aug 2026 03:35:28 +0530 Subject: [PATCH 3/9] fix: address review feedback on the markdown to JSON generator Recovers content the parser was dropping, and makes the output tree safe to regenerate. handle_paragraph() did not treat an iframe as a paragraph terminator, so an iframe that followed prose without a blank line was absorbed into the paragraph and then stripped as an HTML tag. Twelve embedded CircuitVerse simulators were missing from the generated JSON as a result; they are now emitted. collect_list() accepted ordered and unordered markers at the same root level, so a bullet list followed by a numbered list merged into one widget typed from the first item. Collection now stops when a root-level marker type changes. Nested items still mix markers freely, which is how {:.quiz} encodes answers. CitationRegistry numbered keys it could not resolve but rendered() omitted them, so an inline [2] could point at the first entry in the list. Unresolved keys now render a visible placeholder, keeping markers and list positions aligned. The output tree is generator-owned: it is now built in a staging directory and swapped in only after a successful run, so a removed or renumbered section cannot leave stale JSON and a failure part-way through cannot leave a partial tree. Also: - _find_repo_root() fell back to parents[2], one level above the repository, since it is passed utils/md2json rather than the config file itself - write_document() raised ValueError labelling any path outside OUTPUT_PATH - dropped the "1. TOC" placeholder guard, which never fired on any page because the dropped-section handling already removes it, and could only misfire on a legitimate single-item list - fixed a docstring opening with four quote characters - documented the remaining undocumented functions (45/45) Verified against a pre-change snapshot: the only content differences are the twelve recovered iframe widgets. --- utils/md2json/bibliography.py | 14 ++++-- utils/md2json/blocks.py | 29 ++++++++++--- utils/md2json/book.py | 3 ++ utils/md2json/cli.py | 80 +++++++++++++++++++++++------------ utils/md2json/config.py | 2 +- utils/md2json/output.py | 13 +++++- 6 files changed, 103 insertions(+), 38 deletions(-) diff --git a/utils/md2json/bibliography.py b/utils/md2json/bibliography.py index 4b382339..50c85029 100644 --- a/utils/md2json/bibliography.py +++ b/utils/md2json/bibliography.py @@ -36,6 +36,7 @@ def _read_braced(text: str, start: int) -> tuple[str, int]: def _clean(value: str) -> str: + """Strip LaTeX escapes, grouping braces and stray whitespace from a value.""" for escape, plain in LATEX_ESCAPES.items(): value = value.replace(escape, plain) value = value.replace("{", "").replace("}", "") @@ -43,6 +44,7 @@ def _clean(value: str) -> str: def _read_value(text: str, i: int) -> tuple[str, int]: + """Read one field value, braced, quoted or bare, from position `i`.""" while i < len(text) and text[i].isspace(): i += 1 if i < len(text) and text[i] == "{": @@ -94,7 +96,7 @@ def load_bibliography() -> dict[str, dict[str, str]]: def _format_author(author: str) -> str: - """"Donzellini, G. and Oneto, L." -> "G. Donzellini, L. Oneto".""" + """Reorder BibTeX names: "Donzellini, G. and Oneto, L." -> "G. Donzellini, L. Oneto".""" names = [n.strip() for n in re.split(r"\s+and\s+", author) if n.strip()] formatted = [] for name in names: @@ -162,6 +164,7 @@ class CitationRegistry: """Numbers the citations on one page, in order of first appearance.""" def __init__(self, entries: dict[str, dict[str, str]]): + """Start an empty registry backed by the merged BibTeX entries.""" self.entries = entries self.order: list[str] = [] self.missing: list[str] = [] @@ -178,9 +181,14 @@ def mark(self, keys: list[str]) -> str: return ", ".join(f"[{n}]" for n in numbers) def rendered(self) -> list[str]: - """The reference list, in citation order.""" + """The reference list, in citation order. + + Every marked key gets an entry, including keys with no BibTeX record, so + that an inline "[2]" always points at the second item in this list. + """ return [ format_entry(self.entries[key]) - for key in self.order if key in self.entries + else f"{key} (no entry found in _bibliography)" + for key in self.order ] diff --git a/utils/md2json/blocks.py b/utils/md2json/blocks.py index e40f5235..8dd94670 100644 --- a/utils/md2json/blocks.py +++ b/utils/md2json/blocks.py @@ -44,6 +44,7 @@ class Parser: def __init__(self, body: str, *, keep_title: bool = False, bibliography=None): + """Prepare to scan one page body, minus its front matter.""" self.lines = body.split("\n") self.i = 0 self.keep_title = keep_title @@ -65,10 +66,12 @@ def text(self, raw: str) -> str: return inline_text(raw, self.citations) def peek(self, offset: int = 0) -> str | None: + """The line `offset` ahead of the cursor, or None past the end.""" index = self.i + offset return self.lines[index] if index < len(self.lines) else None def add_text(self, size: str, content: str, scroll_to: int | None = None) -> None: + """Append a text view, skipping empties and tagging TOC anchors.""" if not content: return view: dict = {"type": "text", "size": size, "content": content} @@ -77,6 +80,7 @@ def add_text(self, size: str, content: str, scroll_to: int | None = None) -> Non self.views.append(view) def add_widget(self, sub_type: str, **payload) -> None: + """Append a widget view of `sub_type` carrying `payload`.""" self.views.append({"type": "widget", "sub_type": sub_type, **payload}) def add_section_heading(self, text: str, *, in_toc: bool) -> None: @@ -91,6 +95,7 @@ def add_section_heading(self, text: str, *, in_toc: bool) -> None: # -- entry point ------------------------------------------------------ # def parse(self) -> tuple[list[dict], list[str], list[str]]: + """Scan the body once, returning its views, TOC entries and warnings.""" while self.i < len(self.lines): line = self.lines[self.i] @@ -133,6 +138,7 @@ def handle_attribute(self, line: str) -> None: self.i += 1 def handle_heading(self, match: re.Match) -> None: + """Emit a heading, or route plumbing headings to their special cases.""" level, raw = len(match.group(1)), match.group(2) self.i += 1 @@ -178,6 +184,7 @@ def handle_heading(self, match: re.Match) -> None: self.add_text(SIZE_SUBHEADING, text) def handle_paragraph(self) -> None: + """Collect consecutive prose lines into one body-text view.""" chunk: list[str] = [] while self.i < len(self.lines): line = self.lines[self.i] @@ -190,6 +197,7 @@ def handle_paragraph(self) -> None: or LIST_ITEM_RE.match(line) or LIQUID_RE.match(line) or HR_RE.match(line) + or " None: self.add_text(SIZE_BODY, self.text(" ".join(chunk))) def handle_fence(self, language: str) -> None: + """Turn a fenced code block into a clipboard widget.""" self.i += 1 code: list[str] = [] while self.i < len(self.lines) and not FENCE_RE.match(self.lines[self.i]): @@ -216,6 +225,7 @@ def handle_fence(self, language: str) -> None: self.add_widget(WIDGET_CLIPBOARD, content=f"{label}{body}") def handle_table(self) -> None: + """Collect a pipe table into a heading row plus body rows.""" rows: list[list[str]] = [] while self.i < len(self.lines) and TABLE_ROW_RE.match(self.lines[self.i]): line = self.lines[self.i] @@ -231,6 +241,7 @@ def handle_table(self) -> None: self.add_widget(WIDGET_TABLE, content={"heading": heading, "rows": body}) def handle_list(self) -> None: + """Emit a list as a quiz, numbered list or bullet list.""" items = self.collect_list() if self.skipping_section: return @@ -242,10 +253,6 @@ def handle_list(self) -> None: self.add_widget(WIDGET_QUIZ, content=self.build_quiz(items)) return - # A bare "1. TOC" placeholder belongs to the dropped TOC section. - if len(items) == 1 and items[0]["text"].strip().upper() == "TOC": - return - if items[0]["ordered"]: self.add_widget(WIDGET_NUMBERED_LIST, items=[self.build_item(i) for i in items]) else: @@ -257,6 +264,7 @@ def collect_list(self) -> list[dict]: root: list[dict] = [] current: list[dict] = root base_indent: int | None = None + root_ordered: bool | None = None while self.i < len(self.lines): line = self.lines[self.i] @@ -275,11 +283,17 @@ def collect_list(self) -> list[dict]: indent = len(match.group(1).expandtabs(4)) ordered = match.group(3) is not None - text = self.text(match.group(4)) - self.i += 1 if base_indent is None: base_indent = indent + root_ordered = ordered + elif not stack and indent <= base_indent and ordered != root_ordered: + # "- a" then "1. b" are two lists; handle_list picks one widget + # type per call, so stop and let the next call own the second. + break + + text = self.text(match.group(4)) + self.i += 1 if indent > base_indent: parent = current[-1] if current else None @@ -296,6 +310,7 @@ def collect_list(self) -> list[dict]: return root def build_item(self, item: dict) -> dict: + """Shape one collected list item for a numbered_list widget.""" entry: dict = {"content": item["text"]} if item["children"]: entry["children"] = [c["text"] for c in item["children"]] @@ -331,6 +346,7 @@ def render_bibliography(self) -> None: self.add_widget(WIDGET_NUMBERED_LIST, items=[{"content": e} for e in entries]) def handle_liquid(self, tag: str) -> None: + """Translate a Liquid tag into its widget, or warn if unmapped.""" self.i += 1 if tag.startswith("bibliography"): @@ -366,6 +382,7 @@ def handle_liquid(self, tag: str) -> None: self.warnings.append(f"unhandled include: {name}") def handle_iframe(self) -> None: + """Turn an embedded iframe into an image widget carrying its src.""" block: list[str] = [] while self.i < len(self.lines): block.append(self.lines[self.i]) diff --git a/utils/md2json/book.py b/utils/md2json/book.py index b665ef88..391c4ba3 100644 --- a/utils/md2json/book.py +++ b/utils/md2json/book.py @@ -11,6 +11,7 @@ def build_page(path: Path, *, keep_title: bool = False, bibliography=None) -> Page: + """Parse one markdown file into a Page of view documents.""" front, body = parse_front_matter(path.read_text(encoding="utf-8")) parser = Parser(body, keep_title=keep_title, bibliography=bibliography) views, _toc, warnings = parser.parse() @@ -32,6 +33,7 @@ def chapter_contents(chapter: Chapter) -> list[dict]: def discover_chapters(docs: Path) -> list[Chapter]: + """Find every chapter in `docs`, ordering chapters and sections by nav_order.""" chapters: list[Chapter] = [] for index_path in sorted(docs.glob("*/index.md")): front, _ = parse_front_matter(index_path.read_text(encoding="utf-8")) @@ -65,6 +67,7 @@ def discover_chapters(docs: Path) -> list[Chapter]: def build_navbar(chapters: list[Chapter]) -> dict: + """Build the navigation document listing chapters and their sections.""" return { "chapters": [ { diff --git a/utils/md2json/cli.py b/utils/md2json/cli.py index 54d12d76..ecd7ea90 100644 --- a/utils/md2json/cli.py +++ b/utils/md2json/cli.py @@ -2,6 +2,8 @@ from __future__ import annotations +import shutil +import tempfile from pathlib import Path from .bibliography import load_bibliography @@ -17,16 +19,19 @@ from .output import write_document -def main() -> int: - if not DOCS_PATH.is_dir(): - print(f"docs directory not found: {DOCS_PATH}") - return 1 - +def generate(staging: Path) -> tuple[int, list[tuple[Path, str]]]: + """Write every document into `staging`, returning the count and warnings.""" bibliography = load_bibliography() chapters = discover_chapters(DOCS_PATH) written = 0 warnings: list[tuple[Path, str]] = [] + def emit(relative: str, document: dict) -> None: + """Write one document at `relative` inside the staging directory.""" + nonlocal written + write_document(staging / relative, document, staging) + written += 1 + # Standalone pages that live at the root of the output directory. for source, target in ROOT_PAGES.items(): path = resolve_root_page(DOCS_PATH, source) @@ -35,41 +40,64 @@ def main() -> int: continue page = build_page(path, keep_title=True, bibliography=bibliography) warnings += [(path, w) for w in page.warnings] - write_document(OUTPUT_PATH / target, {"name": page.name, "views": page.views}) - written += 1 + emit(target, {"name": page.name, "views": page.views}) - write_document(OUTPUT_PATH / "navbar.json", build_navbar(chapters)) - written += 1 + emit("navbar.json", build_navbar(chapters)) for chapter in chapters: - index = build_page(chapter.index_path, keep_title=True, bibliography=bibliography) + index = build_page( + chapter.index_path, keep_title=True, bibliography=bibliography + ) warnings += [(chapter.index_path, w) for w in index.warnings] for view in index.views: if view.get("sub_type") == WIDGET_CHAPTER_CONTENTS: view["items"] = chapter_contents(chapter) - write_document( - OUTPUT_PATH / chapter.directory / "0.json", - {"name": index.name, "views": index.views}, - ) - written += 1 + emit(f"{chapter.directory}/0.json", {"name": index.name, "views": index.views}) for number, section in enumerate(chapter.sections, start=1): page = build_page(section.path, bibliography=bibliography) warnings += [(section.path, w) for w in page.warnings] - write_document( - OUTPUT_PATH / chapter.directory / f"{number}.json", + emit( + f"{chapter.directory}/{number}.json", {"name": page.name, "views": page.views}, ) - written += 1 - if warnings: - print("\nwarnings:") - for path, warning in warnings: - try: - shown = path.relative_to(REPO_ROOT) - except ValueError: - shown = path - print(f" {shown}: {warning}") + return written, warnings + + +def report(warnings: list[tuple[Path, str]]) -> None: + """Print the parser warnings collected during a run.""" + if not warnings: + return + print("\nwarnings:") + for path, warning in warnings: + try: + shown = path.relative_to(REPO_ROOT) + except ValueError: + shown = path + print(f" {shown}: {warning}") + + +def main() -> int: + """Regenerate the output tree from docs/, replacing it only on success.""" + if not DOCS_PATH.is_dir(): + print(f"docs directory not found: {DOCS_PATH}") + return 1 + + # The output tree is generator-owned: build it beside the real one and swap, + # so a removed or renumbered section cannot leave stale JSON behind and a + # failure part-way through cannot leave a half-written tree. + OUTPUT_PATH.parent.mkdir(parents=True, exist_ok=True) + staging = Path(tempfile.mkdtemp(dir=OUTPUT_PATH.parent, prefix=".md2json-")) + try: + written, warnings = generate(staging) + if OUTPUT_PATH.exists(): + shutil.rmtree(OUTPUT_PATH) + staging.replace(OUTPUT_PATH) + except BaseException: + shutil.rmtree(staging, ignore_errors=True) + raise + report(warnings) print(f"\n{written} files written to {OUTPUT_PATH.relative_to(REPO_ROOT)}/") return 0 diff --git a/utils/md2json/config.py b/utils/md2json/config.py index fd7292b2..146caf9c 100644 --- a/utils/md2json/config.py +++ b/utils/md2json/config.py @@ -13,7 +13,7 @@ def _find_repo_root(start: Path) -> Path: for candidate in (start, *start.parents): if (candidate / "_config.yml").is_file(): return candidate - return start.parents[2] # utils/md2json/config.py -> repo root + return start.parents[1] # start is utils/md2json, so parents[1] is the repo REPO_ROOT = _find_repo_root(Path(__file__).resolve().parent) diff --git a/utils/md2json/output.py b/utils/md2json/output.py index ad0c2977..6cd6a0df 100644 --- a/utils/md2json/output.py +++ b/utils/md2json/output.py @@ -9,11 +9,20 @@ def dumps(document: dict) -> str: + """Serialise one view document as indented, UTF-8 friendly JSON.""" return json.dumps(document, indent=4, ensure_ascii=False) + "\n" -def write_document(path: Path, document: dict) -> None: +def label(path: Path, root: Path | None = None) -> str: + """Path shown in the run log, relative to the output root where possible.""" + try: + return str(path.relative_to(root or OUTPUT_PATH)) + except ValueError: + return str(path) + + +def write_document(path: Path, document: dict, root: Path | None = None) -> None: """Write one document, overwriting whatever was there.""" path.parent.mkdir(parents=True, exist_ok=True) path.write_text(dumps(document), encoding="utf-8") - print(f"[write] {path.relative_to(OUTPUT_PATH)}") + print(f"[write] {label(path, root)}") From 9dd9a2cce279b37fd8cef97be358c14f3c052d60 Mon Sep 17 00:00:00 2001 From: SantamRC Date: Sat, 29 Aug 2026 04:11:20 +0530 Subject: [PATCH 4/9] fix: never leave the output tree missing if the swap fails The previous swap deleted the existing output before moving the staging tree into place, so a failure in that move left no output at all, and concurrent readers could observe the path as absent between the two operations. Move the existing tree to a sibling backup instead of deleting it, then swap the staging tree in, then discard the backup. If the swap fails the backup is restored. If both the swap and the restore fail the backup is kept rather than cleaned up, since it then holds the only copy, and its path is reported. Verified by fault injection: a failure during generation leaves the tree untouched, a failure during the swap restores all 57 files, and a failure of both leaves a named backup behind. No staging or backup directories are left in any case. --- utils/md2json/cli.py | 28 ++++++++++++++++++++++++---- 1 file changed, 24 insertions(+), 4 deletions(-) diff --git a/utils/md2json/cli.py b/utils/md2json/cli.py index ecd7ea90..45c26c62 100644 --- a/utils/md2json/cli.py +++ b/utils/md2json/cli.py @@ -2,6 +2,7 @@ from __future__ import annotations +import os import shutil import tempfile from pathlib import Path @@ -86,17 +87,36 @@ def main() -> int: # The output tree is generator-owned: build it beside the real one and swap, # so a removed or renumbered section cannot leave stale JSON behind and a - # failure part-way through cannot leave a half-written tree. + # failure part-way through cannot leave a half-written tree. The previous + # tree is moved aside rather than deleted, and restored if the swap fails, + # so a failure never leaves the output missing. OUTPUT_PATH.parent.mkdir(parents=True, exist_ok=True) staging = Path(tempfile.mkdtemp(dir=OUTPUT_PATH.parent, prefix=".md2json-")) + backup = OUTPUT_PATH.with_name(f".{OUTPUT_PATH.name}.backup-{os.getpid()}") + moved = swapped = False try: written, warnings = generate(staging) - if OUTPUT_PATH.exists(): - shutil.rmtree(OUTPUT_PATH) - staging.replace(OUTPUT_PATH) + shutil.rmtree(backup, ignore_errors=True) # leftover from an earlier crash + moved = OUTPUT_PATH.exists() + if moved: + OUTPUT_PATH.replace(backup) + try: + staging.replace(OUTPUT_PATH) + swapped = True + except BaseException: + if moved: + backup.replace(OUTPUT_PATH) + raise except BaseException: shutil.rmtree(staging, ignore_errors=True) raise + finally: + # Only discard the backup once the new tree is in place. If the swap and + # the restore both failed it holds the only copy, so say where it is. + if swapped: + shutil.rmtree(backup, ignore_errors=True) + elif moved and backup.exists(): + print(f"previous output preserved at {backup}") report(warnings) print(f"\n{written} files written to {OUTPUT_PATH.relative_to(REPO_ROOT)}/") From 151da846d0283e521a14908cbbcbaf2d053ec4ad Mon Sep 17 00:00:00 2001 From: SantamRC Date: Sat, 29 Aug 2026 12:43:36 +0530 Subject: [PATCH 5/9] feat: publish the book API with the site deployment The generator wrote to a directory that never reached the deploy, so the JSON was only ever available locally. The existing page API is published by building the site into out/ and having utils/api_generator.py write out/_api before peaceiris/actions-gh-pages deploys out/ to GitHub Pages. Serve the book API the same way. - write to out/api/ instead of a top-level directory, so the deploy publishes it - run the generator in the deploy workflow, after the jekyll build, since jekyll clears its destination directory - refuse to generate into the repository root or any ancestor of docs/, as the output tree is now inside the build output and is replaced wholesale Endpoints become /api/navbar.json, /api/about.json and /api//.json, alongside the existing /_api/pages/. Verified in deploy order: with out/ already populated by a build, the run adds out/api/ with all 57 documents, leaves the rest of the site untouched, and leaves no staging or backup directories that would be published. --- .github/workflows/deploy.yml | 82 +++++++++++++++++++----------------- .gitignore | 2 - utils/md2json/cli.py | 6 +++ utils/md2json/config.py | 7 ++- 4 files changed, 54 insertions(+), 43 deletions(-) diff --git a/.github/workflows/deploy.yml b/.github/workflows/deploy.yml index 2d4dbfcc..408bc964 100644 --- a/.github/workflows/deploy.yml +++ b/.github/workflows/deploy.yml @@ -1,39 +1,43 @@ -name: Interactive Book Deployment - -on: - push: - branches: - - master - -jobs: - github-pages: - runs-on: ubuntu-latest - steps: - - name: Checkout code - uses: actions/checkout@v2 - - - uses: ruby/setup-ruby@v1 - with: - ruby-version: 3.3 # Not needed with a .ruby-version file - bundler-cache: true # runs 'bundle install' and caches installed gems automatically - - - name: Build - run: bundle exec jekyll build -d out - - - name: Run temp server - run: bundle exec jekyll serve --port=4000 --detach - - - name: API Generation - run: sudo python utils/api_generator.py - - - name: Kill Temporary Server - run: pkill -f jekyll - - - name: Deploy - uses: peaceiris/actions-gh-pages@v3 - with: - github_token: ${{ secrets.GITHUB_TOKEN }} - publish_dir: ./out - cname: learn.circuitverse.org - - +name: Interactive Book Deployment + +on: + push: + branches: + - master + +jobs: + github-pages: + runs-on: ubuntu-latest + steps: + - name: Checkout code + uses: actions/checkout@v2 + + - uses: ruby/setup-ruby@v1 + with: + ruby-version: 3.3 # Not needed with a .ruby-version file + bundler-cache: true # runs 'bundle install' and caches installed gems automatically + + - name: Build + run: bundle exec jekyll build -d out + + # After the build: jekyll clears the destination directory. + - name: Book API Generation + run: python3 utils/md2json + + - name: Run temp server + run: bundle exec jekyll serve --port=4000 --detach + + - name: API Generation + run: sudo python utils/api_generator.py + + - name: Kill Temporary Server + run: pkill -f jekyll + + - name: Deploy + uses: peaceiris/actions-gh-pages@v3 + with: + github_token: ${{ secrets.GITHUB_TOKEN }} + publish_dir: ./out + cname: learn.circuitverse.org + + diff --git a/.gitignore b/.gitignore index fcddb672..4b4186cc 100644 --- a/.gitignore +++ b/.gitignore @@ -9,7 +9,5 @@ out /.idea /chapters/ -/chapters2/ -/TestJSON/ __pycache__/ *.py[co] diff --git a/utils/md2json/cli.py b/utils/md2json/cli.py index 45c26c62..bf0b6e93 100644 --- a/utils/md2json/cli.py +++ b/utils/md2json/cli.py @@ -85,6 +85,12 @@ def main() -> int: print(f"docs directory not found: {DOCS_PATH}") return 1 + # The output tree is replaced wholesale, so refuse to point it at the repo + # itself or at anything containing the sources. + if OUTPUT_PATH == REPO_ROOT or OUTPUT_PATH in DOCS_PATH.parents: + print(f"refusing to generate into {OUTPUT_PATH}: it contains the sources") + return 1 + # The output tree is generator-owned: build it beside the real one and swap, # so a removed or renumbered section cannot leave stale JSON behind and a # failure part-way through cannot leave a half-written tree. The previous diff --git a/utils/md2json/config.py b/utils/md2json/config.py index 146caf9c..acdec115 100644 --- a/utils/md2json/config.py +++ b/utils/md2json/config.py @@ -18,8 +18,11 @@ def _find_repo_root(start: Path) -> Path: REPO_ROOT = _find_repo_root(Path(__file__).resolve().parent) DOCS_PATH = REPO_ROOT / "docs" -# The generated tree. Regenerated in full on every run. -OUTPUT_PATH = REPO_ROOT / "TestJSON" +# The generated tree, written inside the Jekyll build output so the deploy +# workflow publishes it to GitHub Pages alongside the site, the same way +# utils/api_generator.py publishes out/_api. Served at /api/. +# Must stay a subdirectory of out/: it is replaced wholesale on every run. +OUTPUT_PATH = REPO_ROOT / "out" / "api" # BibTeX sources behind {% cite %} / {% bibliography %}. BIBLIOGRAPHY_PATH = REPO_ROOT / "_bibliography" From e391ac8de4f5eb73c367218784bdb42a6465b348 Mon Sep 17 00:00:00 2001 From: SantamRC Date: Sat, 29 Aug 2026 13:01:09 +0530 Subject: [PATCH 6/9] feat: replace the markdown page API with the structured JSON API The site published two APIs. utils/api_generator.py crawled jekyll-admin's endpoint to mirror out/_api, serving each page as Jekyll-rendered HTML inside a JSON envelope. utils/md2json now publishes out/api, serving the same pages as a structured view model. The app consumes the structured form, so the markdown API is retired. Removing it also simplifies the deploy. Generating _api required booting a detached jekyll server, crawling it over HTTP with an unqualified `sudo python`, then pkill-ing the server: three steps, a background process and a second interpreter invocation. The generator reads docs/ directly, so the workflow drops to build, generate, deploy. jekyll-admin stays in the Gemfile: it is in the :jekyll_plugins group and still powers the local /admin editing UI for contributors. BREAKING CHANGE: https://learn.circuitverse.org/_api/ is no longer published. Consumers should move to https://learn.circuitverse.org/api/, which serves navbar.json, about.json, guidelines.json and /.json. --- .github/workflows/deploy.yml | 13 +---- utils/api_generator.py | 92 ------------------------------------ utils/md2json/config.py | 3 +- 3 files changed, 3 insertions(+), 105 deletions(-) delete mode 100644 utils/api_generator.py diff --git a/.github/workflows/deploy.yml b/.github/workflows/deploy.yml index 408bc964..50444ddd 100644 --- a/.github/workflows/deploy.yml +++ b/.github/workflows/deploy.yml @@ -20,18 +20,9 @@ jobs: - name: Build run: bundle exec jekyll build -d out - # After the build: jekyll clears the destination directory. - - name: Book API Generation - run: python3 utils/md2json - - - name: Run temp server - run: bundle exec jekyll serve --port=4000 --detach - + # Runs after the build: jekyll clears its destination directory. - name: API Generation - run: sudo python utils/api_generator.py - - - name: Kill Temporary Server - run: pkill -f jekyll + run: python3 utils/md2json - name: Deploy uses: peaceiris/actions-gh-pages@v3 diff --git a/utils/api_generator.py b/utils/api_generator.py deleted file mode 100644 index b591d789..00000000 --- a/utils/api_generator.py +++ /dev/null @@ -1,92 +0,0 @@ -#!/usr/bin/python3 -# -# Python script for Interactive Book to generate Page APIs from -# Jekyll-admin and store them in build directory -# -# Copyright (C) 2021, Manjot Sidhu -# (C) 2021, CircuitVerse - -# Imports -import sys, json, urllib.request, shutil, os, argparse - -# Constants -host = "localhost" -port = "4000" -base_path = "pages" - -# Hosted API URL -deployed_host = "learn.circuitverse.org" -deployed_url = f'https://{deployed_host}/_api' - -# Local API URL -base_api_url = f'http://{host}:{port}/_api' -base_url = f'{base_api_url}/{base_path}' - -# Output Directories -build_dir = 'out' -api_dir = f'{build_dir}/_api' - -# Arguments Parser -parser = argparse.ArgumentParser(description='Generate Page APIs from Jekyll-admin and store them in build directory') -parser.add_argument('--src_port', type=int, dest='src_port', default=port, help='Destination port address where the generated API will be hosted.') -parser.add_argument('--dest_host', type=str, dest='dest_host', default=deployed_host, help='Destination host address of the site where the API will be hosted.') -args = parser.parse_args() - -# Overide Arguments -port = args.src_port -deployed_host = args.dest_host - - -def get_response(url): - return urllib.request.urlopen(url).read().decode('utf-8') - - -def get_json(response): - return json.loads(response) - - -def save_content(content, file_path): - print(f'Saving to {file_path}') - os.makedirs(os.path.dirname(file_path), exist_ok=True) - f = open(file_path, "w", encoding='utf-8') - f.write(content) - f.close() - - -def get_effective_file_path(path): - return f'{api_dir}/{path}' - - -def get_effective_content(content): - return content.replace(f'http://{host}:{port}', f'https://{deployed_host}') - - -def save_api_page(res, api_path): - save_content(get_effective_content(res), get_effective_file_path(api_path) + ".json") - - -def save_page(api_url): - res = get_response(api_url) - api_data = get_json(res) - save_content(get_effective_content(res), get_effective_file_path(f'{base_path}/{api_data["path"]}')) - - -def recursive_scan(api_url, path): - res = get_response(api_url) - api_data = get_json(res) - save_api_page(res, path) - - for item in api_data: - if "type" not in item: - save_page(item["api_url"]) - elif item["type"] == "directory": - recursive_scan(item["api_url"], f'{base_path}/{item["path"]}') - - -def main(): - print("CV Interactive Book API Generator v0.1\n") - recursive_scan(base_url, base_path) - - -if __name__ == '__main__': - main() diff --git a/utils/md2json/config.py b/utils/md2json/config.py index acdec115..1e0fdab1 100644 --- a/utils/md2json/config.py +++ b/utils/md2json/config.py @@ -19,8 +19,7 @@ def _find_repo_root(start: Path) -> Path: REPO_ROOT = _find_repo_root(Path(__file__).resolve().parent) DOCS_PATH = REPO_ROOT / "docs" # The generated tree, written inside the Jekyll build output so the deploy -# workflow publishes it to GitHub Pages alongside the site, the same way -# utils/api_generator.py publishes out/_api. Served at /api/. +# workflow publishes it to GitHub Pages alongside the site. Served at /api/. # Must stay a subdirectory of out/: it is replaced wholesale on every run. OUTPUT_PATH = REPO_ROOT / "out" / "api" # BibTeX sources behind {% cite %} / {% bibliography %}. From 5235a1c23e218b1289df607bda9263e06da50e03 Mon Sep 17 00:00:00 2001 From: SantamRC Date: Sat, 29 Aug 2026 13:09:00 +0530 Subject: [PATCH 7/9] feat: make the navbar self-describing for remote clients navbar.json identified each chapter by id and display name only, while its documents are generated into a slug directory. As a local bundle a client could carry its own ordered list of directories; as a remote API there was no way to build a URL from the navbar at all, and any change to a chapter's nav_order would silently repoint a hardcoded list at the wrong chapter. Each chapter now carries the directory it was generated into, so a client can request "//0.json" for a chapter index and "//.json" for a section. The field is additive; existing consumers ignore it. Verified that every URL derivable from navbar.json resolves to valid JSON, and that no generated document is unreachable from it. --- utils/md2json/book.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/utils/md2json/book.py b/utils/md2json/book.py index 391c4ba3..39e34200 100644 --- a/utils/md2json/book.py +++ b/utils/md2json/book.py @@ -67,12 +67,18 @@ def discover_chapters(docs: Path) -> list[Chapter]: def build_navbar(chapters: list[Chapter]) -> dict: - """Build the navigation document listing chapters and their sections.""" + """Build the navigation document listing chapters and their sections. + + Each chapter carries the directory it was generated into, so a client can + build "//.json" for a section and "//0.json" for + the chapter index without knowing the chapter ordering in advance. + """ return { "chapters": [ { "id": chapter_id, "name": chapter.title, + "path": chapter.directory, "sub-chapters": [ {"id": number, "name": section.title} for number, section in enumerate(chapter.sections, start=1) From aecb0f199328f560d838290385e9f429b39318b2 Mon Sep 17 00:00:00 2001 From: SantamRC Date: Sat, 29 Aug 2026 13:12:56 +0530 Subject: [PATCH 8/9] fix: reject output paths overlapping the source trees The guard rejected only paths strictly above docs/, so OUTPUT_PATH == DOCS_PATH passed it. The swap would then move docs/ aside, write the generated JSON in its place, and delete the backup once the swap succeeded, destroying the book source. A path inside docs/ was likewise allowed and would have replaced a chapter directory. Reject overlap in either direction, equal to a source tree, containing it, or inside it, and resolve the paths first so a relative path or symlink cannot slip past. _bibliography/ is protected alongside docs/, since it carries the same risk. The repository root is covered by the containment check. Verified: out/api generates normally, while docs/, docs/api, docs/logic-design, _bibliography/, the repository root, / and out/../docs are all refused. --- utils/md2json/cli.py | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/utils/md2json/cli.py b/utils/md2json/cli.py index bf0b6e93..53e6895a 100644 --- a/utils/md2json/cli.py +++ b/utils/md2json/cli.py @@ -10,6 +10,7 @@ from .bibliography import load_bibliography from .book import build_navbar, build_page, chapter_contents, discover_chapters from .config import ( + BIBLIOGRAPHY_PATH, DOCS_PATH, OUTPUT_PATH, REPO_ROOT, @@ -85,11 +86,15 @@ def main() -> int: print(f"docs directory not found: {DOCS_PATH}") return 1 - # The output tree is replaced wholesale, so refuse to point it at the repo - # itself or at anything containing the sources. - if OUTPUT_PATH == REPO_ROOT or OUTPUT_PATH in DOCS_PATH.parents: - print(f"refusing to generate into {OUTPUT_PATH}: it contains the sources") - return 1 + # The output tree is replaced wholesale and its predecessor is then deleted, + # so refuse any path overlapping a source tree in either direction: equal to + # it, containing it, or inside it. Resolved first so a relative path or a + # symlink cannot slip past. + output = OUTPUT_PATH.resolve() + for source in (DOCS_PATH.resolve(), BIBLIOGRAPHY_PATH.resolve()): + if output == source or output in source.parents or source in output.parents: + print(f"refusing to generate into {output}: it overlaps sources at {source}") + return 1 # The output tree is generator-owned: build it beside the real one and swap, # so a removed or renumbered section cannot leave stale JSON behind and a From 834cfe772b20338b129843fc69627e363df564fc Mon Sep 17 00:00:00 2001 From: SantamRC Date: Sat, 29 Aug 2026 13:46:08 +0530 Subject: [PATCH 9/9] style: restore CRLF line endings in deploy.yml The workflow file is stored with CRLF. Editing it rewrote the whole file with LF, so the diff showed all 34 lines as changed rather than just the three removed steps. No content change. --- .github/workflows/deploy.yml | 68 ++++++++++++++++++------------------ 1 file changed, 34 insertions(+), 34 deletions(-) diff --git a/.github/workflows/deploy.yml b/.github/workflows/deploy.yml index 50444ddd..7fef1151 100644 --- a/.github/workflows/deploy.yml +++ b/.github/workflows/deploy.yml @@ -1,34 +1,34 @@ -name: Interactive Book Deployment - -on: - push: - branches: - - master - -jobs: - github-pages: - runs-on: ubuntu-latest - steps: - - name: Checkout code - uses: actions/checkout@v2 - - - uses: ruby/setup-ruby@v1 - with: - ruby-version: 3.3 # Not needed with a .ruby-version file - bundler-cache: true # runs 'bundle install' and caches installed gems automatically - - - name: Build - run: bundle exec jekyll build -d out - - # Runs after the build: jekyll clears its destination directory. - - name: API Generation - run: python3 utils/md2json - - - name: Deploy - uses: peaceiris/actions-gh-pages@v3 - with: - github_token: ${{ secrets.GITHUB_TOKEN }} - publish_dir: ./out - cname: learn.circuitverse.org - - +name: Interactive Book Deployment + +on: + push: + branches: + - master + +jobs: + github-pages: + runs-on: ubuntu-latest + steps: + - name: Checkout code + uses: actions/checkout@v2 + + - uses: ruby/setup-ruby@v1 + with: + ruby-version: 3.3 # Not needed with a .ruby-version file + bundler-cache: true # runs 'bundle install' and caches installed gems automatically + + - name: Build + run: bundle exec jekyll build -d out + + # Runs after the build: jekyll clears its destination directory. + - name: API Generation + run: python3 utils/md2json + + - name: Deploy + uses: peaceiris/actions-gh-pages@v3 + with: + github_token: ${{ secrets.GITHUB_TOKEN }} + publish_dir: ./out + cname: learn.circuitverse.org + +