121 lines
4.4 KiB
Python
121 lines
4.4 KiB
Python
"""Obsidian vault writer — normalized Markdown notes with YAML frontmatter.
|
|
|
|
Invariants:
|
|
|
|
* Frontmatter values are escaped (backslash + double quote), so titles
|
|
like ``The "Real" Deal`` always produce valid YAML.
|
|
* Re-ingesting a source refreshes its note in place.
|
|
* A *different* source with the same title never clobbers an existing
|
|
note — it claims a deterministic hash-suffixed filename.
|
|
* An empty slug (symbol-only titles) lands on ``untitled``.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import itertools
|
|
import re
|
|
from collections.abc import Iterator
|
|
from datetime import datetime
|
|
from pathlib import Path
|
|
|
|
# Maximum frontmatter lines examined when resolving filename collisions
|
|
# (SEI CERT FIO39-C spirit: bounded reads on files we do not own).
|
|
_FRONTMATTER_SCAN_LIMIT = 32
|
|
|
|
_FRONTMATTER_SOURCE_RE = re.compile(r'^source_file:\s*"(.*)"\s*$')
|
|
_FRONTMATTER_KEYS = ("title:", "source_file:", "ingested_at:", "tags:", "- ")
|
|
|
|
|
|
def _yaml_escape(value: str) -> str:
|
|
"""Escape a string for a double-quoted YAML scalar."""
|
|
return value.replace("\\", "\\\\").replace('"', '\\"')
|
|
|
|
|
|
def _unescape_yaml(value: str) -> str:
|
|
"""Inverse of _yaml_escape for values we wrote ourselves."""
|
|
return value.replace('\\"', '"').replace("\\\\", "\\")
|
|
|
|
|
|
def _frontmatter_lines(fh) -> Iterator[str]:
|
|
"""Yield stripped frontmatter lines, stopping at the block end."""
|
|
for raw in fh:
|
|
stripped = raw.strip()
|
|
if not stripped or stripped == "---":
|
|
continue # block delimiters and blank lines
|
|
yield stripped
|
|
if not stripped.startswith(_FRONTMATTER_KEYS):
|
|
return # first non-frontmatter line ends the block
|
|
|
|
|
|
def _read_frontmatter_source(path: Path) -> str | None:
|
|
"""The ``source_file:`` value from an existing note's frontmatter;
|
|
None when absent or unreadable."""
|
|
try:
|
|
with path.open("r", encoding="utf-8", errors="ignore") as fh:
|
|
bounded = list(itertools.islice(fh, _FRONTMATTER_SCAN_LIMIT))
|
|
except OSError:
|
|
return None
|
|
for line in _frontmatter_lines(bounded):
|
|
match = _FRONTMATTER_SOURCE_RE.match(line)
|
|
if match:
|
|
return _unescape_yaml(match.group(1))
|
|
return None
|
|
|
|
|
|
class ObsidianVaultWriter:
|
|
"""Formats and writes extracted text into normalized Obsidian
|
|
Markdown files under ``<vault>/Ingested_Brain``."""
|
|
|
|
def __init__(self, vault_path: Path, output_dirname: str = "Ingested_Brain"):
|
|
self.output_dir = vault_path / output_dirname
|
|
self.output_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
def write(self, title: str, content: str, source_path: Path,
|
|
source_uri: str | None = None) -> Path:
|
|
"""Write (or refresh) the note for *source_path*. Returns the
|
|
note path — which carries a hash suffix when a *different*
|
|
source already claimed the plain slug. *source_uri* records the
|
|
object-storage location when the archive stage ran first."""
|
|
from slugify import slugify # lazy: only needed once ingesting
|
|
|
|
slug = slugify(title) or "untitled"
|
|
target_file = self._claim_filename(slug, source_path.name)
|
|
|
|
archive_line = (
|
|
f'source_uri: "{_yaml_escape(source_uri)}"\n'
|
|
if source_uri else ""
|
|
)
|
|
frontmatter = (
|
|
"---\n"
|
|
f'title: "{_yaml_escape(title)}"\n'
|
|
f'source_file: "{_yaml_escape(source_path.name)}"\n'
|
|
f'{archive_line}'
|
|
f'ingested_at: "{datetime.now().isoformat()}"\n'
|
|
"tags:\n"
|
|
" - brain/ingested\n"
|
|
f" - source/{source_path.suffix.lstrip('.')}\n"
|
|
"---\n\n"
|
|
)
|
|
|
|
header = (
|
|
f"# {title}\n\n"
|
|
f"*Source document: `{source_path.name}`*\n\n"
|
|
"---\n\n"
|
|
)
|
|
|
|
target_file.write_text(frontmatter + header + content, encoding="utf-8")
|
|
return target_file
|
|
|
|
def _claim_filename(self, slug: str, source_name: str) -> Path:
|
|
"""Resolve the note path for a slug: same source reclaims its
|
|
note; a different source gets a digest-suffixed sibling."""
|
|
target = self.output_dir / f"{slug}.md"
|
|
if not target.exists():
|
|
return target
|
|
existing_source = _read_frontmatter_source(target)
|
|
if existing_source == source_name or existing_source is None:
|
|
return target
|
|
digest = hashlib.md5(source_name.encode()).hexdigest()[:6]
|
|
return self.output_dir / f"{slug}-{digest}.md"
|