Files
kman/export/formatter.md
2026-07-17 15:39:39 +08:00

3.3 KiB
Raw Blame History

#!/usr/bin/env python3

-- coding: utf-8 --

import os import re from datetime import datetime from pathlib import Path from pypinyin import lazy_pinyin

INPUT_DIR = "/Users/gavin/kman/export" OUTPUT_DIR = os.path.join(INPUT_DIR, "out") os.makedirs(OUTPUT_DIR, exist_ok=True)

def slugify(title): # 拼音 + 只保留字母 + 最多15个字符 py = "".join(lazy_pinyin(title)) s = re.sub("[^a-zA-Z]", "", py) return s[:15]

def parse_filename(fname): # 去掉 note_bleaf 前缀 name = fname[len("note_bleaf"):] # 分 title / subtitle if "" in name: title, rest = name.split("", 1) # 去掉尾部可能的 -xxxx.md subtitle = re.sub(r'-..md$', '', rest) else: title = re.sub(r'-..md$', '', name).replace(".md","") subtitle = "" return title.strip(), subtitle.strip()

def parse_file(filepath): """逐行解析,保证 CONTENT 中有 | 也能处理""" entries = [] with open(filepath, "r", encoding="utf-8") as f: lines = f.readlines() for line in lines: line = line.strip() if not line or line.startswith("--") or line.startswith("TYPE"): continue parts = line.split("|", 4) # 只分成 5 个字段 if len(parts) < 5: print(f"⚠️ 跳过解析失败行: {line}") continue type_, bookname, author, marktime, content = parts try: marktime_dt = datetime.strptime(marktime.strip(), "%Y/%m/%d %H:%M:%S") except: try: marktime_dt = datetime.strptime(marktime.strip(), "%Y/%m/%d %H:%M") except: marktime_dt = None entries.append({ "TYPE": type_.strip(), "BOOKNAME": bookname.strip(), "AUTHOR": author.strip(), "MARKTIME": marktime_dt, "CONTENT": content.strip() }) return entries

def convert_file(filepath): fname = os.path.basename(filepath) title, subtitle = parse_filename(fname) entries = parse_file(filepath) if not entries: print(f" 解析失败 {filepath}") return

# 最早的时间
times = [e["MARKTIME"] for e in entries if e["MARKTIME"]]
ddd = min(times).strftime("%Y-%m-%d %H:%M:%S") if times else datetime.now().strftime("%Y-%m-%d %H:%M:%S")

slug = slugify(title)
output_fname = f"note_{title}.md"
output_path = os.path.join(OUTPUT_DIR, output_fname)

lines = [
    "---",
    f"title: '{title}'",
    f"subtitle: '{subtitle}'",
    f"author: '大童'",
    f"date: {ddd}",
    f"slug: {slug}",
    "description:",
    "---",
    ""
]

# 作者显示第一条的 AUTHOR
author_line = f"作者:{entries[0]['AUTHOR']}"
lines.append(author_line)
lines.append("")

for e in entries:
    if e["TYPE"] == "HL":
        lines.append(e["CONTENT"])
        lines.append("")
    elif e["TYPE"] == "NT":
        lines.append(f"> {e['CONTENT']}")
content_text = "\n".join(lines)

with open(output_path, "w", encoding="utf-8") as f:
    f.write(content_text)

print(f"✅ 生成 {output_path}")

def main(): for fname in os.listdir(INPUT_DIR): if fname.startswith("note_bleaf") and fname.endswith(".md"): convert_file(os.path.join(INPUT_DIR, fname))

if name == "main": main()