update at 2026-07-17 15:39:39
This commit is contained in:
115
export/formatter.md
Normal file
115
export/formatter.md
Normal file
@@ -0,0 +1,115 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
import os
|
||||
import re
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from pypinyin import lazy_pinyin
|
||||
|
||||
INPUT_DIR = "/Users/gavin/kman/export"
|
||||
OUTPUT_DIR = os.path.join(INPUT_DIR, "out")
|
||||
os.makedirs(OUTPUT_DIR, exist_ok=True)
|
||||
|
||||
def slugify(title):
|
||||
# 拼音 + 只保留字母 + 最多15个字符
|
||||
py = "".join(lazy_pinyin(title))
|
||||
s = re.sub("[^a-zA-Z]", "", py)
|
||||
return s[:15]
|
||||
|
||||
def parse_filename(fname):
|
||||
# 去掉 note_bleaf 前缀
|
||||
name = fname[len("note_bleaf"):]
|
||||
# 分 title / subtitle
|
||||
if ":" in name:
|
||||
title, rest = name.split(":", 1)
|
||||
# 去掉尾部可能的 -xxxx.md
|
||||
subtitle = re.sub(r'-.*\.md$', '', rest)
|
||||
else:
|
||||
title = re.sub(r'-.*\.md$', '', name).replace(".md","")
|
||||
subtitle = ""
|
||||
return title.strip(), subtitle.strip()
|
||||
|
||||
def parse_file(filepath):
|
||||
"""逐行解析,保证 CONTENT 中有 | 也能处理"""
|
||||
entries = []
|
||||
with open(filepath, "r", encoding="utf-8") as f:
|
||||
lines = f.readlines()
|
||||
for line in lines:
|
||||
line = line.strip()
|
||||
if not line or line.startswith("--") or line.startswith("TYPE"):
|
||||
continue
|
||||
parts = line.split("|", 4) # 只分成 5 个字段
|
||||
if len(parts) < 5:
|
||||
print(f"⚠️ 跳过解析失败行: {line}")
|
||||
continue
|
||||
type_, bookname, author, marktime, content = parts
|
||||
try:
|
||||
marktime_dt = datetime.strptime(marktime.strip(), "%Y/%m/%d %H:%M:%S")
|
||||
except:
|
||||
try:
|
||||
marktime_dt = datetime.strptime(marktime.strip(), "%Y/%m/%d %H:%M")
|
||||
except:
|
||||
marktime_dt = None
|
||||
entries.append({
|
||||
"TYPE": type_.strip(),
|
||||
"BOOKNAME": bookname.strip(),
|
||||
"AUTHOR": author.strip(),
|
||||
"MARKTIME": marktime_dt,
|
||||
"CONTENT": content.strip()
|
||||
})
|
||||
return entries
|
||||
|
||||
def convert_file(filepath):
|
||||
fname = os.path.basename(filepath)
|
||||
title, subtitle = parse_filename(fname)
|
||||
entries = parse_file(filepath)
|
||||
if not entries:
|
||||
print(f"❌ 解析失败 {filepath}")
|
||||
return
|
||||
|
||||
# 最早的时间
|
||||
times = [e["MARKTIME"] for e in entries if e["MARKTIME"]]
|
||||
ddd = min(times).strftime("%Y-%m-%d %H:%M:%S") if times else datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
|
||||
slug = slugify(title)
|
||||
output_fname = f"note_{title}.md"
|
||||
output_path = os.path.join(OUTPUT_DIR, output_fname)
|
||||
|
||||
lines = [
|
||||
"---",
|
||||
f"title: '{title}'",
|
||||
f"subtitle: '{subtitle}'",
|
||||
f"author: '大童'",
|
||||
f"date: {ddd}",
|
||||
f"slug: {slug}",
|
||||
"description:",
|
||||
"---",
|
||||
""
|
||||
]
|
||||
|
||||
# 作者显示第一条的 AUTHOR
|
||||
author_line = f"作者:{entries[0]['AUTHOR']}"
|
||||
lines.append(author_line)
|
||||
lines.append("")
|
||||
|
||||
for e in entries:
|
||||
if e["TYPE"] == "HL":
|
||||
lines.append(e["CONTENT"])
|
||||
lines.append("")
|
||||
elif e["TYPE"] == "NT":
|
||||
lines.append(f"> {e['CONTENT']}")
|
||||
content_text = "\n".join(lines)
|
||||
|
||||
with open(output_path, "w", encoding="utf-8") as f:
|
||||
f.write(content_text)
|
||||
|
||||
print(f"✅ 生成 {output_path}")
|
||||
|
||||
def main():
|
||||
for fname in os.listdir(INPUT_DIR):
|
||||
if fname.startswith("note_bleaf") and fname.endswith(".md"):
|
||||
convert_file(os.path.join(INPUT_DIR, fname))
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user