jung/tools/make_rootlist.py

299 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""tools/make_rootlist.py — generate READING-LIST.md at repo root.
Mirrors the ISAP Zurich reading-list page: one table per section, every item with
its listed title, author, RU edition, card link, and links to ALL downloaded files
(cross-section books link to the canonical section's file, tagged «secNN»).
Sources: sections/NN/INDEX.md (items) + downloads/NN/MANIFEST.md (files) +
sections/NN/NN-*.md cards (RU title). Regenerable, no manual edits.
"""
import os, re, sys, datetime
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DL_DIR = {"09-comparison-of-psychodynamic-concepts": "09-comparison",
"10-psychopathology-psychiatry": "10-psychopathology"}
SECS = ["01-fundamentals", "02-dreams", "03-myths-fairy-tales", "04-pictures",
"05-ethnology", "06-religion", "07-complexes", "08-developmental",
"09-comparison-of-psychodynamic-concepts", "10-psychopathology-psychiatry",
"11-individuation", "12-practical-case"]
DATE = datetime.date.today().isoformat()
def dldir(sec):
return DL_DIR.get(sec, sec)
SECNUM_TO_DIR = {s[:2]: dldir(s) for s in SECS}
SECNUM_TO_FULL = {s[:2]: s for s in SECS}
def human(size):
if size >= 1024*1024:
v = size/1048576
return (f"{v:.1f}M" if v < 10 else f"{v:.0f}M")
return f"{size/1024:.0f}K"
def parse_index(sec):
p = os.path.join(BASE, "sections", sec, "INDEX.md")
h1 = re.search(r'^# Section \d{2} — (.+)$', open(p).read(), re.M).group(1)
name = re.sub(r':\s*index.*$|:\s*INDEX.*$', '', h1, flags=re.I).strip()
items = {}
for line in open(p, encoding="utf-8"):
m = re.match(r'^\|\s*(\d{2})\s*\|\s*([^|]*)\|\s*([^|]*)\s*\|\s*([^|]*)\s*\|\s*([^|]+?)\s*\|\s*([^|]+)\|', line)
if m:
num, grp, title, author, status, card = m.groups()
is_xref = status.strip().startswith("xref")
em = re.search(r'[✅🔶❌⬜]', status)
items[num] = dict(title=title.strip(), author=author.strip(),
status=(em.group(0) if em else status.strip()),
is_xref=is_xref, card=card.strip())
return name, items
def parse_manifest(sec):
"""items -> dict(num -> {'files': [(name,size,source)], 'xref': [(target_sec, target_item)]})"""
p = os.path.join(BASE, "downloads", dldir(sec), "MANIFEST.md")
out = {}
cur = None
for line in open(p, encoding="utf-8"):
m = re.match(r'^### (\d{2}) — .*(?:— files in (sec\d\d(?:\+\d\d)*))?\s*$', line)
if line.startswith("### "):
num = line.split(" — ")[0].replace("### ", "").strip()
cur = num
out.setdefault(num, {"files": [], "xref": []})
if re.search(r'— files in sec', line) and not out[num]["files"]:
tgts = re.findall(r'sec(\d\d)', line)
out[num]["xref"].extend(("sec" + t, None, None) for t in tgts)
continue
if line.startswith("## "):
cur = None
continue
if cur is None:
continue
# legacy prose: Файлы: `path1`, `path2`.
mp = re.match(r'Файлы:\s*(.+)$', line)
if mp:
for path in re.findall(r'`([^`]+)`', mp.group(1)):
nm = os.path.basename(path)
out[cur]["files"].append(("@" + path, "", ""))
continue
if not line.startswith("|"):
continue
cells = [c.strip() for c in line.strip("|").split("|")]
if not cells or cells[0] in ("file", "") or cells[0].startswith(":"):
continue
first = cells[0]
# legacy 5-col rows: | EN/RU | file | size | source | verify |
if first in ("EN", "RU", "DE", "ES", "FR") and len(cells) > 2 and re.search(r'\.\w{2,4}$', cells[1]):
out[cur]["files"].append((cells[1], cells[2], cells[3] if len(cells) > 3 else ""))
continue
m = re.match(r'— \(', first)
if m:
tfile = None
fm = re.search(r'(\d\d-[\w.\-]+\.\w{2,4}|\d\d-[\w.\-]+\*)', first)
if fm:
tfile = fm.group(1)
secmm = re.search(r'sec\d\d', first)
out[cur]["xref"].append((secmm.group(0) if secmm else "sec??", None, tfile))
continue
for sec_m, item_m in re.findall(r'sec(\d\d)(?: #(\d{2}))?', first):
out[cur]["xref"].append(("sec" + sec_m, item_m or None, None))
continue
if re.search(r'\.\w{2,4}$', first):
size = cells[1] if len(cells) > 1 else ""
out[cur]["files"].append((first, size, cells[2] if len(cells) > 2 else ""))
return out
def size_from_str(s):
m = re.match(r'([\d.]+)\s*(MB|GB|KB|pp)', s)
if not m:
return None
v = float(m.group(1))
return {"KB": v*1024, "MB": v*1048576, "GB": v*1073741824}[m.group(2)]
def card_path(sec, card):
p = os.path.join(BASE, "sections", sec, card)
if os.path.exists(p):
return p
# INDEX name drift: match by item-number prefix
num = card[:2]
d = os.path.join(BASE, "sections", sec)
cands = [f for f in os.listdir(d) if f.startswith(num + "-") and f.endswith(".md")]
if len(cands) == 1:
return os.path.join(d, cands[0])
return None
def card_ru_title(sec, card):
p = card_path(sec, card)
if p is None:
return ""
s = open(p, encoding="utf-8").read()
m = re.search(r'## Editions — RU\s*\n+\|.*?\n\|[-| ]*\n((?:\|.*\n?)+)', s)
if not m:
return ""
rows = [r for r in m.group(1).strip().splitlines() if r.startswith("|")]
if not rows:
return ""
first = [c.strip() for c in rows[0].strip("|").split("|")]
if first and (first[0].startswith("—") or first[0] in ("", "RU title")):
return "— нет"
t = first[0].strip("«» ")
return t[:90]
def file_links(sec, files, ddir):
out = []
for name, size_s, _src in files:
if name.startswith("@"):
path = name[1:]
if not os.path.exists(os.path.join(BASE, path)):
import glob as _glob
cands = _glob.glob(os.path.join(BASE, "downloads", "*", os.path.basename(path)))
if len(cands) == 1:
path = os.path.relpath(cands[0], BASE)
else:
continue
nm = os.path.basename(path)
secnum = os.path.basename(os.path.dirname(path))[:2]
lang = "RU" if re.search(r'-ru[.\-]', nm) else ("EN" if re.search(r'-en[.\-]', nm) else "")
sz = os.path.getsize(os.path.join(BASE, path))
out.append(f"[{lang} {nm.rsplit('.', 1)[-1]} {human(sz)} →sec{secnum}]({path})")
continue
lang = "RU" if re.search(r'-ru[.\-]', name) else ("EN" if re.search(r'-en[.\-]', name) else "")
sz = size_from_str(size_s) or os.path.getsize(os.path.join(BASE, "downloads", ddir, name))
fmt = name.rsplit(".", 1)[-1]
tag = f"{lang} {fmt} {human(sz)}"
out.append(f"[{tag}](downloads/{ddir}/{name})")
return out
def mklink(name, ddir, sec, sz):
lang = "RU " if re.search(r'-ru[.\-]', name) else ("EN " if re.search(r'-en[.\-]', name) else "")
label = f"{lang}{name.rsplit('.', 1)[-1]} {human(sz) if sz else ''} →sec{sec}".replace(" ", " ").strip()
return f"[{label}](downloads/{ddir}/{name})"
def mkpath(name, sz_hint=""):
"""build (label, relpath) for a file name or @path entry"""
if name.startswith("@"):
path = name[1:]
import glob as _glob
if not os.path.exists(os.path.join(BASE, path)):
cands = _glob.glob(os.path.join(BASE, "downloads", "*", os.path.basename(path)))
if len(cands) == 1:
path = os.path.relpath(cands[0], BASE)
else:
return None
else:
return None # caller handles plain names
nm = os.path.basename(path)
secnum = os.path.basename(os.path.dirname(path))[:2]
lang = "RU " if re.search(r'-ru[.\-]', nm) else ("EN " if re.search(r'-en[.\-]', nm) else "")
sz = os.path.getsize(os.path.join(BASE, path))
return f"{lang}{nm.rsplit('.', 1)[-1]} {human(sz)} →sec{secnum}", path
def main():
# pre-parse all manifests (needed for cross-section resolution)
manifs = {sec: parse_manifest(sec) for sec in SECS}
idxs = {}
for sec in SECS:
idxs[sec] = parse_index(sec)
out = []
out.append(f"# ISAP Zurich — Reading List: RU editions & downloaded files ({DATE})")
out.append("")
out.append("> Все 12 секций. После каждой книги — ссылки на скачанные файлы; у книг из")
out.append("> предыдущих секций файлы лежат в канонической секции (тег `secNN` = где лежит файл).")
out.append("> `✅` = RU-издание найдено, `🔶` = частично (глава/статья), `❌` = RU нет (verified).")
out.append("> Карточка каждой книги — по клику на название. Детали изданий/каталогов — в карточке,")
out.append("> провалы поиска — в `sections/NN/MISSING.md`.")
out.append("")
grand_i = grand_f = grand_l = 0
for sec in SECS:
name, items = idxs[sec]
mf = manifs[sec]
ddir = dldir(sec)
n = int(sec[:2])
dfiles_n = len([f for f in os.listdir(os.path.join(BASE, "downloads", ddir)) if f != "MANIFEST.md"])
stat = {e: 0 for e in "✅🔶❌"}
for it in items.values():
if it.get("is_xref"):
continue # xref items are counted in their canonical section
stat[it["status"]] = stat.get(it["status"], 0) + 1
out.append(f"## {n}. {name}")
out.append("")
out.append(f"_{len(items)} items — ✅ {stat.get('✅',0)} / 🔶 {stat.get('🔶',0)} / ❌ {stat.get('❌',0)} · файлов в секции: {dfiles_n}_")
out.append("")
out.append("| # | Book | Author | RU edition | Files |")
out.append("|---|------|--------|------------|-------|")
nfiles_xref = 0
nlinks = 0
for num in sorted(items):
it = items[num]
md = mf.get(num, {"files": [], "xref": []})
# files: local + resolved xref
links = file_links(sec, md["files"], ddir)
resolved_secs = set()
for tsec, titem, tfile in md["xref"]:
tsec2 = tsec[-2:]
tmf = manifs.get(SECNUM_TO_FULL.get(tsec2, ""), {})
tinfo = tmf.get(titem, {}) if titem else {}
tdir = SECNUM_TO_DIR.get(tsec2)
if tfile and tdir:
import glob as _glob
pats = [os.path.join(BASE, "downloads", tdir, tfile)] if "*" not in tfile else _glob.glob(os.path.join(BASE, "downloads", tdir, tfile))
for pth in pats:
if not os.path.exists(pth):
continue
nm = os.path.basename(pth)
links.append(mklink(nm, tdir, tsec2, os.path.getsize(pth)))
nfiles_xref += 1
resolved_secs.add(tsec2)
continue
elif tdir and tinfo.get("files"):
for name, size_s, _s in tinfo["files"]:
if name.startswith("@"):
mp = mkpath(name)
if mp:
label, path = mp
links.append(f"[{label}]({path})")
nfiles_xref += 1
else:
lang = "RU" if re.search(r'-ru[.\-]', name) else ("EN" if re.search(r'-en[.\-]', name) else "")
sz = size_from_str(size_s) or 0
links.append(f"[{lang} {name.rsplit('.',1)[-1]} {human(sz)} →sec{tsec2[-2:]}](downloads/{tdir}/{name})")
nfiles_xref += 1
resolved_secs.add(tsec2)
elif tdir and tsec2 not in resolved_secs:
# target item number unknown -> link to the section manifest
links.append(f"[→sec{tsec2[-2:]}](downloads/{tdir}/MANIFEST.md)")
# RU title from card
ru = card_ru_title(sec, it["card"])
status = it["status"]
ru_cell = f"{status} {ru}" if ru and status in "✅🔶" else (status if not ru else ru)
if status == "❌":
ru_cell = "—"
# drop bare fallback links if a resolved file link for the same section exists
def bare_sec(lk):
m = re.match(r'^\[→sec(\d\d)\]\(', lk)
return m.group(1) if m else None
resolved_secs2 = set()
for lk in links:
m = re.search(r'→sec(\d\d)\](?!.*MANIFEST)', lk)
if m and not lk.startswith(f"[→sec{m.group(1)}]"):
resolved_secs2.add(m.group(1))
links = [lk for lk in links if bare_sec(lk) is None or bare_sec(lk) not in resolved_secs2]
links_cell = " · ".join(dict.fromkeys(links)) if links else "—"
nlinks += len(dict.fromkeys(links))
# Book link: use real card path if INDEX drifted
cp = card_path(sec, it["card"])
card_link = f"sections/{sec}/{os.path.basename(cp)}" if cp else f"sections/{sec}/{it['card']}"
out.append(f"| {num} | [{it['title']}]({card_link}) | {it['author']} | {ru_cell} | {links_cell} |")
grand_i += len(items)
grand_f += dfiles_n
grand_l += nlinks
out.append("")
out.append("---")
out.append(f"_Итого: {grand_i} позиций · {grand_f} файлов · {grand_l} ссылок (включая cross-section) · {DATE}._")
out.append("")
text = "\n".join(out)
if "--check" not in sys.argv:
open(os.path.join(BASE, "READING-LIST.md"), "w", encoding="utf-8").write(text)
print(f"READING-LIST.md written: {len(text)} chars, {text.count(chr(10))} lines")
else:
print(text[:2000])
if __name__ == "__main__":
main()