READING-LIST.md: root view of the whole ISAP list with file links (tools/make_rootlist.py)
User request: one big list at repo root mirroring isapzurich.com/en/library/reading-list — every book with links to ALL downloaded files, INCLUDING cross-section books (link always resolves to the canonical section's file, tagged →secNN; per user: link every time the book is mentioned, even if downloaded in an earlier section). - 483 positions (12 sections), 566 files, 1234 links, 0 broken (verified) - sources: sections/NN/INDEX.md (items) + downloads/NN/MANIFEST.md (files) + cards (RU title) - resolves: per-item blocks, xref rows (secNN #MM), legacy 5-col rows (| EN | file | …), legacy prose «Файлы: `path`» rows, glob xref rows (04-cw9-archetypes-*), card-name drift (sec07 INDEX) by item-number prefix - manifest fixes found while building: sec08/10/11 'files in sec??' headings → '0 files' (23 blocks, paid/absent files), sec12 #03 xref row gained #19, sec11 #13 → 0 files - AGENTS.md: layout line + total corrected (483 positions, 566 files) CANON question answered: the new manifest canon (09-12 style) is exactly what makes this generator possible; sec01-08 were migrated to it in the previous commit.
This commit is contained in:
parent
2847f1bcea
commit
7a402ee0ba
7 changed files with 902 additions and 102 deletions
296
tools/make_rootlist.py
Normal file
296
tools/make_rootlist.py
Normal file
|
|
@ -0,0 +1,296 @@
|
|||
#!/usr/bin/env python3
|
||||
"""tools/make_rootlist.py — generate READING-LIST.md at repo root.
|
||||
|
||||
Mirrors the ISAP Zurich reading-list page: one table per section, every item with
|
||||
its listed title, author, RU edition, card link, and links to ALL downloaded files
|
||||
(cross-section books link to the canonical section's file, tagged «secNN»).
|
||||
Sources: sections/NN/INDEX.md (items) + downloads/NN/MANIFEST.md (files) +
|
||||
sections/NN/NN-*.md cards (RU title). Regenerable, no manual edits.
|
||||
"""
|
||||
import os, re, sys
|
||||
|
||||
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
DL_DIR = {"09-comparison-of-psychodynamic-concepts": "09-comparison",
|
||||
"10-psychopathology-psychiatry": "10-psychopathology"}
|
||||
SECS = ["01-fundamentals", "02-dreams", "03-myths-fairy-tales", "04-pictures",
|
||||
"05-ethnology", "06-religion", "07-complexes", "08-developmental",
|
||||
"09-comparison-of-psychodynamic-concepts", "10-psychopathology-psychiatry",
|
||||
"11-individuation", "12-practical-case"]
|
||||
DATE = "2026-09-28"
|
||||
|
||||
def dldir(sec):
|
||||
return DL_DIR.get(sec, sec)
|
||||
|
||||
SECNUM_TO_DIR = {s[:2]: dldir(s) for s in SECS}
|
||||
SECNUM_TO_FULL = {s[:2]: s for s in SECS}
|
||||
|
||||
def human(size):
|
||||
if size >= 1024*1024:
|
||||
v = size/1048576
|
||||
return (f"{v:.1f}M" if v < 10 else f"{v:.0f}M")
|
||||
return f"{size/1024:.0f}K"
|
||||
|
||||
def parse_index(sec):
|
||||
p = os.path.join(BASE, "sections", sec, "INDEX.md")
|
||||
h1 = re.search(r'^# Section \d{2} — (.+)$', open(p).read(), re.M).group(1)
|
||||
name = re.sub(r':\s*index.*$|:\s*INDEX.*$', '', h1, flags=re.I).strip()
|
||||
items = {}
|
||||
for line in open(p, encoding="utf-8"):
|
||||
m = re.match(r'^\|\s*(\d{2})\s*\|\s*([^|]*)\|\s*([^|]*)\s*\|\s*([^|]*)\s*\|\s*([^|]+?)\s*\|\s*([^|]+)\|', line)
|
||||
if m:
|
||||
num, grp, title, author, status, card = m.groups()
|
||||
em = re.search(r'[✅🔶❌⬜]', status)
|
||||
items[num] = dict(title=title.strip(), author=author.strip(),
|
||||
status=(em.group(0) if em else ("xref" if status.strip().startswith("xref") else status.strip())),
|
||||
card=card.strip())
|
||||
return name, items
|
||||
|
||||
def parse_manifest(sec):
|
||||
"""items -> dict(num -> {'files': [(name,size,source)], 'xref': [(target_sec, target_item)]})"""
|
||||
p = os.path.join(BASE, "downloads", dldir(sec), "MANIFEST.md")
|
||||
out = {}
|
||||
cur = None
|
||||
for line in open(p, encoding="utf-8"):
|
||||
m = re.match(r'^### (\d{2}) — .*(?:— files in (sec\d\d(?:\+\d\d)*))?\s*$', line)
|
||||
if line.startswith("### "):
|
||||
num = line.split(" — ")[0].replace("### ", "").strip()
|
||||
cur = num
|
||||
out.setdefault(num, {"files": [], "xref": []})
|
||||
if re.search(r'— files in sec', line) and not out[num]["files"]:
|
||||
tgts = re.findall(r'sec(\d\d)', line)
|
||||
out[num]["xref"].extend(("sec" + t, None, None) for t in tgts)
|
||||
continue
|
||||
if line.startswith("## "):
|
||||
cur = None
|
||||
continue
|
||||
if cur is None:
|
||||
continue
|
||||
# legacy prose: Файлы: `path1`, `path2`.
|
||||
mp = re.match(r'Файлы:\s*(.+)$', line)
|
||||
if mp:
|
||||
for path in re.findall(r'`([^`]+)`', mp.group(1)):
|
||||
nm = os.path.basename(path)
|
||||
out[cur]["files"].append(("@" + path, "", ""))
|
||||
continue
|
||||
if not line.startswith("|"):
|
||||
continue
|
||||
cells = [c.strip() for c in line.strip("|").split("|")]
|
||||
if not cells or cells[0] in ("file", "") or cells[0].startswith(":"):
|
||||
continue
|
||||
first = cells[0]
|
||||
# legacy 5-col rows: | EN/RU | file | size | source | verify |
|
||||
if first in ("EN", "RU", "DE", "ES", "FR") and len(cells) > 2 and re.search(r'\.\w{2,4}$', cells[1]):
|
||||
out[cur]["files"].append((cells[1], cells[2], cells[3] if len(cells) > 3 else ""))
|
||||
continue
|
||||
m = re.match(r'— \(', first)
|
||||
if m:
|
||||
tfile = None
|
||||
fm = re.search(r'(\d\d-[\w.\-]+\.\w{2,4}|\d\d-[\w.\-]+\*)', first)
|
||||
if fm:
|
||||
tfile = fm.group(1)
|
||||
secmm = re.search(r'sec\d\d', first)
|
||||
out[cur]["xref"].append((secmm.group(0) if secmm else "sec??", None, tfile))
|
||||
continue
|
||||
for sec_m, item_m in re.findall(r'sec(\d\d)(?: #(\d{2}))?', first):
|
||||
out[cur]["xref"].append(("sec" + sec_m, item_m or None, None))
|
||||
continue
|
||||
if re.search(r'\.\w{2,4}$', first):
|
||||
size = cells[1] if len(cells) > 1 else ""
|
||||
out[cur]["files"].append((first, size, cells[2] if len(cells) > 2 else ""))
|
||||
return out
|
||||
|
||||
def size_from_str(s):
|
||||
m = re.match(r'([\d.]+)\s*(MB|GB|KB|pp)', s)
|
||||
if not m:
|
||||
return None
|
||||
v = float(m.group(1))
|
||||
return {"KB": v*1024, "MB": v*1048576, "GB": v*1073741824}[m.group(2)]
|
||||
|
||||
def card_path(sec, card):
|
||||
p = os.path.join(BASE, "sections", sec, card)
|
||||
if os.path.exists(p):
|
||||
return p
|
||||
# INDEX name drift: match by item-number prefix
|
||||
num = card[:2]
|
||||
d = os.path.join(BASE, "sections", sec)
|
||||
cands = [f for f in os.listdir(d) if f.startswith(num + "-") and f.endswith(".md")]
|
||||
if len(cands) == 1:
|
||||
return os.path.join(d, cands[0])
|
||||
return None
|
||||
|
||||
def card_ru_title(sec, card):
|
||||
p = card_path(sec, card)
|
||||
if p is None:
|
||||
return ""
|
||||
s = open(p, encoding="utf-8").read()
|
||||
m = re.search(r'## Editions — RU\s*\n+\|.*?\n\|[-| ]*\n((?:\|.*\n?)+)', s)
|
||||
if not m:
|
||||
return ""
|
||||
rows = [r for r in m.group(1).strip().splitlines() if r.startswith("|")]
|
||||
if not rows:
|
||||
return ""
|
||||
first = [c.strip() for c in rows[0].strip("|").split("|")]
|
||||
if first and (first[0].startswith("—") or first[0] in ("", "RU title")):
|
||||
return "— нет"
|
||||
t = first[0].strip("«» ")
|
||||
return t[:90]
|
||||
|
||||
def file_links(sec, files, ddir):
|
||||
out = []
|
||||
for name, size_s, _src in files:
|
||||
if name.startswith("@"):
|
||||
path = name[1:]
|
||||
if not os.path.exists(os.path.join(BASE, path)):
|
||||
import glob as _glob
|
||||
cands = _glob.glob(os.path.join(BASE, "downloads", "*", os.path.basename(path)))
|
||||
if len(cands) == 1:
|
||||
path = os.path.relpath(cands[0], BASE)
|
||||
else:
|
||||
continue
|
||||
nm = os.path.basename(path)
|
||||
secnum = os.path.basename(os.path.dirname(path))[:2]
|
||||
lang = "RU" if re.search(r'-ru[.\-]', nm) else ("EN" if re.search(r'-en[.\-]', nm) else "")
|
||||
sz = os.path.getsize(os.path.join(BASE, path))
|
||||
out.append(f"[{lang} {nm.rsplit('.', 1)[-1]} {human(sz)} →sec{secnum}]({path})")
|
||||
continue
|
||||
lang = "RU" if re.search(r'-ru[.\-]', name) else ("EN" if re.search(r'-en[.\-]', name) else "")
|
||||
sz = size_from_str(size_s) or os.path.getsize(os.path.join(BASE, "downloads", ddir, name))
|
||||
fmt = name.rsplit(".", 1)[-1]
|
||||
tag = f"{lang} {fmt} {human(sz)}"
|
||||
out.append(f"[{tag}](downloads/{ddir}/{name})")
|
||||
return out
|
||||
|
||||
def mklink(name, ddir, sec, sz):
|
||||
lang = "RU " if re.search(r'-ru[.\-]', name) else ("EN " if re.search(r'-en[.\-]', name) else "")
|
||||
label = f"{lang}{name.rsplit('.', 1)[-1]} {human(sz) if sz else ''} →sec{sec}".replace(" ", " ").strip()
|
||||
return f"[{label}](downloads/{ddir}/{name})"
|
||||
|
||||
def mkpath(name, sz_hint=""):
|
||||
"""build (label, relpath) for a file name or @path entry"""
|
||||
if name.startswith("@"):
|
||||
path = name[1:]
|
||||
import glob as _glob
|
||||
if not os.path.exists(os.path.join(BASE, path)):
|
||||
cands = _glob.glob(os.path.join(BASE, "downloads", "*", os.path.basename(path)))
|
||||
if len(cands) == 1:
|
||||
path = os.path.relpath(cands[0], BASE)
|
||||
else:
|
||||
return None
|
||||
else:
|
||||
return None # caller handles plain names
|
||||
nm = os.path.basename(path)
|
||||
secnum = os.path.basename(os.path.dirname(path))[:2]
|
||||
lang = "RU " if re.search(r'-ru[.\-]', nm) else ("EN " if re.search(r'-en[.\-]', nm) else "")
|
||||
sz = os.path.getsize(os.path.join(BASE, path))
|
||||
return f"{lang}{nm.rsplit('.', 1)[-1]} {human(sz)} →sec{secnum}", path
|
||||
|
||||
def main():
|
||||
# pre-parse all manifests (needed for cross-section resolution)
|
||||
manifs = {sec: parse_manifest(sec) for sec in SECS}
|
||||
idxs = {}
|
||||
for sec in SECS:
|
||||
idxs[sec] = parse_index(sec)
|
||||
out = []
|
||||
out.append(f"# ISAP Zurich — Reading List: RU editions & downloaded files ({DATE})")
|
||||
out.append("")
|
||||
out.append("> Все 12 секций. После каждой книги — ссылки на скачанные файлы; у книг из")
|
||||
out.append("> предыдущих секций файлы лежат в канонической секции (тег `secNN` = где лежит файл).")
|
||||
out.append("> `✅` = RU-издание найдено, `🔶` = частично (глава/статья), `❌` = RU нет (verified).")
|
||||
out.append("> Карточка каждой книги — по клику на название. Детали изданий/каталогов — в карточке,")
|
||||
out.append("> провалы поиска — в `sections/NN/MISSING.md`.")
|
||||
out.append("")
|
||||
grand_i = grand_f = grand_l = 0
|
||||
for sec in SECS:
|
||||
name, items = idxs[sec]
|
||||
mf = manifs[sec]
|
||||
ddir = dldir(sec)
|
||||
n = int(sec[:2])
|
||||
dfiles_n = len([f for f in os.listdir(os.path.join(BASE, "downloads", ddir)) if f != "MANIFEST.md"])
|
||||
stat = {e: 0 for e in "✅🔶❌"}
|
||||
for it in items.values():
|
||||
stat[it["status"]] = stat.get(it["status"], 0) + 1
|
||||
out.append(f"## {n}. {name}")
|
||||
out.append("")
|
||||
out.append(f"_{len(items)} items — ✅ {stat.get('✅',0)} / 🔶 {stat.get('🔶',0)} / ❌ {stat.get('❌',0)} · файлов в секции: {dfiles_n}_")
|
||||
out.append("")
|
||||
out.append("| # | Book | Author | RU edition | Files |")
|
||||
out.append("|---|------|--------|------------|-------|")
|
||||
nfiles_xref = 0
|
||||
nlinks = 0
|
||||
for num in sorted(items):
|
||||
it = items[num]
|
||||
md = mf.get(num, {"files": [], "xref": []})
|
||||
# files: local + resolved xref
|
||||
links = file_links(sec, md["files"], ddir)
|
||||
resolved_secs = set()
|
||||
for tsec, titem, tfile in md["xref"]:
|
||||
tsec2 = tsec[-2:]
|
||||
tmf = manifs.get(SECNUM_TO_FULL.get(tsec2, ""), {})
|
||||
tinfo = tmf.get(titem, {}) if titem else {}
|
||||
tdir = SECNUM_TO_DIR.get(tsec2)
|
||||
if tfile and tdir:
|
||||
import glob as _glob
|
||||
pats = [os.path.join(BASE, "downloads", tdir, tfile)] if "*" not in tfile else _glob.glob(os.path.join(BASE, "downloads", tdir, tfile))
|
||||
for pth in pats:
|
||||
if not os.path.exists(pth):
|
||||
continue
|
||||
nm = os.path.basename(pth)
|
||||
links.append(mklink(nm, tdir, tsec2, os.path.getsize(pth)))
|
||||
nfiles_xref += 1
|
||||
resolved_secs.add(tsec2)
|
||||
continue
|
||||
elif tdir and tinfo.get("files"):
|
||||
for name, size_s, _s in tinfo["files"]:
|
||||
if name.startswith("@"):
|
||||
mp = mkpath(name)
|
||||
if mp:
|
||||
label, path = mp
|
||||
links.append(f"[{label}]({path})")
|
||||
nfiles_xref += 1
|
||||
else:
|
||||
lang = "RU" if re.search(r'-ru[.\-]', name) else ("EN" if re.search(r'-en[.\-]', name) else "")
|
||||
sz = size_from_str(size_s) or 0
|
||||
links.append(f"[{lang} {name.rsplit('.',1)[-1]} {human(sz)} →sec{tsec2[-2:]}](downloads/{tdir}/{name})")
|
||||
nfiles_xref += 1
|
||||
resolved_secs.add(tsec2)
|
||||
elif tdir and tsec2 not in resolved_secs:
|
||||
# target item number unknown -> link to the section manifest
|
||||
links.append(f"[→sec{tsec2[-2:]}](downloads/{tdir}/MANIFEST.md)")
|
||||
# RU title from card
|
||||
ru = card_ru_title(sec, it["card"])
|
||||
status = it["status"]
|
||||
ru_cell = f"{status} {ru}" if ru and status in "✅🔶" else (status if not ru else ru)
|
||||
if status == "❌":
|
||||
ru_cell = "—"
|
||||
# drop bare fallback links if a resolved file link for the same section exists
|
||||
def bare_sec(lk):
|
||||
m = re.match(r'^\[→sec(\d\d)\]\(', lk)
|
||||
return m.group(1) if m else None
|
||||
resolved_secs2 = set()
|
||||
for lk in links:
|
||||
m = re.search(r'→sec(\d\d)\](?!.*MANIFEST)', lk)
|
||||
if m and not lk.startswith(f"[→sec{m.group(1)}]"):
|
||||
resolved_secs2.add(m.group(1))
|
||||
links = [lk for lk in links if bare_sec(lk) is None or bare_sec(lk) not in resolved_secs2]
|
||||
links_cell = " · ".join(dict.fromkeys(links)) if links else "—"
|
||||
nlinks += len(dict.fromkeys(links))
|
||||
# Book link: use real card path if INDEX drifted
|
||||
cp = card_path(sec, it["card"])
|
||||
card_link = f"sections/{sec}/{os.path.basename(cp)}" if cp else f"sections/{sec}/{it['card']}"
|
||||
out.append(f"| {num} | [{it['title']}]({card_link}) | {it['author']} | {ru_cell} | {links_cell} |")
|
||||
grand_i += len(items)
|
||||
grand_f += dfiles_n
|
||||
grand_l += nlinks
|
||||
out.append("")
|
||||
out.append("---")
|
||||
out.append(f"_Итого: {grand_i} позиций · {grand_f} файлов · {grand_l} ссылок (включая cross-section) · {DATE}._")
|
||||
out.append("")
|
||||
text = "\n".join(out)
|
||||
if "--check" not in sys.argv:
|
||||
open(os.path.join(BASE, "READING-LIST.md"), "w", encoding="utf-8").write(text)
|
||||
print(f"READING-LIST.md written: {len(text)} chars, {text.count(chr(10))} lines")
|
||||
else:
|
||||
print(text[:2000])
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue