READING-LIST.md: root view of the whole ISAP list with file links (tools/make_rootlist.py)

User request: one big list at repo root mirroring isapzurich.com/en/library/reading-list —
every book with links to ALL downloaded files, INCLUDING cross-section books (link always
resolves to the canonical section's file, tagged →secNN; per user: link every time the
book is mentioned, even if downloaded in an earlier section).

- 483 positions (12 sections), 566 files, 1234 links, 0 broken (verified)
- sources: sections/NN/INDEX.md (items) + downloads/NN/MANIFEST.md (files) + cards (RU title)
- resolves: per-item blocks, xref rows (secNN #MM), legacy 5-col rows (| EN | file | …),
  legacy prose «Файлы: `path`» rows, glob xref rows (04-cw9-archetypes-*), card-name drift
  (sec07 INDEX) by item-number prefix
- manifest fixes found while building: sec08/10/11 'files in sec??' headings → '0 files'
  (23 blocks, paid/absent files), sec12 #03 xref row gained #19, sec11 #13 → 0 files
- AGENTS.md: layout line + total corrected (483 positions, 566 files)

CANON question answered: the new manifest canon (09-12 style) is exactly what makes this
generator possible; sec01-08 were migrated to it in the previous commit.
This commit is contained in:
Dmitry Kokorin 2026-09-29 00:07:02 +03:00
parent 2847f1bcea
commit 7a402ee0ba
7 changed files with 902 additions and 102 deletions

296
tools/make_rootlist.py Normal file
View file

@ -0,0 +1,296 @@
#!/usr/bin/env python3
"""tools/make_rootlist.py — generate READING-LIST.md at repo root.
Mirrors the ISAP Zurich reading-list page: one table per section, every item with
its listed title, author, RU edition, card link, and links to ALL downloaded files
(cross-section books link to the canonical section's file, tagged «secNN»).
Sources: sections/NN/INDEX.md (items) + downloads/NN/MANIFEST.md (files) +
sections/NN/NN-*.md cards (RU title). Regenerable, no manual edits.
"""
import os, re, sys
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DL_DIR = {"09-comparison-of-psychodynamic-concepts": "09-comparison",
"10-psychopathology-psychiatry": "10-psychopathology"}
SECS = ["01-fundamentals", "02-dreams", "03-myths-fairy-tales", "04-pictures",
"05-ethnology", "06-religion", "07-complexes", "08-developmental",
"09-comparison-of-psychodynamic-concepts", "10-psychopathology-psychiatry",
"11-individuation", "12-practical-case"]
DATE = "2026-09-28"
def dldir(sec):
return DL_DIR.get(sec, sec)
SECNUM_TO_DIR = {s[:2]: dldir(s) for s in SECS}
SECNUM_TO_FULL = {s[:2]: s for s in SECS}
def human(size):
if size >= 1024*1024:
v = size/1048576
return (f"{v:.1f}M" if v < 10 else f"{v:.0f}M")
return f"{size/1024:.0f}K"
def parse_index(sec):
p = os.path.join(BASE, "sections", sec, "INDEX.md")
h1 = re.search(r'^# Section \d{2} — (.+)$', open(p).read(), re.M).group(1)
name = re.sub(r':\s*index.*$|:\s*INDEX.*$', '', h1, flags=re.I).strip()
items = {}
for line in open(p, encoding="utf-8"):
m = re.match(r'^\|\s*(\d{2})\s*\|\s*([^|]*)\|\s*([^|]*)\s*\|\s*([^|]*)\s*\|\s*([^|]+?)\s*\|\s*([^|]+)\|', line)
if m:
num, grp, title, author, status, card = m.groups()
em = re.search(r'[✅🔶❌⬜]', status)
items[num] = dict(title=title.strip(), author=author.strip(),
status=(em.group(0) if em else ("xref" if status.strip().startswith("xref") else status.strip())),
card=card.strip())
return name, items
def parse_manifest(sec):
"""items -> dict(num -> {'files': [(name,size,source)], 'xref': [(target_sec, target_item)]})"""
p = os.path.join(BASE, "downloads", dldir(sec), "MANIFEST.md")
out = {}
cur = None
for line in open(p, encoding="utf-8"):
m = re.match(r'^### (\d{2}) — .*(?:— files in (sec\d\d(?:\+\d\d)*))?\s*$', line)
if line.startswith("### "):
num = line.split(" — ")[0].replace("### ", "").strip()
cur = num
out.setdefault(num, {"files": [], "xref": []})
if re.search(r'— files in sec', line) and not out[num]["files"]:
tgts = re.findall(r'sec(\d\d)', line)
out[num]["xref"].extend(("sec" + t, None, None) for t in tgts)
continue
if line.startswith("## "):
cur = None
continue
if cur is None:
continue
# legacy prose: Файлы: `path1`, `path2`.
mp = re.match(r'Файлы:\s*(.+)$', line)
if mp:
for path in re.findall(r'`([^`]+)`', mp.group(1)):
nm = os.path.basename(path)
out[cur]["files"].append(("@" + path, "", ""))
continue
if not line.startswith("|"):
continue
cells = [c.strip() for c in line.strip("|").split("|")]
if not cells or cells[0] in ("file", "") or cells[0].startswith(":"):
continue
first = cells[0]
# legacy 5-col rows: | EN/RU | file | size | source | verify |
if first in ("EN", "RU", "DE", "ES", "FR") and len(cells) > 2 and re.search(r'\.\w{2,4}$', cells[1]):
out[cur]["files"].append((cells[1], cells[2], cells[3] if len(cells) > 3 else ""))
continue
m = re.match(r'— \(', first)
if m:
tfile = None
fm = re.search(r'(\d\d-[\w.\-]+\.\w{2,4}|\d\d-[\w.\-]+\*)', first)
if fm:
tfile = fm.group(1)
secmm = re.search(r'sec\d\d', first)
out[cur]["xref"].append((secmm.group(0) if secmm else "sec??", None, tfile))
continue
for sec_m, item_m in re.findall(r'sec(\d\d)(?: #(\d{2}))?', first):
out[cur]["xref"].append(("sec" + sec_m, item_m or None, None))
continue
if re.search(r'\.\w{2,4}$', first):
size = cells[1] if len(cells) > 1 else ""
out[cur]["files"].append((first, size, cells[2] if len(cells) > 2 else ""))
return out
def size_from_str(s):
m = re.match(r'([\d.]+)\s*(MB|GB|KB|pp)', s)
if not m:
return None
v = float(m.group(1))
return {"KB": v*1024, "MB": v*1048576, "GB": v*1073741824}[m.group(2)]
def card_path(sec, card):
p = os.path.join(BASE, "sections", sec, card)
if os.path.exists(p):
return p
# INDEX name drift: match by item-number prefix
num = card[:2]
d = os.path.join(BASE, "sections", sec)
cands = [f for f in os.listdir(d) if f.startswith(num + "-") and f.endswith(".md")]
if len(cands) == 1:
return os.path.join(d, cands[0])
return None
def card_ru_title(sec, card):
p = card_path(sec, card)
if p is None:
return ""
s = open(p, encoding="utf-8").read()
m = re.search(r'## Editions — RU\s*\n+\|.*?\n\|[-| ]*\n((?:\|.*\n?)+)', s)
if not m:
return ""
rows = [r for r in m.group(1).strip().splitlines() if r.startswith("|")]
if not rows:
return ""
first = [c.strip() for c in rows[0].strip("|").split("|")]
if first and (first[0].startswith("—") or first[0] in ("", "RU title")):
return "— нет"
t = first[0].strip("«» ")
return t[:90]
def file_links(sec, files, ddir):
out = []
for name, size_s, _src in files:
if name.startswith("@"):
path = name[1:]
if not os.path.exists(os.path.join(BASE, path)):
import glob as _glob
cands = _glob.glob(os.path.join(BASE, "downloads", "*", os.path.basename(path)))
if len(cands) == 1:
path = os.path.relpath(cands[0], BASE)
else:
continue
nm = os.path.basename(path)
secnum = os.path.basename(os.path.dirname(path))[:2]
lang = "RU" if re.search(r'-ru[.\-]', nm) else ("EN" if re.search(r'-en[.\-]', nm) else "")
sz = os.path.getsize(os.path.join(BASE, path))
out.append(f"[{lang} {nm.rsplit('.', 1)[-1]} {human(sz)} →sec{secnum}]({path})")
continue
lang = "RU" if re.search(r'-ru[.\-]', name) else ("EN" if re.search(r'-en[.\-]', name) else "")
sz = size_from_str(size_s) or os.path.getsize(os.path.join(BASE, "downloads", ddir, name))
fmt = name.rsplit(".", 1)[-1]
tag = f"{lang} {fmt} {human(sz)}"
out.append(f"[{tag}](downloads/{ddir}/{name})")
return out
def mklink(name, ddir, sec, sz):
lang = "RU " if re.search(r'-ru[.\-]', name) else ("EN " if re.search(r'-en[.\-]', name) else "")
label = f"{lang}{name.rsplit('.', 1)[-1]} {human(sz) if sz else ''} →sec{sec}".replace(" ", " ").strip()
return f"[{label}](downloads/{ddir}/{name})"
def mkpath(name, sz_hint=""):
"""build (label, relpath) for a file name or @path entry"""
if name.startswith("@"):
path = name[1:]
import glob as _glob
if not os.path.exists(os.path.join(BASE, path)):
cands = _glob.glob(os.path.join(BASE, "downloads", "*", os.path.basename(path)))
if len(cands) == 1:
path = os.path.relpath(cands[0], BASE)
else:
return None
else:
return None # caller handles plain names
nm = os.path.basename(path)
secnum = os.path.basename(os.path.dirname(path))[:2]
lang = "RU " if re.search(r'-ru[.\-]', nm) else ("EN " if re.search(r'-en[.\-]', nm) else "")
sz = os.path.getsize(os.path.join(BASE, path))
return f"{lang}{nm.rsplit('.', 1)[-1]} {human(sz)} →sec{secnum}", path
def main():
# pre-parse all manifests (needed for cross-section resolution)
manifs = {sec: parse_manifest(sec) for sec in SECS}
idxs = {}
for sec in SECS:
idxs[sec] = parse_index(sec)
out = []
out.append(f"# ISAP Zurich — Reading List: RU editions & downloaded files ({DATE})")
out.append("")
out.append("> Все 12 секций. После каждой книги — ссылки на скачанные файлы; у книг из")
out.append("> предыдущих секций файлы лежат в канонической секции (тег `secNN` = где лежит файл).")
out.append("> `✅` = RU-издание найдено, `🔶` = частично (глава/статья), `❌` = RU нет (verified).")
out.append("> Карточка каждой книги — по клику на название. Детали изданий/каталогов — в карточке,")
out.append("> провалы поиска — в `sections/NN/MISSING.md`.")
out.append("")
grand_i = grand_f = grand_l = 0
for sec in SECS:
name, items = idxs[sec]
mf = manifs[sec]
ddir = dldir(sec)
n = int(sec[:2])
dfiles_n = len([f for f in os.listdir(os.path.join(BASE, "downloads", ddir)) if f != "MANIFEST.md"])
stat = {e: 0 for e in "✅🔶❌"}
for it in items.values():
stat[it["status"]] = stat.get(it["status"], 0) + 1
out.append(f"## {n}. {name}")
out.append("")
out.append(f"_{len(items)} items — ✅ {stat.get('✅',0)} / 🔶 {stat.get('🔶',0)} / ❌ {stat.get('❌',0)} · файлов в секции: {dfiles_n}_")
out.append("")
out.append("| # | Book | Author | RU edition | Files |")
out.append("|---|------|--------|------------|-------|")
nfiles_xref = 0
nlinks = 0
for num in sorted(items):
it = items[num]
md = mf.get(num, {"files": [], "xref": []})
# files: local + resolved xref
links = file_links(sec, md["files"], ddir)
resolved_secs = set()
for tsec, titem, tfile in md["xref"]:
tsec2 = tsec[-2:]
tmf = manifs.get(SECNUM_TO_FULL.get(tsec2, ""), {})
tinfo = tmf.get(titem, {}) if titem else {}
tdir = SECNUM_TO_DIR.get(tsec2)
if tfile and tdir:
import glob as _glob
pats = [os.path.join(BASE, "downloads", tdir, tfile)] if "*" not in tfile else _glob.glob(os.path.join(BASE, "downloads", tdir, tfile))
for pth in pats:
if not os.path.exists(pth):
continue
nm = os.path.basename(pth)
links.append(mklink(nm, tdir, tsec2, os.path.getsize(pth)))
nfiles_xref += 1
resolved_secs.add(tsec2)
continue
elif tdir and tinfo.get("files"):
for name, size_s, _s in tinfo["files"]:
if name.startswith("@"):
mp = mkpath(name)
if mp:
label, path = mp
links.append(f"[{label}]({path})")
nfiles_xref += 1
else:
lang = "RU" if re.search(r'-ru[.\-]', name) else ("EN" if re.search(r'-en[.\-]', name) else "")
sz = size_from_str(size_s) or 0
links.append(f"[{lang} {name.rsplit('.',1)[-1]} {human(sz)} →sec{tsec2[-2:]}](downloads/{tdir}/{name})")
nfiles_xref += 1
resolved_secs.add(tsec2)
elif tdir and tsec2 not in resolved_secs:
# target item number unknown -> link to the section manifest
links.append(f"[→sec{tsec2[-2:]}](downloads/{tdir}/MANIFEST.md)")
# RU title from card
ru = card_ru_title(sec, it["card"])
status = it["status"]
ru_cell = f"{status} {ru}" if ru and status in "✅🔶" else (status if not ru else ru)
if status == "❌":
ru_cell = "—"
# drop bare fallback links if a resolved file link for the same section exists
def bare_sec(lk):
m = re.match(r'^\[→sec(\d\d)\]\(', lk)
return m.group(1) if m else None
resolved_secs2 = set()
for lk in links:
m = re.search(r'→sec(\d\d)\](?!.*MANIFEST)', lk)
if m and not lk.startswith(f"[→sec{m.group(1)}]"):
resolved_secs2.add(m.group(1))
links = [lk for lk in links if bare_sec(lk) is None or bare_sec(lk) not in resolved_secs2]
links_cell = " · ".join(dict.fromkeys(links)) if links else "—"
nlinks += len(dict.fromkeys(links))
# Book link: use real card path if INDEX drifted
cp = card_path(sec, it["card"])
card_link = f"sections/{sec}/{os.path.basename(cp)}" if cp else f"sections/{sec}/{it['card']}"
out.append(f"| {num} | [{it['title']}]({card_link}) | {it['author']} | {ru_cell} | {links_cell} |")
grand_i += len(items)
grand_f += dfiles_n
grand_l += nlinks
out.append("")
out.append("---")
out.append(f"_Итого: {grand_i} позиций · {grand_f} файлов · {grand_l} ссылок (включая cross-section) · {DATE}._")
out.append("")
text = "\n".join(out)
if "--check" not in sys.argv:
open(os.path.join(BASE, "READING-LIST.md"), "w", encoding="utf-8").write(text)
print(f"READING-LIST.md written: {len(text)} chars, {text.count(chr(10))} lines")
else:
print(text[:2000])
if __name__ == "__main__":
main()