344 lines
16 KiB
Python
344 lines
16 KiB
Python
#!/usr/bin/env python3
|
||
"""tools/make_rootlist.py — generate READING-LIST.md at repo root.
|
||
|
||
Mirrors the ISAP Zurich reading-list page: one table per section, every item with
|
||
its listed title, author, RU edition, card link, and links to ALL downloaded files
|
||
(cross-section books link to the canonical section's file, tagged «secNN»).
|
||
Sources: sections/NN/INDEX.md (items) + downloads/NN/MANIFEST.md (files) +
|
||
sections/NN/NN-*.md cards (RU title). Regenerable, no manual edits.
|
||
"""
|
||
import os, re, sys, datetime
|
||
|
||
def is_en(f):
|
||
return "-en-" in f or re.search(r"-en(\.b)?\.\w{2,4}$", f) is not None
|
||
|
||
def is_ru(f):
|
||
return "-ru-" in f or re.search(r"-ru(\.b)?\.\w{2,4}$", f) is not None
|
||
|
||
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
DL_DIR = {"09-comparison-of-psychodynamic-concepts": "09-comparison",
|
||
"10-psychopathology-psychiatry": "10-psychopathology"}
|
||
SECS = ["01-fundamentals", "02-dreams", "03-myths-fairy-tales", "04-pictures",
|
||
"05-ethnology", "06-religion", "07-complexes", "08-developmental",
|
||
"09-comparison-of-psychodynamic-concepts", "10-psychopathology-psychiatry",
|
||
"11-individuation", "12-practical-case"]
|
||
DATE = datetime.date.today().isoformat()
|
||
|
||
def dldir(sec):
|
||
return DL_DIR.get(sec, sec)
|
||
|
||
SECNUM_TO_DIR = {s[:2]: dldir(s) for s in SECS}
|
||
SECNUM_TO_FULL = {s[:2]: s for s in SECS}
|
||
|
||
def human(size):
|
||
if size >= 1024*1024:
|
||
v = size/1048576
|
||
return (f"{v:.1f}M" if v < 10 else f"{v:.0f}M")
|
||
return f"{size/1024:.0f}K"
|
||
|
||
def parse_index(sec):
|
||
p = os.path.join(BASE, "sections", sec, "INDEX.md")
|
||
h1 = re.search(r'^# Section \d{2} — (.+)$', open(p).read(), re.M).group(1)
|
||
name = re.sub(r':\s*index.*$|:\s*INDEX.*$', '', h1, flags=re.I).strip()
|
||
items = {}
|
||
for line in open(p, encoding="utf-8"):
|
||
m = re.match(r'^\|\s*(\d{2})\s*\|\s*([^|]*)\|\s*([^|]*)\s*\|\s*([^|]*)\s*\|\s*([^|]+?)\s*\|\s*([^|]+)\|', line)
|
||
if m:
|
||
num, grp, title, author, status, card = m.groups()
|
||
is_xref = status.strip().startswith("xref") or card.strip().endswith("-xref.md")
|
||
em = re.search(r'[✅🔶❌⬜]', status)
|
||
items[num] = dict(title=title.strip(), author=author.strip(),
|
||
status=(em.group(0) if em else status.strip()),
|
||
is_xref=is_xref, card=card.strip())
|
||
return name, items
|
||
|
||
def parse_manifest(sec):
|
||
"""items -> dict(num -> {'files': [(name,size,source)], 'xref': [(target_sec, target_item)]})"""
|
||
p = os.path.join(BASE, "downloads", dldir(sec), "MANIFEST.md")
|
||
out = {}
|
||
cur = None
|
||
for line in open(p, encoding="utf-8"):
|
||
m = re.match(r'^### (\d{2}) — .*(?:— files in (sec\d\d(?:\+\d\d)*))?\s*$', line)
|
||
if line.startswith("### "):
|
||
num = line.split(" — ")[0].replace("### ", "").strip()
|
||
cur = num
|
||
out.setdefault(num, {"files": [], "xref": []})
|
||
if re.search(r'— files in sec', line) and not out[num]["files"]:
|
||
tgts = re.findall(r'sec(\d\d)', line)
|
||
out[num]["xref"].extend(("sec" + t, None, None) for t in tgts)
|
||
continue
|
||
if line.startswith("## "):
|
||
cur = None
|
||
continue
|
||
if cur is None:
|
||
continue
|
||
# legacy prose: Файлы: `path1`, `path2`.
|
||
mp = re.match(r'Файлы:\s*(.+)$', line)
|
||
if mp:
|
||
for path in re.findall(r'`([^`]+)`', mp.group(1)):
|
||
nm = os.path.basename(path)
|
||
out[cur]["files"].append(("@" + path, "", ""))
|
||
continue
|
||
if not line.startswith("|"):
|
||
continue
|
||
cells = [c.strip() for c in line.strip("|").split("|")]
|
||
if not cells or cells[0] in ("file", "") or cells[0].startswith(":"):
|
||
continue
|
||
first = cells[0]
|
||
# legacy 5-col rows: | EN/RU | file | size | source | verify |
|
||
if first in ("EN", "RU", "DE", "ES", "FR") and len(cells) > 2 and re.search(r'\.\w{2,4}$', cells[1]):
|
||
out[cur]["files"].append((cells[1], cells[2], cells[3] if len(cells) > 3 else ""))
|
||
continue
|
||
m = re.match(r'— \(', first)
|
||
if m:
|
||
tfile = None
|
||
fm = re.search(r'(\d\d-[\w.\-]+\.\w{2,4}|\d\d-[\w.\-]+\*)', first)
|
||
if fm:
|
||
tfile = fm.group(1)
|
||
secmm = re.search(r'sec\d\d', first)
|
||
out[cur]["xref"].append((secmm.group(0) if secmm else "sec??", None, tfile))
|
||
continue
|
||
for sec_m, item_m in re.findall(r'sec(\d\d)(?: #(\d{2}))?', first):
|
||
out[cur]["xref"].append(("sec" + sec_m, item_m or None, None))
|
||
continue
|
||
if re.search(r'\.\w{2,4}$', first):
|
||
size = cells[1] if len(cells) > 1 else ""
|
||
out[cur]["files"].append((first, size, cells[2] if len(cells) > 2 else ""))
|
||
return out
|
||
|
||
def size_from_str(s):
|
||
m = re.match(r'([\d.]+)\s*(MB|GB|KB|pp)', s)
|
||
if not m:
|
||
return None
|
||
v = float(m.group(1))
|
||
return {"KB": v*1024, "MB": v*1048576, "GB": v*1073741824}[m.group(2)]
|
||
|
||
def card_path(sec, card):
|
||
p = os.path.join(BASE, "sections", sec, card)
|
||
if os.path.exists(p):
|
||
return p
|
||
# INDEX name drift: match by item-number prefix
|
||
num = card[:2]
|
||
d = os.path.join(BASE, "sections", sec)
|
||
cands = [f for f in os.listdir(d) if f.startswith(num + "-") and f.endswith(".md")]
|
||
if len(cands) == 1:
|
||
return os.path.join(d, cands[0])
|
||
return None
|
||
|
||
def card_ru_title(sec, card, _depth=0):
|
||
p = card_path(sec, card)
|
||
if p is None:
|
||
return ""
|
||
s = open(p, encoding="utf-8").read()
|
||
m = re.search(r'## Editions — RU\s*\n(.*?)(?=\n## )', s, re.S)
|
||
if m:
|
||
for r in [x for x in m.group(1).splitlines() if x.strip().startswith("|")]:
|
||
cells = [c.strip() for c in r.strip("|").split("|")]
|
||
if not cells or all(set(c) <= set("-: ") for c in cells):
|
||
continue # separator
|
||
if cells[0].lower().endswith("title") or cells[0].startswith("Title"):
|
||
continue # header
|
||
if cells[0] in ("EN", "RU", "DE", "ES", "FR", "IT"):
|
||
continue # lang-first table: no title column
|
||
tt = cells[0].strip("«» ")
|
||
if tt.startswith("—"):
|
||
return "— нет"
|
||
if tt:
|
||
return tt[:90]
|
||
# RU title via cross-ref in Status line (cross-section / xref cards)
|
||
if _depth < 2:
|
||
stat = re.search(r'\*\*Status:\*\*([^\n]*)', s)
|
||
if stat:
|
||
mm = (re.search(r'cross-ref (?:IWT: )?sec(\d\d) #(\d{2})', stat.group(1))
|
||
or re.search(r'see `../(\d\d)-[\w\-]+/`(?:\s*#|item )(\d{2})', stat.group(1)))
|
||
if mm:
|
||
full = SECNUM_TO_FULL.get(mm.group(1))
|
||
if full:
|
||
_name, items = idxs[full]
|
||
it = items.get(mm.group(2))
|
||
if it:
|
||
sub = card_ru_title(full, it["card"], _depth + 1)
|
||
if sub and sub != "— нет":
|
||
return f"{sub} (sec{mm.group(1)})"
|
||
return ""
|
||
|
||
def file_links(sec, files, ddir):
|
||
out = []
|
||
for name, size_s, _src in files:
|
||
if name.startswith("@"):
|
||
path = name[1:]
|
||
if not os.path.exists(os.path.join(BASE, path)):
|
||
import glob as _glob
|
||
cands = _glob.glob(os.path.join(BASE, "downloads", "*", os.path.basename(path)))
|
||
if len(cands) == 1:
|
||
path = os.path.relpath(cands[0], BASE)
|
||
else:
|
||
continue
|
||
nm = os.path.basename(path)
|
||
secnum = os.path.basename(os.path.dirname(path))[:2]
|
||
lang = "RU" if re.search(r'-ru[.\-]', nm) else ("EN" if re.search(r'-en[.\-]', nm) else "")
|
||
sz = os.path.getsize(os.path.join(BASE, path))
|
||
out.append(f"[{lang} {nm.rsplit('.', 1)[-1]} {human(sz)} →sec{secnum}]({path})")
|
||
continue
|
||
lang = "RU" if re.search(r'-ru[.\-]', name) else ("EN" if re.search(r'-en[.\-]', name) else "")
|
||
sz = size_from_str(size_s) or os.path.getsize(os.path.join(BASE, "downloads", ddir, name))
|
||
fmt = name.rsplit(".", 1)[-1]
|
||
tag = f"{lang} {fmt} {human(sz)}"
|
||
out.append(f"[{tag}](downloads/{ddir}/{name})")
|
||
return out
|
||
|
||
def mklink(name, ddir, sec, sz):
|
||
lang = "RU " if re.search(r'-ru[.\-]', name) else ("EN " if re.search(r'-en[.\-]', name) else "")
|
||
label = f"{lang}{name.rsplit('.', 1)[-1]} {human(sz) if sz else ''} →sec{sec}".replace(" ", " ").strip()
|
||
return f"[{label}](downloads/{ddir}/{name})"
|
||
|
||
def mkpath(name, sz_hint=""):
|
||
"""build (label, relpath) for a file name or @path entry"""
|
||
if name.startswith("@"):
|
||
path = name[1:]
|
||
import glob as _glob
|
||
if not os.path.exists(os.path.join(BASE, path)):
|
||
cands = _glob.glob(os.path.join(BASE, "downloads", "*", os.path.basename(path)))
|
||
if len(cands) == 1:
|
||
path = os.path.relpath(cands[0], BASE)
|
||
else:
|
||
return None
|
||
else:
|
||
return None # caller handles plain names
|
||
nm = os.path.basename(path)
|
||
secnum = os.path.basename(os.path.dirname(path))[:2]
|
||
lang = "RU " if re.search(r'-ru[.\-]', nm) else ("EN " if re.search(r'-en[.\-]', nm) else "")
|
||
sz = os.path.getsize(os.path.join(BASE, path))
|
||
return f"{lang}{nm.rsplit('.', 1)[-1]} {human(sz)} →sec{secnum}", path
|
||
|
||
def main():
|
||
# pre-parse all manifests (needed for cross-section resolution)
|
||
manifs = {sec: parse_manifest(sec) for sec in SECS}
|
||
global idxs
|
||
idxs = {}
|
||
for sec in SECS:
|
||
idxs[sec] = parse_index(sec)
|
||
out = []
|
||
out.append(f"# ISAP Zurich — Reading List: RU editions & downloaded files ({DATE})")
|
||
out.append("")
|
||
out.append("> Все 12 секций. После каждой книги — ссылки на скачанные файлы; у книг из")
|
||
out.append("> предыдущих секций файлы лежат в канонической секции (тег `secNN` = где лежит файл).")
|
||
out.append("> `✅` = RU-издание найдено, `🔶` = частично (глава/статья), `❌` = RU нет (verified).")
|
||
out.append("> Карточка каждой книги — по клику на название. Детали изданий/каталогов — в карточке,")
|
||
out.append("> провалы поиска — в `sections/NN/MISSING.md`.")
|
||
out.append("")
|
||
grand_i = grand_f = grand_l = 0
|
||
for sec in SECS:
|
||
name, items = idxs[sec]
|
||
mf = manifs[sec]
|
||
ddir = dldir(sec)
|
||
n = int(sec[:2])
|
||
dfiles = [f for f in os.listdir(os.path.join(BASE, "downloads", ddir)) if f != "MANIFEST.md"]
|
||
dfiles_n = len(dfiles)
|
||
f_en = sum(1 for f in dfiles if is_en(f))
|
||
f_ru = sum(1 for f in dfiles if is_ru(f))
|
||
f_other = dfiles_n - f_en - f_ru
|
||
stat = {e: 0 for e in "✅🔶❌"}
|
||
new_items = [it for it in items.values() if not it.get("is_xref")]
|
||
for it in new_items:
|
||
stat[it["status"]] = stat.get(it["status"], 0) + 1
|
||
en_dl = ru_dl = 0
|
||
for num, it in items.items():
|
||
if it.get("is_xref"):
|
||
continue
|
||
fns = [f[0] for f in mf.get(num, {"files": []}).get("files", [])]
|
||
if any(is_en(f) for f in fns): en_dl += 1
|
||
if any(is_ru(f) for f in fns): ru_dl += 1
|
||
xref_n = len(items) - len(new_items)
|
||
comp = " (все NEW)" if xref_n == 0 else f": {len(new_items)} NEW + {xref_n} cross-ref"
|
||
ru_found = stat["✅"] + stat["🔶"]
|
||
ru_dl_s = f"{ru_dl} из {ru_found}" if ru_found else "—"
|
||
out.append(f"## {n}. {name}")
|
||
out.append("")
|
||
out.append(f"{len(items)} книг{comp}\\")
|
||
out.append(f"RU-издание: ✅ {stat['✅']} найдено · 🔶 {stat['🔶']} частично · ❌ {stat['❌']} нет\\")
|
||
out.append(f"EN-оригинал скачан: {en_dl} из {len(new_items)}\\")
|
||
out.append(f"RU-перевод скачан: {ru_dl_s}\\")
|
||
out.append(f"Файлов: {dfiles_n} (EN {f_en} / RU {f_ru} / other {f_other})")
|
||
out.append("")
|
||
out.append("| # | Book | Author | RU edition | Files |")
|
||
out.append("|---|------|--------|------------|-------|")
|
||
nfiles_xref = 0
|
||
nlinks = 0
|
||
for num in sorted(items):
|
||
it = items[num]
|
||
md = mf.get(num, {"files": [], "xref": []})
|
||
# files: local + resolved xref
|
||
links = file_links(sec, md["files"], ddir)
|
||
resolved_secs = set()
|
||
for tsec, titem, tfile in md["xref"]:
|
||
tsec2 = tsec[-2:]
|
||
tmf = manifs.get(SECNUM_TO_FULL.get(tsec2, ""), {})
|
||
tinfo = tmf.get(titem, {}) if titem else {}
|
||
tdir = SECNUM_TO_DIR.get(tsec2)
|
||
if tfile and tdir:
|
||
import glob as _glob
|
||
pats = [os.path.join(BASE, "downloads", tdir, tfile)] if "*" not in tfile else _glob.glob(os.path.join(BASE, "downloads", tdir, tfile))
|
||
for pth in pats:
|
||
if not os.path.exists(pth):
|
||
continue
|
||
nm = os.path.basename(pth)
|
||
links.append(mklink(nm, tdir, tsec2, os.path.getsize(pth)))
|
||
nfiles_xref += 1
|
||
resolved_secs.add(tsec2)
|
||
continue
|
||
elif tdir and tinfo.get("files"):
|
||
for name, size_s, _s in tinfo["files"]:
|
||
if name.startswith("@"):
|
||
mp = mkpath(name)
|
||
if mp:
|
||
label, path = mp
|
||
links.append(f"[{label}]({path})")
|
||
nfiles_xref += 1
|
||
else:
|
||
lang = "RU" if re.search(r'-ru[.\-]', name) else ("EN" if re.search(r'-en[.\-]', name) else "")
|
||
sz = size_from_str(size_s) or 0
|
||
links.append(f"[{lang} {name.rsplit('.',1)[-1]} {human(sz)} →sec{tsec2[-2:]}](downloads/{tdir}/{name})")
|
||
nfiles_xref += 1
|
||
resolved_secs.add(tsec2)
|
||
elif tdir and tsec2 not in resolved_secs:
|
||
# target item number unknown -> link to the section manifest
|
||
links.append(f"[→sec{tsec2[-2:]}](downloads/{tdir}/MANIFEST.md)")
|
||
# RU title from card
|
||
ru = card_ru_title(sec, it["card"])
|
||
status = it["status"]
|
||
ru_cell = f"{status} {ru}" if ru and status in "✅🔶" else (status if not ru else ru)
|
||
if status == "❌":
|
||
ru_cell = "—"
|
||
# drop bare fallback links if a resolved file link for the same section exists
|
||
def bare_sec(lk):
|
||
m = re.match(r'^\[→sec(\d\d)\]\(', lk)
|
||
return m.group(1) if m else None
|
||
resolved_secs2 = set()
|
||
for lk in links:
|
||
m = re.search(r'→sec(\d\d)\](?!.*MANIFEST)', lk)
|
||
if m and not lk.startswith(f"[→sec{m.group(1)}]"):
|
||
resolved_secs2.add(m.group(1))
|
||
links = [lk for lk in links if bare_sec(lk) is None or bare_sec(lk) not in resolved_secs2]
|
||
links_cell = " · ".join(dict.fromkeys(links)) if links else "—"
|
||
nlinks += len(dict.fromkeys(links))
|
||
# Book link: use real card path if INDEX drifted
|
||
cp = card_path(sec, it["card"])
|
||
card_link = f"sections/{sec}/{os.path.basename(cp)}" if cp else f"sections/{sec}/{it['card']}"
|
||
out.append(f"| {num} | [{it['title']}]({card_link}) | {it['author']} | {ru_cell} | {links_cell} |")
|
||
grand_i += len(items)
|
||
grand_f += dfiles_n
|
||
grand_l += nlinks
|
||
out.append("")
|
||
out.append("---")
|
||
out.append(f"_Итого: {grand_i} позиций · {grand_f} файлов · {grand_l} ссылок (включая cross-section) · {DATE}._")
|
||
out.append("")
|
||
text = "\n".join(out)
|
||
if "--check" not in sys.argv:
|
||
open(os.path.join(BASE, "READING-LIST.md"), "w", encoding="utf-8").write(text)
|
||
print(f"READING-LIST.md written: {len(text)} chars, {text.count(chr(10))} lines")
|
||
else:
|
||
print(text[:2000])
|
||
|
||
if __name__ == "__main__":
|
||
main()
|