content audit sec01-06: fixes (audit_content.py + 12 file/card corrections)

- tools/audit_content.py: per-file vs item-title triage (SUSPECT/SMALL/NO-TEXT/NO-CYR)
- sec01: .part removed; 10 files restored .pdf ext (LFS rename); #04/#19/#27 verify notes
- sec02 #16: corrupted double-encoded txt -> clean flib fb2 b/844509; #17/#22 doc verified via catdoc
- sec03 #11: -b epub = Spanish von Franz (Paidós 1983, forged EN OPF) -> bonus label
- sec03 #42: КАРО 2012 'Irish Tales' = EN reader (not RU, not the listed book) -> bonus rename
- sec03 #56: Onians b/c/d/e = Cambridge 24-87KB previews -> labeled
- sec06 #18: filename 1981->1989 (Perera 'Descent to the Goddess' 1989; 1981 = other book)
- sec06 #28: ia Zimmer OCR layer = foreign Devanagari text; images = RKP (p.100 verified)
- sec04 #30: buksmart 2020 = RU/FR bilingual w/ VeryPDF watermarks -> note
- linter: azw3/mobi ext + underscore in fname regex; manifest cross-check ignores off-disk rows;
  8 legacy cards section-order fixed (CR before DL)
- AGENTS.md: 'Content audit (2026-09-25)' lessons section
- check-md: 0/219
This commit is contained in:
Dmitry Kokorin 2026-09-26 01:32:02 +03:00
parent 85c41acd3c
commit f5cb34195e
44 changed files with 20976 additions and 83 deletions

103
tools/audit_content.py Normal file
View file

@ -0,0 +1,103 @@
#!/usr/bin/env python3
"""audit_content.py — content-triage of ALL downloaded files vs their item title.
For each file NN-... in downloads/0N-*/: expected title = per-item header
'### NN — <title>' of that section's MANIFEST. Extract a text sample
(pdf pdftotext -l 3 / fb2 body / epub first xhtml / djvu djvutxt / doc strings)
and score title-token overlap. Flags: SUSPECT (0 overlap), SMALL, NO-TEXT.
RU files (name contains '-ru') get a light check (Cyrillic present, non-empty).
Run: python3 tools/audit_content.py [sec] (sec = 01..06, default all)
"""
import os, re, sys, subprocess, json, html
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DL = os.path.join(BASE, 'downloads')
STOP = {'the','and','for','with','from','into','upon','about','its','his','her','vol','vols','chaps','ch','part','study','essays','essay','selected','works','collected','complete','book','pages','original','edition','second','first'}
def tokens(title):
t = re.sub(r'\(.*?\)', ' ', title)
t = re.sub(r'[^a-zа-яё0-9 ]', '', t.lower())
return [w for w in t.split() if len(w) >= 4 and w not in STOP]
def item_titles(sec_dir):
mf = os.path.join(DL, sec_dir, 'MANIFEST.md')
t = open(mf, encoding='utf8').read()
part = t.split('## Not downloadable')[0]
out = {}
for m in re.finditer(r'^### (\d+) — (.+?) — (.+)$', part, re.M):
out[m.group(1)] = m.group(2).strip()
# items with '0 files' headers too (no local files, skip)
return out
def sample(path):
ext = path.rsplit('.', 1)[-1].lower()
try:
if ext == 'pdf':
r = subprocess.run(['pdftotext', '-l', '3', path, '-'], capture_output=True, timeout=120)
txt = r.stdout.decode('utf8', errors='ignore')
if not txt.strip():
r = subprocess.run(['pdftotext', path, '-'], capture_output=True, timeout=300)
txt = r.stdout.decode('utf8', errors='ignore')[:3000]
return txt[:3000]
if ext == 'fb2':
raw = open(path, encoding='utf8', errors='ignore').read()
body = re.search(r'<body>(.*)', raw, re.S)
txt = re.sub(r'<[^>]+>', ' ', body.group(1) if body else raw[:50000])
return re.sub(r'\s+', ' ', txt)[:3000]
if ext == 'epub':
import zipfile
z = zipfile.ZipFile(path)
names = [n for n in z.namelist() if re.search(r'\.(x?html?|htm)$', n) and 'nav' not in n.lower()]
txt = ''
for n in names[:8]:
txt += z.read(n).decode('utf8', errors='ignore')
if len(txt) > 4000: break
return re.sub(r'<[^>]+>', ' ', txt)[:3000]
if ext == 'djvu':
r = subprocess.run(['djvutxt', path, '-'], capture_output=True, timeout=300)
txt = r.stdout.decode('utf8', errors='ignore')
return txt[:3000]
if ext in ('doc', 'txt', 'html', 'mobi'):
r = subprocess.run(['strings', '-n', '6', path], capture_output=True, timeout=60)
return r.stdout.decode('utf8', errors='ignore')[:3000]
except Exception as e:
return f"__ERR__ {e}"
return ''
def main():
secs = sys.argv[1:] or sorted(d for d in os.listdir(DL) if re.match(r'^0\d-', d))
results = []
for sec in secs:
titles = item_titles(sec)
files = [f for f in os.listdir(os.path.join(DL, sec)) if f != 'MANIFEST.md' and re.match(r'^\d\d-', f)]
for f in sorted(files):
num = f[:2]
title = titles.get(num, '')
if not title:
results.append((sec, f, 'NO-TITLE', 0, 0, ''))
continue
tok = tokens(title)
is_ru = '-ru' in f or re.search(r'-(t\d-ru|ru-)', f)
size = os.path.getsize(os.path.join(DL, sec, f))
s = sample(os.path.join(DL, sec, f))
low = s.lower()
hits = sum(1 for w in tok if w in low) if tok else 0
score = hits / len(tok) if tok else 0
flags = []
if s.startswith('__ERR__'):
flags.append('ERR')
elif is_ru:
if not re.search(r'[а-яё]', s): flags.append('NO-CYR')
else:
if not s.strip(): flags.append('NO-TEXT')
elif score == 0 and tok: flags.append('SUSPECT')
if size < 200000 and f.endswith('.pdf'): flags.append('SMALL')
if size < 80000 and not f.endswith('.pdf'): flags.append('SMALL')
if flags:
results.append((sec, f, ','.join(flags), round(score, 2), size, s[:150].replace('\n', ' ')))
for sec, f, fl, sc, sz, snip in results:
print(f"{sec}/{f}\t{fl}\tscore={sc}\t{sz}\t{snip[:80]}")
print(f"\nTOTAL flagged: {len(results)}", file=sys.stderr)
if __name__ == '__main__':
main()

View file

@ -63,9 +63,9 @@ def lint_card(path, manifest_files):
# Downloads table cross-check
dm = re.search(r'^## Downloads\n(.*?)(?=^## |\Z)', c, re.M | re.S)
if dm:
rows = re.findall(r'^\|\s*(EN|RU\??|\?)\s*\|\s*([0-9]{2}-[a-zA-Z0-9\.\-]+)\s*\|', dm.group(1), re.M)
rows = re.findall(r'^\|\s*(EN|RU\??|\?)\s*\|\s*([0-9]{2}-[a-zA-Z0-9_\.\-]+)\s*\|', dm.group(1), re.M)
disk = set(os.path.basename(f) for f in glob.glob(f"downloads/{sec_of(path)}/*"))
manifest = set(f for f, _, _ in manifest_files.get(num, []))
manifest = set(f for f, _, _ in manifest_files.get(num, []) if f in disk) # skip legacy/renamed rows
for lang, fname in rows:
if fname not in disk:
issues.append(f'Downloads file not on disk: {fname}')

View file

@ -19,7 +19,7 @@ def parse_manifest_files(manifest_path):
if not os.path.exists(manifest_path):
return items
c = open(manifest_path, encoding='utf-8').read()
fname_re = r'([0-9]{2}-[a-zA-Z0-9\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip))'
fname_re = r'([0-9]{2}-[a-zA-Z0-9_\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip|mobi|azw3))'
# sec01/02: table rows | file | size | source | note |
for m in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*([^|]+)\|\s*([^|]+)\|', c, re.M):
f, size, src = m.group(1), m.group(2).strip(), m.group(3).strip()