jung/tools/audit_content.py
Dmitry Kokorin f5cb34195e content audit sec01-06: fixes (audit_content.py + 12 file/card corrections)
- tools/audit_content.py: per-file vs item-title triage (SUSPECT/SMALL/NO-TEXT/NO-CYR)
- sec01: .part removed; 10 files restored .pdf ext (LFS rename); #04/#19/#27 verify notes
- sec02 #16: corrupted double-encoded txt -> clean flib fb2 b/844509; #17/#22 doc verified via catdoc
- sec03 #11: -b epub = Spanish von Franz (Paidós 1983, forged EN OPF) -> bonus label
- sec03 #42: КАРО 2012 'Irish Tales' = EN reader (not RU, not the listed book) -> bonus rename
- sec03 #56: Onians b/c/d/e = Cambridge 24-87KB previews -> labeled
- sec06 #18: filename 1981->1989 (Perera 'Descent to the Goddess' 1989; 1981 = other book)
- sec06 #28: ia Zimmer OCR layer = foreign Devanagari text; images = RKP (p.100 verified)
- sec04 #30: buksmart 2020 = RU/FR bilingual w/ VeryPDF watermarks -> note
- linter: azw3/mobi ext + underscore in fname regex; manifest cross-check ignores off-disk rows;
  8 legacy cards section-order fixed (CR before DL)
- AGENTS.md: 'Content audit (2026-09-25)' lessons section
- check-md: 0/219
2026-09-26 01:32:02 +03:00

103 lines
4.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""audit_content.py — content-triage of ALL downloaded files vs their item title.
For each file NN-... in downloads/0N-*/: expected title = per-item header
'### NN — <title>' of that section's MANIFEST. Extract a text sample
(pdf pdftotext -l 3 / fb2 body / epub first xhtml / djvu djvutxt / doc strings)
and score title-token overlap. Flags: SUSPECT (0 overlap), SMALL, NO-TEXT.
RU files (name contains '-ru') get a light check (Cyrillic present, non-empty).
Run: python3 tools/audit_content.py [sec] (sec = 01..06, default all)
"""
import os, re, sys, subprocess, json, html
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DL = os.path.join(BASE, 'downloads')
STOP = {'the','and','for','with','from','into','upon','about','its','his','her','vol','vols','chaps','ch','part','study','essays','essay','selected','works','collected','complete','book','pages','original','edition','second','first'}
def tokens(title):
t = re.sub(r'\(.*?\)', ' ', title)
t = re.sub(r'[^a-zа-яё0-9 ]', '', t.lower())
return [w for w in t.split() if len(w) >= 4 and w not in STOP]
def item_titles(sec_dir):
mf = os.path.join(DL, sec_dir, 'MANIFEST.md')
t = open(mf, encoding='utf8').read()
part = t.split('## Not downloadable')[0]
out = {}
for m in re.finditer(r'^### (\d+) — (.+?) — (.+)$', part, re.M):
out[m.group(1)] = m.group(2).strip()
# items with '0 files' headers too (no local files, skip)
return out
def sample(path):
ext = path.rsplit('.', 1)[-1].lower()
try:
if ext == 'pdf':
r = subprocess.run(['pdftotext', '-l', '3', path, '-'], capture_output=True, timeout=120)
txt = r.stdout.decode('utf8', errors='ignore')
if not txt.strip():
r = subprocess.run(['pdftotext', path, '-'], capture_output=True, timeout=300)
txt = r.stdout.decode('utf8', errors='ignore')[:3000]
return txt[:3000]
if ext == 'fb2':
raw = open(path, encoding='utf8', errors='ignore').read()
body = re.search(r'<body>(.*)', raw, re.S)
txt = re.sub(r'<[^>]+>', ' ', body.group(1) if body else raw[:50000])
return re.sub(r'\s+', ' ', txt)[:3000]
if ext == 'epub':
import zipfile
z = zipfile.ZipFile(path)
names = [n for n in z.namelist() if re.search(r'\.(x?html?|htm)$', n) and 'nav' not in n.lower()]
txt = ''
for n in names[:8]:
txt += z.read(n).decode('utf8', errors='ignore')
if len(txt) > 4000: break
return re.sub(r'<[^>]+>', ' ', txt)[:3000]
if ext == 'djvu':
r = subprocess.run(['djvutxt', path, '-'], capture_output=True, timeout=300)
txt = r.stdout.decode('utf8', errors='ignore')
return txt[:3000]
if ext in ('doc', 'txt', 'html', 'mobi'):
r = subprocess.run(['strings', '-n', '6', path], capture_output=True, timeout=60)
return r.stdout.decode('utf8', errors='ignore')[:3000]
except Exception as e:
return f"__ERR__ {e}"
return ''
def main():
secs = sys.argv[1:] or sorted(d for d in os.listdir(DL) if re.match(r'^0\d-', d))
results = []
for sec in secs:
titles = item_titles(sec)
files = [f for f in os.listdir(os.path.join(DL, sec)) if f != 'MANIFEST.md' and re.match(r'^\d\d-', f)]
for f in sorted(files):
num = f[:2]
title = titles.get(num, '')
if not title:
results.append((sec, f, 'NO-TITLE', 0, 0, ''))
continue
tok = tokens(title)
is_ru = '-ru' in f or re.search(r'-(t\d-ru|ru-)', f)
size = os.path.getsize(os.path.join(DL, sec, f))
s = sample(os.path.join(DL, sec, f))
low = s.lower()
hits = sum(1 for w in tok if w in low) if tok else 0
score = hits / len(tok) if tok else 0
flags = []
if s.startswith('__ERR__'):
flags.append('ERR')
elif is_ru:
if not re.search(r'[а-яё]', s): flags.append('NO-CYR')
else:
if not s.strip(): flags.append('NO-TEXT')
elif score == 0 and tok: flags.append('SUSPECT')
if size < 200000 and f.endswith('.pdf'): flags.append('SMALL')
if size < 80000 and not f.endswith('.pdf'): flags.append('SMALL')
if flags:
results.append((sec, f, ','.join(flags), round(score, 2), size, s[:150].replace('\n', ' ')))
for sec, f, fl, sc, sz, snip in results:
print(f"{sec}/{f}\t{fl}\tscore={sc}\t{sz}\t{snip[:80]}")
print(f"\nTOTAL flagged: {len(results)}", file=sys.stderr)
if __name__ == '__main__':
main()