sec06: key-vs-f_id discovery — 12 EN files recovered (6 rescues) + precheck v2
- USER FINDING: libgen edition files map {key: {f_id, md5}} — KEY and f_id VALUE
are DIFFERENT file records; we were downloading/prechecking by the KEY (wrong file).
- Recovered by f_id: 09-t1 (Eliade UCP 1978), 13 Sungod (Cornell 2010), 14 Solomon
(Karnac 2007), 17 God-Image (Inner Light 1992), 19 Hornung (UNC 1982), 20 Psyche in
Scripture (1995), 23 Dionysos (Bollingen LXV/2 Princeton) + backups 12/16/24/25/26/27.
All content-verified. '24 systemic mislinks' = our bug, not libgen.
- lg.py: objects[]=f,e,s,a,p,w + topics[] (file-level search; found Solomon md5) +
md5 in output. lgdl.py: bymd5 subcommand.
- precheck v2: rev-value match = OK; locator words >=2 shared = OK; words present
0 shared = BAD; hash/ISBN/empty locator = no signal (SUSPECT); digit-runs masked
before tokenization (hex 'aaae' trap).
- MANIFEST/SUMMARY/cards/AGENTS.md updated. 34 files, 608 MB (28 backup still downloading).
This commit is contained in:
parent
96c0e1190d
commit
b7ddfac17e
30 changed files with 298 additions and 137 deletions
17
tools/lg.py
17
tools/lg.py
|
|
@ -33,10 +33,14 @@ def fetch(url, tries=3):
|
|||
def main():
|
||||
q = sys.argv[1]
|
||||
limit = int(sys.argv[2]) if len(sys.argv) > 2 else 10
|
||||
url = "https://libgen.vg/index.php?" + urllib.parse.urlencode({
|
||||
"req": q, "res": "100", "dlt": "0", "ln": "0",
|
||||
"columns[]": ["t","a","s","y","p","i","l","x","sz"]
|
||||
})
|
||||
# objects[] MUST include 'f' (files): editions-only search misses files that
|
||||
# exist in libgen but are not linked to the right edition (verified 2026-09-25:
|
||||
# Solomon "Self in Transformation" found by file search, missing in edition search).
|
||||
params = {"req": q, "res": "100"}
|
||||
params["columns[]"] = ["t","a","s","y","p","i"]
|
||||
params["objects[]"] = ["f","e","s","a","p","w"]
|
||||
params["topics[]"] = ["l","c","f","a","m","r","s"]
|
||||
url = "https://libgen.vg/index.php?" + urllib.parse.urlencode(params, doseq=True)
|
||||
html = fetch(url)
|
||||
# pagination hint
|
||||
m = re.search(r'page=(\d+)', html)
|
||||
|
|
@ -52,8 +56,11 @@ def main():
|
|||
if len(tds) < 6:
|
||||
continue
|
||||
# first cell is usually a link with id, title is the big cell
|
||||
title = max(tds, key=len) if tds else ""
|
||||
line = " | ".join(tds[:9])
|
||||
# file rows: surface the md5 (ads.php link) — feed to tools/lgdl.py bymd5
|
||||
md5s = re.findall(r'ads\.php\?md5=([0-9a-f]{32})', r)
|
||||
if md5s:
|
||||
line += f" [md5={md5s[0]}]"
|
||||
print(line[:300])
|
||||
count += 1
|
||||
if count >= limit:
|
||||
|
|
|
|||
148
tools/lgdl.py
148
tools/lgdl.py
|
|
@ -16,11 +16,16 @@ Usage:
|
|||
tools/lgdl.py precheck <file_id> <edition_id> [expected_pages]
|
||||
|
||||
precheck (no download, ~2 libgen calls): file record vs the target edition.
|
||||
RED (exit 2) — reverse-map `editions` non-empty and does NOT contain the
|
||||
target edition (file belongs to another book), or locator path is a
|
||||
foreign-language batch (spa/English/fiction) for a Cyrillic book.
|
||||
YELLOW(exit 1) — reverse map empty, or size implausible vs expected pages.
|
||||
OK (exit 0) — reverse map matches target, or empty with plausible locator.
|
||||
The ONLY reliable ownership signal is the file `locator` (the importer's
|
||||
original Windows path — it names the real book). libgen's `file.editions`
|
||||
reverse-map uses internal IDs that do NOT resolve via object=e (they return
|
||||
unrelated books), so it is logged as a hint but never gates the verdict.
|
||||
RED (exit 2) — locator non-empty and shares NO content word with the target
|
||||
title/author (locator names a different book), or a foreign-language
|
||||
batch path for a Cyrillic target.
|
||||
OK (exit 0) — locator non-empty and shares content words with the target.
|
||||
SUSPECT(exit 1) — locator empty (no signal), too few target tokens to judge,
|
||||
or size implausible vs expected pages. ALWAYS post-verify content.
|
||||
`dl` with the optional 4th arg runs precheck first and aborts on RED.
|
||||
|
||||
Verification on dl: final size must equal libgen filesize; magic-byte sniff
|
||||
|
|
@ -212,64 +217,100 @@ def download(f_id, out_dir, name, tries=16):
|
|||
return None
|
||||
|
||||
|
||||
# stop-ish / generic tokens to ignore when matching locator vs target
|
||||
_STOP = set("the of and in to a an for with from on by edition new second first "
|
||||
"vol volume pdf epub fb2 doc djvu mobi txt part pt series book "
|
||||
"translated translation revised updated complete full original"
|
||||
.split())
|
||||
|
||||
def _tokens(s):
|
||||
return {w for w in re.findall(r"[a-z\u0400-\u04ff]{4,}", (s or "").lower())
|
||||
if w not in _STOP}
|
||||
|
||||
def _loc_tokens(s):
|
||||
# locator words: mask alphanumeric runs containing digits (IDs/hashes/ISBNs)
|
||||
# so "d1e2f4...aaae..." does not yield fake word 'aaae'
|
||||
masked = re.sub(r"[a-z0-9]+\d[a-z0-9]*", " ", (s or "").lower())
|
||||
return _tokens(masked)
|
||||
|
||||
def precheck(f_id, e_id, expected_pages=None):
|
||||
"""Mislinked-file gate. Returns 0=OK, 1=SUSPECT, 2=BAD (see module doc)."""
|
||||
"""Mislinked-file gate. Returns 0=OK, 1=SUSPECT, 2=BAD (see module doc).
|
||||
|
||||
CRITICAL: pass the edition files-map VALUE (`f_id`), not the map key —
|
||||
the key is a different (often stale/mislinked) file record (verified
|
||||
2026-09-25: Solomon — key 92709688 = Russian book, f_id 93253044 = real book).
|
||||
|
||||
Signals:
|
||||
1) reverse map: for a real f_id record the `editions` values ARE the public
|
||||
edition IDs — value == target is a strong OK signal; for a key's record
|
||||
it contains other numbers (no match => weak negative, never hard BAD).
|
||||
2) locator: words in the locator vs target title/author words.
|
||||
- no content words (hash/ISBN/number) => no signal (SUSPECT)
|
||||
- >=2 shared words (or 1 for a short target) => OK
|
||||
- words present, none shared => BAD (names a different book)
|
||||
ALWAYS content-verify after download (ground truth).
|
||||
"""
|
||||
fi = file_info(f_id)
|
||||
time.sleep(2)
|
||||
ext, size = fi.get("extension"), int(fi.get("filesize") or 0)
|
||||
locator = fi.get("locator") or ""
|
||||
rev = {eid: int(x.get("e_id") or 0) for eid, x in (fi.get("editions") or {}).items()}
|
||||
rev = {k: v for k, v in rev.items() if v}
|
||||
verdict, reasons = "OK", []
|
||||
locator = (fi.get("locator") or "").strip()
|
||||
reasons = []
|
||||
|
||||
# 1) reverse map: who does this file BELONG to?
|
||||
rev_match = None
|
||||
rev = {k: v.get("e_id") for k, v in (fi.get("editions") or {}).items()}
|
||||
rev_match = int(e_id) in {int(v) for v in rev.values() if v}
|
||||
if rev:
|
||||
if int(e_id) in rev.values():
|
||||
rev_match = True
|
||||
reasons.append(f"reverse-map matches target e_id {e_id}")
|
||||
else:
|
||||
verdict = "BAD"
|
||||
owners = []
|
||||
for owner_eid in set(rev.values()):
|
||||
try:
|
||||
oe = jget(f"json.php?object=e&addkeys=title,author,year&ids={owner_eid}")
|
||||
o = oe.get(str(owner_eid), {})
|
||||
owners.append(f"e_id {owner_eid} «{o.get('title')}» {o.get('author')} {o.get('year')}")
|
||||
time.sleep(2)
|
||||
except Exception:
|
||||
owners.append(f"e_id {owner_eid} (?)")
|
||||
reasons.append("reverse-map points to ANOTHER edition: " + "; ".join(owners))
|
||||
reasons.append(f"reverse-map: {rev} {'MATCHES target' if rev_match else '(no target)'}")
|
||||
|
||||
try:
|
||||
ed = jget(f"json.php?object=e&addkeys=title,author&ids={e_id}").get(str(e_id), {})
|
||||
time.sleep(2)
|
||||
target = f"{ed.get('title','')} {ed.get('author','')}"
|
||||
except Exception:
|
||||
target = ""
|
||||
t_tok, l_tok = _tokens(target), _loc_tokens(locator)
|
||||
shared = t_tok & l_tok
|
||||
|
||||
if not locator or not l_tok:
|
||||
loc_signal = "none"
|
||||
reasons.append(f"locator has no content words (hash/ISBN/empty) — no signal: {locator[:90]}")
|
||||
elif len(shared) >= 2 or (len(shared) == 1 and len(t_tok) <= 3):
|
||||
loc_signal = "match"
|
||||
reasons.append(f"locator names the target (shared: {sorted(shared)}): {locator[:120]}")
|
||||
else:
|
||||
rev_match = False
|
||||
verdict = "SUSPECT" if verdict == "OK" else verdict
|
||||
reasons.append("reverse-map EMPTY (old file) — rely on locator + post-verify")
|
||||
|
||||
# 2) locator: foreign-language batch path. Only meaningful when the reverse
|
||||
# map does NOT confirm the target (matching map = the edition itself comes
|
||||
# from that batch — an origin note, not a mislink).
|
||||
low = locator.lower()
|
||||
foreign = re.search(r"(spa\\|spanish|english|epublibre|0day|\\fiction\\)", low)
|
||||
cyrillic = re.search(r"[\u0400-\u04FF]", locator)
|
||||
if locator and foreign and not cyrillic:
|
||||
if rev_match is True:
|
||||
reasons.append(f"origin note: edition comes from a foreign batch: {locator[:120]}")
|
||||
loc_signal = "mismatch"
|
||||
low = locator.lower()
|
||||
target_cyr = re.search(r"[\u0400-\u04FF]", target)
|
||||
cyrillic_loc = re.search(r"[\u0400-\u04FF]", locator)
|
||||
foreign = re.search(r"(spa\\|spanish|english|epublibre|0day|\\fiction\\)", low)
|
||||
if target_cyr and not cyrillic_loc:
|
||||
reasons.append(f"Cyrillic target but locator is non-Cyrillic: {locator[:120]}")
|
||||
elif foreign:
|
||||
reasons.append(f"locator is a foreign batch naming another book: {locator[:120]}")
|
||||
else:
|
||||
verdict = "BAD" if verdict == "OK" else verdict
|
||||
reasons.append(f"locator is a foreign batch path: {locator[:120]}")
|
||||
elif locator and rev_match is not True:
|
||||
reasons.append(f"locator ok: {locator[:120]}")
|
||||
reasons.append(f"locator names a DIFFERENT book (no shared words): {locator[:120]}")
|
||||
|
||||
# 3) size sanity vs expected pages (text formats are tiny: ~200B/pp;
|
||||
# scans: 50KB..15MB/pp)
|
||||
if rev_match and loc_signal != "mismatch":
|
||||
verdict = "OK"
|
||||
elif loc_signal == "match":
|
||||
verdict = "OK"
|
||||
elif loc_signal == "mismatch" and not rev_match:
|
||||
verdict = "BAD"
|
||||
else:
|
||||
verdict = "SUSPECT" # none/none, or contradiction (rev match vs locator mismatch)
|
||||
if rev_match and loc_signal == "mismatch":
|
||||
reasons.append("CONTRADICTION: reverse-map matches but locator names another book — verify content")
|
||||
|
||||
# size sanity vs expected pages (text ~200B/pp; scans 50KB..15MB/pp)
|
||||
if expected_pages and size:
|
||||
per_lo = 200 if ext in ("fb2", "epub", "txt", "html") else 10_000
|
||||
lo, hi = expected_pages * per_lo, expected_pages * 15_000_000
|
||||
if size < lo:
|
||||
verdict = "SUSPECT" if verdict == "OK" else verdict
|
||||
if verdict == "OK":
|
||||
verdict = "SUSPECT"
|
||||
reasons.append(f"size {size}b < {lo}b for ~{expected_pages} pp (too small — fragment?)")
|
||||
elif size > hi:
|
||||
verdict = "SUSPECT" if verdict == "OK" else verdict
|
||||
if verdict == "OK":
|
||||
verdict = "SUSPECT"
|
||||
reasons.append(f"size {size}b > {hi}b for ~{expected_pages} pp (too big — scan mismatch?)")
|
||||
|
||||
print(f"precheck f_id={f_id} vs e_id={e_id}: {verdict} [{ext}, {size}b]")
|
||||
|
|
@ -308,6 +349,17 @@ def main():
|
|||
sys.exit(precheck(f_id, e_id, pages))
|
||||
elif cmd == "md5":
|
||||
print(file_info(sys.argv[2])["md5"])
|
||||
elif cmd == "bymd5":
|
||||
# md5 -> f_id + metadata (for files found via lg.py file search)
|
||||
d = jget(f"json.php?object=f&addkeys=*&md5={sys.argv[2]}")
|
||||
if not d:
|
||||
print("not found"); sys.exit(1)
|
||||
fid = list(d.keys())[0]
|
||||
f = d[fid]
|
||||
print(f"f_id={fid} {f.get('extension')} {int(f.get('filesize') or 0)}b pages={f.get('pages')} ocr={f.get('ocr')} added={f.get('time_added')}")
|
||||
print(f"locator: {f.get('locator','')}")
|
||||
rev = {k: v.get('e_id') for k, v in (f.get('editions') or {}).items()}
|
||||
if rev: print(f"editions (hint): {rev}")
|
||||
else:
|
||||
sys.exit(__doc__)
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue