jung/tools/rutitles.py
Dmitry Kokorin 5441dddbb8 pipeline v3: RU-title candidates (rutitles+ru-dict+pymorphy3), batch sweep driver, OpenAlex/Crossref
- tools/rutitles.py: EN->RU token translation (data/ru-dict.tsv, 540+ pairs, Kogito/Castalia
  conventions) + Jaccard diff against data/ru-titles.jsonl (692 titles: Kogito 518, Litres 24,
  OPP 55, flibusta a/5272+57639+118921+122883+193690+193689). Matching is lemmatized
  (pymorphy3) — case endings handled: 'великой матери' -> 'Великая мать' 1.25.
  'The Great Mother' -> 'Великая мать' 1.25 top hit; 'The Symbolic Quest' -> correct negative.
- tools/sweep.py: batch driver for Phases 1-2 (rsl+lg+alib+flib+cogito+SearXNG-OZON-snippet
  parse+rutitles diff per item; --nlr optional). Query log data/queries.log (JSONL).
  Fixed: cogito nav-menu leak (parse bx_product_item only), libgen robot-block (lg.py curl fallback).
- tools/oa.py: OpenAlex + Crossref (Phase 0 identity/ISBN, no key; found Margaret Wilkinson,
  Karen Evers-Fahey, Symbolic Quest Princeton ISBN).
- data/sweeps/01-fundamentals/input.tsv: 16 ❌ items loaded; background sweep running.
- AGENTS.md: source 10b (oa.py), flibusta .su = reduced mirror (dropped from pipeline),
  pipeline v3 section (rutitles/oa/sweep/pymorphy3 note: pymorphy2 broken on py3.12).
- data/ru-titles.jsonl committed as the RU market universe asset.
2026-09-18 10:56:36 +03:00

205 lines
8.5 KiB
Python
Executable file
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""RU-title candidates: EN->RU token translation + match against known RU title DB.
Usage:
rutitles.py build # rebuild data/ru-titles.jsonl from data/catalogs/*
rutitles.py translate "The Symbolic Quest" # EN title -> RU token bag (+unknown EN words)
rutitles.py match "символический поиск" [N] # top-N fuzzy matches in the RU title DB
rutitles.py both "The Symbolic Quest" # translate + match in one go
DB sources (data/catalogs/*.md, title lines after the header):
cogito Jungian series (518), litres 140409, OPP/MISP list.
Extend: `rutitles.py addflib <authorId>` appends flibusta.is authorall book titles.
Matching: token Jaccard over normalized lowercase tokens (stopwords removed).
Order-agnostic — RU word order irrelevant. This is a DIFF against the known
RU market universe, NOT a translation check — verify identity on real hits.
"""
import json, os, re, sys, difflib
try:
import pymorphy3
_MORPH = pymorphy3.MorphAnalyzer()
except Exception:
_MORPH = None
def lemma(w):
if _MORPH and w.isalpha() and "а" <= w[0] <= "я":
try:
return _MORPH.parse(w)[0].normal_form
except Exception:
return w
return w
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DICT = os.path.join(ROOT, "data", "ru-dict.tsv")
DB = os.path.join(ROOT, "data", "ru-titles.jsonl")
CATS = os.path.join(ROOT, "data", "catalogs")
STOP = set("""в вв во не на наа наа не при по от из до к к к к у за для с со и и и или но а же ли бы то что это тот эта этот эта эти те они он она мы вы они ее его их наш твой мой ваш их их мой твой его ее их наш ваш свой чей какой какой какие какое какое каких каких какие какие какой какой какой какой какая какое какие сколько сколько многие несколько весь все всё весь все всё весь все всё каждый каждый каждая каждое какие любой любая любое любые любой любая любое любые некий некая некое некие иной иная иное иные иной иная иное иные чужой чужая чужое чужие другой другая другое другие другой другая другое другие сам сама само сами сами сами сам сама само сами сам сам сама само сами""".split())
STOP = set(w for w in STOP if len(w) > 1)
STOP |= {"в", "на", "во", "о", "об", "с", "со", "из", "от", "по", "к", "у", "за", "и", "а", "но",
"не", "что", "который", "которая", "которое", "которые", "для", "про", "же", "ли"}
def load_dict():
phrases, words = [], {}
for line in open(DICT, encoding="utf-8"):
line = line.rstrip("\n")
if not line or line.startswith("#") or "\t" not in line:
continue
k, v = line.split("\t", 1)
k, v = k.strip(), v.strip()
if " " in k:
phrases.append((k.lower(), v))
else:
words[k.lower()] = v
phrases.sort(key=lambda kv: -len(kv[0]))
return phrases, words
def norm_tokens(s):
s = s.lower()
s = re.sub(r"[«»\"'’\-–—().,;:!?/\\|]", " ", s)
toks = [t for t in s.split() if t]
return [lemma(t) for t in toks if t not in STOP]
PHR, WORD = load_dict()
def translate(title):
"""EN title -> (ru_token_bag, [unknown_en_words])"""
text = title.lower().strip()
text = re.sub(r"[«»\"'’\-–—().,;:!?/\\|]", " ", text)
tokens, unknown = [], []
# phrases first (longest)
for ph, ru in PHR:
pat = re.compile(r"(?<![a-z])" + re.escape(ph) + r"(?![a-z])")
m = pat.search(text)
if m:
tokens.append(ru)
text = text[:m.start()] + " " * (m.end() - m.start()) + text[m.end():]
for w in text.split():
w = w.strip()
if not w or w in STOP or not w.isalpha():
continue
if w in WORD:
tokens.append(WORD[w])
elif re.fullmatch(r"[a-z]+", w):
unknown.append(w)
# expand variants for matching: keep all
return tokens, unknown
def load_db():
if not os.path.exists(DB):
return []
return [json.loads(l) for l in open(DB, encoding="utf-8") if l.strip()]
def build():
rows, seen = [], set()
# preserve non-catalog entries (flibusta author lists added via addflib)
if os.path.exists(DB):
for l in open(DB, encoding="utf-8"):
if l.strip():
r = json.loads(l)
if r["src"].startswith("flib:"):
rows.append(r)
seen.add(r["norm"])
for fn in sorted(os.listdir(CATS)):
if not fn.endswith((".md", ".txt")):
continue
src = os.path.basename(fn)
for line in open(os.path.join(CATS, fn), encoding="utf-8", errors="replace"):
line = line.strip()
if not line or line.startswith(("#", "-", "Источник", "Дата", "##")):
continue
if len(line) < 8:
continue
key = " ".join(norm_tokens(line))
if key in seen:
continue
seen.add(key)
rows.append({"title": line, "norm": key, "src": src})
with open(DB, "w", encoding="utf-8") as f:
for r in rows:
f.write(json.dumps(r, ensure_ascii=False) + "\n")
print("DB built: %d titles from %d catalogs" % (len(rows), len(os.listdir(CATS))))
def match(needle, n=10):
nt = set(norm_tokens(needle))
if not nt:
print("empty needle"); return
rows = load_db()
scored = []
for r in rows:
rt = set(r["norm"].split())
if not rt:
continue
inter = nt & rt
union = nt | rt
j = len(inter) / len(union) if union else 0
# boost: all needle tokens present
if nt <= rt:
j += 0.25
if j >= 0.25:
scored.append((j, r))
scored.sort(key=lambda x: -x[0])
for j, r in scored[:n]:
print("%.2f [%s] %s" % (j, r["src"][:12], r["title"][:100]))
def both(title, n=10):
tokens, unknown = translate(title)
print("RU tokens:", " | ".join(tokens))
if unknown:
print("UNTRANSLATED:", " ".join(unknown))
if tokens:
print("--- matches ---")
match(" ".join(tokens), n)
def addflib(author_id):
import subprocess
out = subprocess.run([sys.executable, os.path.join(ROOT, "tools", "flib.py"), "authorall", str(author_id)],
capture_output=True, text=True, timeout=600).stdout
rows = [json.loads(l) for l in open(DB, encoding="utf-8") if l.strip()] if os.path.exists(DB) else []
seen = set(r["norm"] for r in rows)
added = 0
for line in out.splitlines():
# flib.py authorall: "b/<id> <year> <fmt> <author> — <title>[ [пер. X]]"
m = re.match(r"^b/(\d+)\s+(\d+|\?)\s+\S+\s+([^—]+?)\s*—\s*(.+)$", line)
if not m:
continue
title = re.sub(r"\s*\[пер\. .+\]$", "", m.group(4)).strip()
key = " ".join(norm_tokens(title))
if not key or key in seen:
continue
seen.add(key)
rows.append({"title": title, "norm": key, "src": "flib:a%s" % author_id})
added += 1
with open(DB, "w", encoding="utf-8") as f:
for r in rows:
f.write(json.dumps(r, ensure_ascii=False) + "\n")
print("flib a/%s: +%d titles (DB now %d)" % (author_id, added, len(rows)))
def main():
if len(sys.argv) < 2:
print(__doc__); sys.exit(1)
cmd = sys.argv[1]
if cmd == "build":
build()
elif cmd == "translate":
tokens, unknown = translate(" ".join(sys.argv[2:]))
print(" | ".join(tokens))
if unknown:
print("UNTRANSLATED:", " ".join(unknown))
elif cmd == "match":
args = sys.argv[2:]
n = 10
if args and args[-1].isdigit():
n, args = int(args[-1]), args[:-1]
match(" ".join(args), n)
elif cmd == "both":
both(" ".join(sys.argv[2:]), int(sys.argv[3]) if len(sys.argv) > 3 else 10)
elif cmd == "addflib":
addflib(sys.argv[2])
else:
print(__doc__); sys.exit(1)
if __name__ == "__main__":
main()