#!/usr/bin/env python3 """RU-title candidates: EN->RU token translation + match against known RU title DB. Usage: rutitles.py build # rebuild data/ru-titles.jsonl from data/catalogs/* rutitles.py translate "The Symbolic Quest" # EN title -> RU token bag (+unknown EN words) rutitles.py match "символический поиск" [N] # top-N fuzzy matches in the RU title DB rutitles.py both "The Symbolic Quest" # translate + match in one go DB sources (data/catalogs/*.md, title lines after the header): cogito Jungian series (518), litres 140409, OPP/MISP list. Extend: `rutitles.py addflib ` appends flibusta.is authorall book titles. Matching: token Jaccard over normalized lowercase tokens (stopwords removed). Order-agnostic — RU word order irrelevant. This is a DIFF against the known RU market universe, NOT a translation check — verify identity on real hits. """ import json, os, re, sys, difflib try: import pymorphy3 _MORPH = pymorphy3.MorphAnalyzer() except Exception: _MORPH = None def lemma(w): if _MORPH and w.isalpha() and "а" <= w[0] <= "я": try: return _MORPH.parse(w)[0].normal_form except Exception: return w return w ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) DICT = os.path.join(ROOT, "data", "ru-dict.tsv") DB = os.path.join(ROOT, "data", "ru-titles.jsonl") CATS = os.path.join(ROOT, "data", "catalogs") STOP = set("""в вв во не на наа наа не при по от из до к к к к у за для с со и и и или но а же ли бы то что это тот эта этот эта эти те они он она мы вы они ее его их наш твой мой ваш их их мой твой его ее их наш ваш свой чей какой какой какие какое какое каких каких какие какие какой какой какой какой какая какое какие сколько сколько многие несколько весь все всё весь все всё весь все всё каждый каждый каждая каждое какие любой любая любое любые любой любая любое любые некий некая некое некие иной иная иное иные иной иная иное иные чужой чужая чужое чужие другой другая другое другие другой другая другое другие сам сама само сами сами сами сам сама само сами сам сам сама само сами""".split()) STOP = set(w for w in STOP if len(w) > 1) STOP |= {"в", "на", "во", "о", "об", "с", "со", "из", "от", "по", "к", "у", "за", "и", "а", "но", "не", "что", "который", "которая", "которое", "которые", "для", "про", "же", "ли"} def load_dict(): phrases, words = [], {} for line in open(DICT, encoding="utf-8"): line = line.rstrip("\n") if not line or line.startswith("#") or "\t" not in line: continue k, v = line.split("\t", 1) k, v = k.strip(), v.strip() if " " in k: phrases.append((k.lower(), v)) else: words[k.lower()] = v phrases.sort(key=lambda kv: -len(kv[0])) return phrases, words def norm_tokens(s): s = s.lower() s = re.sub(r"[«»\"'’\-–—().,;:!?/\\|]", " ", s) toks = [t for t in s.split() if t] return [lemma(t) for t in toks if t not in STOP] PHR, WORD = load_dict() def translate(title): """EN title -> (ru_token_bag, [unknown_en_words])""" text = title.lower().strip() text = re.sub(r"[«»\"'’\-–—().,;:!?/\\|]", " ", text) tokens, unknown = [], [] # phrases first (longest) for ph, ru in PHR: pat = re.compile(r"(?= 0.25: scored.append((j, r)) scored.sort(key=lambda x: -x[0]) for j, r in scored[:n]: print("%.2f [%s] %s" % (j, r["src"][:12], r["title"][:100])) def both(title, n=10): tokens, unknown = translate(title) print("RU tokens:", " | ".join(tokens)) if unknown: print("UNTRANSLATED:", " ".join(unknown)) if tokens: print("--- matches ---") match(" ".join(tokens), n) def addflib(author_id): import subprocess out = subprocess.run([sys.executable, os.path.join(ROOT, "tools", "flib.py"), "authorall", str(author_id)], capture_output=True, text=True, timeout=600).stdout rows = [json.loads(l) for l in open(DB, encoding="utf-8") if l.strip()] if os.path.exists(DB) else [] seen = set(r["norm"] for r in rows) added = 0 for line in out.splitlines(): # flib.py authorall: "b/ — [ [пер. X]]" m = re.match(r"^b/(\d+)\s+(\d+|\?)\s+\S+\s+([^—]+?)\s*—\s*(.+)$", line) if not m: continue title = re.sub(r"\s*\[пер\. .+\]$", "", m.group(4)).strip() key = " ".join(norm_tokens(title)) if not key or key in seen: continue seen.add(key) rows.append({"title": title, "norm": key, "src": "flib:a%s" % author_id}) added += 1 with open(DB, "w", encoding="utf-8") as f: for r in rows: f.write(json.dumps(r, ensure_ascii=False) + "\n") print("flib a/%s: +%d titles (DB now %d)" % (author_id, added, len(rows))) def main(): if len(sys.argv) < 2: print(__doc__); sys.exit(1) cmd = sys.argv[1] if cmd == "build": build() elif cmd == "translate": tokens, unknown = translate(" ".join(sys.argv[2:])) print(" | ".join(tokens)) if unknown: print("UNTRANSLATED:", " ".join(unknown)) elif cmd == "match": args = sys.argv[2:] n = 10 if args and args[-1].isdigit(): n, args = int(args[-1]), args[:-1] match(" ".join(args), n) elif cmd == "both": both(" ".join(sys.argv[2:]), int(sys.argv[3]) if len(sys.argv) > 3 else 10) elif cmd == "addflib": addflib(sys.argv[2]) else: print(__doc__); sys.exit(1) if __name__ == "__main__": main()