zcbot/skills/literature/scripts/merge_publications.py

391 lines
17 KiB
Python

"""Normalize and conservatively merge publication records from multiple backends."""
from __future__ import annotations
import argparse
import json
import re
import unicodedata
from difflib import SequenceMatcher
from pathlib import Path
from typing import Any, Iterable
EVIDENCE_LEVELS = (
"metadata_only",
"abstract_verified",
"snippet_verified",
"fulltext_verified",
)
TYPE_ALIASES = {
"article": "article",
"journal-article": "article",
"journal_article": "article",
"book": "book",
"book-chapter": "book_chapter",
"book_chapter": "book_chapter",
"chapter": "book_chapter",
"conference-paper": "conference_paper",
"conference_paper": "conference_paper",
"proceedings-article": "conference_paper",
"proceedings": "proceedings",
"thesis": "thesis",
"dissertation": "thesis",
"report": "report",
"standard": "standard",
"patent": "patent",
"preprint": "preprint",
"dataset": "dataset_publication",
"dataset_publication": "dataset_publication",
}
def normalize_doi(value: Any) -> str:
doi = str(value or "").strip().lower()
doi = re.sub(r"^(?:doi\s*:\s*|https?://(?:dx\.)?doi\.org/)", "", doi)
return doi.rstrip(".,;:)]}")
def _isbn13_from_10(value: str) -> str:
core = "978" + value[:9]
total = sum((1 if i % 2 == 0 else 3) * int(ch) for i, ch in enumerate(core))
return core + str((10 - total % 10) % 10)
def normalize_isbn(value: Any) -> str:
isbn = re.sub(r"[^0-9Xx]", "", str(value or "")).upper()
if len(isbn) == 10 and isbn[:9].isdigit() and (isbn[-1].isdigit() or isbn[-1] == "X"):
return _isbn13_from_10(isbn)
return isbn if len(isbn) == 13 and isbn.isdigit() else ""
def normalize_title(value: Any) -> str:
title = unicodedata.normalize("NFKC", str(value or "")).casefold()
title = re.sub(r"[^\w]+", " ", title, flags=re.UNICODE)
return " ".join(title.split())
def normalize_creator(value: Any) -> str:
if isinstance(value, dict):
value = value.get("name") or value.get("literal") or ""
return normalize_title(value)
def normalize_type(value: Any) -> tuple[str, str]:
raw = str(value or "").strip()
key = raw.casefold().replace(" ", "-")
return TYPE_ALIASES.get(key, "other"), raw
def _first(record: dict, *keys: str) -> Any:
for key in keys:
value = record.get(key)
if value not in (None, "", [], {}):
return value
return ""
def _creator_list(record: dict) -> list[dict[str, str]]:
raw = _first(record, "creators", "authors")
if not raw:
raw = [_first(record, "first_author", "author", "creator")]
elif not isinstance(raw, list):
raw = [raw]
creators: list[dict[str, str]] = []
for item in raw:
if isinstance(item, dict):
name = str(item.get("name") or item.get("literal") or "").strip()
role = str(item.get("role") or "author").strip()
orcid = str(item.get("orcid") or "").strip()
else:
name, role, orcid = str(item or "").strip(), "author", ""
if name:
creators.append({"name": name, "role": role, "orcid": orcid})
return creators
def _issued(record: dict) -> dict[str, int | None]:
issued = record.get("issued")
current = dict(issued) if isinstance(issued, dict) else {}
date_parts = str(record.get("publication_date") or "").split("-")
value = current.get("year") or _first(record, "publication_year", "year")
try:
year = int(value) if value not in (None, "") else int(date_parts[0]) if date_parts and date_parts[0] else None
except (TypeError, ValueError):
year = None
parts: dict[str, int | None] = {"year": year, "month": None, "day": None}
for index, key in ((1, "month"), (2, "day")):
candidate = current.get(key)
if candidate in (None, "") and len(date_parts) > index:
candidate = date_parts[index]
try:
parts[key] = int(candidate) if candidate not in (None, "") else None
except (TypeError, ValueError):
parts[key] = None
return parts
def _string_list(value: Any, normalizer=None) -> list[str]:
values = value if isinstance(value, list) else [value]
out: list[str] = []
for item in values:
normalized = normalizer(item) if normalizer else str(item or "").strip()
if normalized and normalized not in out:
out.append(normalized)
return out
def normalize_record(record: dict[str, Any], backend: str) -> dict[str, Any]:
identifiers = dict(record.get("identifiers") or {})
doi = normalize_doi(_first(identifiers, "doi") or record.get("doi"))
isbn = _string_list(_first(identifiers, "isbn") or record.get("isbn"), normalize_isbn)
issn = _string_list(_first(identifiers, "issn") or record.get("issn"))
publication_type, raw_type = normalize_type(_first(record, "type", "publication_type"))
title = str(_first(record, "title", "file_name") or "").strip()
creators = _creator_list(record)
abstract = str(record.get("abstract") or "").strip()
snippet = str(_first(record, "snippet", "md_content") or "").strip()
formats = _string_list((record.get("access") or {}).get("formats") if isinstance(record.get("access"), dict) else [])
if record.get("has_fulltext_pdf") and "pdf" not in formats:
formats.append("pdf")
if record.get("has_fulltext_xml") and "xml" not in formats:
formats.append("xml")
evidence = str(record.get("evidence_level") or "metadata_only")
if evidence not in EVIDENCE_LEVELS:
evidence = "metadata_only"
source_id = str(_first(record, "id", "source_id", "file_name") or "")
container = dict(record.get("container") or {})
container.setdefault("title", str(_first(record, "publication_name", "container_title") or ""))
container.setdefault("type", "")
container.setdefault("volume", str(record.get("volume") or ""))
container.setdefault("issue", str(record.get("issue") or ""))
container.setdefault("pages", str(_first(record, "pages", "page") or ""))
raw_publisher = record.get("publisher")
if isinstance(raw_publisher, str):
publisher = {"name": raw_publisher, "place": ""}
else:
publisher = dict(raw_publisher or {})
publisher.setdefault("name", str(record.get("publisher_name") or ""))
publisher.setdefault("place", str(record.get("publisher_place") or ""))
raw_edition = record.get("edition")
if isinstance(raw_edition, str):
edition = {"label": raw_edition, "number": None}
else:
edition = dict(raw_edition or {})
edition.setdefault("label", str(record.get("edition_label") or ""))
edition.setdefault("number", record.get("edition_number"))
raw_relations = record.get("relations")
relations = dict(raw_relations) if isinstance(raw_relations, dict) else {}
relations.setdefault("is_part_of", str(record.get("is_part_of") or ""))
relations["has_parts"] = _string_list(relations.get("has_parts") or record.get("has_parts") or [])
relations.setdefault("edition_of", str(record.get("edition_of") or ""))
relations.setdefault("translation_of", str(record.get("translation_of") or ""))
relations.setdefault("patent_family", str(record.get("patent_family") or ""))
return {
"type": publication_type,
"raw_type": raw_type,
"title": title,
"creators": creators,
"issued": _issued(record),
"container": container,
"publisher": publisher,
"edition": edition,
"identifiers": {
"doi": doi,
"isbn": isbn,
"issn": issn,
"standard_number": str(_first(identifiers, "standard_number") or record.get("standard_number") or "").strip(),
"patent_application_number": str(_first(identifiers, "patent_application_number") or record.get("patent_application_number") or "").strip(),
"patent_publication_number": str(_first(identifiers, "patent_publication_number") or record.get("patent_publication_number") or "").strip(),
"report_number": str(_first(identifiers, "report_number") or record.get("report_number") or "").strip(),
},
"relations": relations,
"language": str(record.get("language") or "").strip(),
"abstract": abstract,
"access": {
"has_abstract": bool(abstract or record.get("has_abstract")),
"has_snippet": bool(snippet),
"has_fulltext": bool(formats or record.get("has_fulltext")),
"formats": formats,
"local_paths": _string_list((record.get("access") or {}).get("local_paths") if isinstance(record.get("access"), dict) else record.get("local_paths") or []),
},
"evidence_level": evidence,
"sources": [{
"backend": backend,
"source_id": source_id,
"retrieved_at": str(record.get("retrieved_at") or ""),
}],
"conflicts": {},
}
def _edition_key(record: dict[str, Any]) -> str:
edition = record.get("edition") or {}
return normalize_title(edition.get("number") or edition.get("label"))
def exact_key(record: dict[str, Any]) -> str:
ids = record["identifiers"]
if ids.get("doi"):
return "doi:" + ids["doi"]
year = record["issued"].get("year") or ""
publisher = normalize_title(record["publisher"].get("name"))
if record["type"] == "standard" and ids.get("standard_number"):
return f"standard:{publisher}:{normalize_title(ids['standard_number'])}:{year}"
if record["type"] == "patent" and ids.get("patent_publication_number"):
return "patent:" + normalize_title(ids["patent_publication_number"])
if record["type"] == "report" and ids.get("report_number"):
return f"report:{publisher}:{normalize_title(ids['report_number'])}"
if record["type"] == "book" and ids.get("isbn"):
return f"book:{ids['isbn'][0]}:{_edition_key(record)}"
title = normalize_title(record.get("title"))
creators = record.get("creators") or []
creator = normalize_creator(creators[0]) if creators else ""
if title and creator and year:
extra = _edition_key(record) if record["type"] == "book" else ""
if record["type"] == "book_chapter":
extra = normalize_title(record["container"].get("title"))
return f"fallback:{record['type']}:{title}:{creator}:{year}:{extra}"
return ""
def _add_conflict(target: dict[str, Any], field: str, value: Any) -> None:
if value in (None, "", [], {}):
return
values = target["conflicts"].setdefault(field, [])
if value not in values:
values.append(value)
def merge_record(target: dict[str, Any], incoming: dict[str, Any]) -> None:
for source in incoming["sources"]:
if source not in target["sources"]:
target["sources"].append(source)
if EVIDENCE_LEVELS.index(incoming["evidence_level"]) > EVIDENCE_LEVELS.index(target["evidence_level"]):
target["evidence_level"] = incoming["evidence_level"]
if not target["abstract"] and incoming["abstract"]:
target["abstract"] = incoming["abstract"]
elif target["abstract"] and incoming["abstract"] and target["abstract"] != incoming["abstract"]:
_add_conflict(target, "abstract", incoming["abstract"])
for flag in ("has_abstract", "has_snippet", "has_fulltext"):
target["access"][flag] = target["access"][flag] or incoming["access"][flag]
for field in ("formats", "local_paths"):
for value in incoming["access"][field]:
if value not in target["access"][field]:
target["access"][field].append(value)
for field in ("doi", "standard_number", "patent_application_number", "patent_publication_number", "report_number"):
current, value = target["identifiers"].get(field), incoming["identifiers"].get(field)
if not current and value:
target["identifiers"][field] = value
elif current and value and current != value:
_add_conflict(target, f"identifiers.{field}", value)
for field in ("isbn", "issn"):
for value in incoming["identifiers"].get(field, []):
if value not in target["identifiers"][field]:
target["identifiers"][field].append(value)
for field in ("title", "raw_type", "language"):
current, value = target.get(field), incoming.get(field)
if not current and value:
target[field] = value
elif current and value and normalize_title(current) != normalize_title(value):
_add_conflict(target, field, value)
for field in ("container", "publisher", "edition"):
for key, value in incoming[field].items():
current = target[field].get(key)
if not current and value not in (None, ""):
target[field][key] = value
elif current and value not in (None, "") and current != value:
_add_conflict(target, f"{field}.{key}", value)
for key, value in incoming["relations"].items():
current = target["relations"].get(key)
if isinstance(value, list):
if not isinstance(current, list):
current = []
target["relations"][key] = current
for item in value:
if item not in current:
current.append(item)
elif not current and value:
target["relations"][key] = value
elif current and value and current != value:
_add_conflict(target, f"relations.{key}", value)
def merge_publications(records: Iterable[tuple[str, dict[str, Any]]]) -> dict[str, Any]:
publications: list[dict[str, Any]] = []
key_to_index: dict[str, int] = {}
for backend, raw in records:
normalized = normalize_record(raw, backend)
key = exact_key(normalized)
if key and key in key_to_index:
merge_record(publications[key_to_index[key]], normalized)
else:
if key:
key_to_index[key] = len(publications)
publications.append(normalized)
possible: list[dict[str, Any]] = []
for left_index, left in enumerate(publications):
left_title = normalize_title(left["title"])
if not left_title:
continue
for right_index in range(left_index + 1, len(publications)):
right = publications[right_index]
if left["type"] != right["type"]:
continue
left_year, right_year = left["issued"]["year"], right["issued"]["year"]
if left_year and right_year and abs(left_year - right_year) > 1:
continue
ratio = SequenceMatcher(None, left_title, normalize_title(right["title"])).ratio()
if ratio >= 0.93:
possible.append({
"left_index": left_index,
"right_index": right_index,
"left_title": left["title"],
"right_title": right["title"],
"title_similarity": round(ratio, 4),
})
return {"publications": publications, "possible_duplicates": possible}
def _load_records(path: Path) -> list[dict[str, Any]]:
data = json.loads(path.read_text(encoding="utf-8"))
if isinstance(data, list):
return data
if not isinstance(data, dict):
raise ValueError(f"{path}: JSON 顶层必须是列表或对象")
for key in ("records", "publications", "results"):
if isinstance(data.get(key), list):
return data[key]
nested = data.get("data")
if isinstance(nested, list):
return nested
raise ValueError(f"{path}: 未找到 records/publications/results 列表")
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--input", action="append", required=True, metavar="BACKEND=PATH")
parser.add_argument("--output", required=True, type=Path)
args = parser.parse_args(argv)
records: list[tuple[str, dict[str, Any]]] = []
for spec in args.input:
if "=" not in spec:
parser.error("--input 必须使用 BACKEND=PATH")
backend, raw_path = spec.split("=", 1)
backend = backend.strip()
if not backend:
parser.error("--input 的 BACKEND 不可为空")
records.extend((backend, item) for item in _load_records(Path(raw_path)))
result = merge_publications(records)
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(f"[OK] merged {len(records)} source records into {len(result['publications'])} publications")
if result["possible_duplicates"]:
print(f"[WARN] possible duplicates: {len(result['possible_duplicates'])}")
return 0
if __name__ == "__main__":
raise SystemExit(main())