257 lines
8.8 KiB
Python
257 lines
8.8 KiB
Python
"""论文标题质量检测与全文标题提取工具。
|
||
|
||
本模块不依赖 Django,管理命令、抓取任务和单元测试都可以复用。全文来源的
|
||
可信度按 XML(标题节点且 DOI 匹配) > PDF 元数据(首页身份匹配) > PDF 首页版式
|
||
推断排序;调用方默认只应自动采用 high 置信度候选。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
from dataclasses import dataclass
|
||
from html import unescape
|
||
import logging
|
||
import os
|
||
import re
|
||
from typing import Optional
|
||
|
||
|
||
_TAG_RE = re.compile(r"<[^>]+>")
|
||
_SPACE_RE = re.compile(r"\s+")
|
||
_WORD_RE = re.compile(r"[\w'-]+", re.UNICODE)
|
||
_DOI_RE = re.compile(r"10\.\d{4,9}/[-._;()/:A-Z0-9]+", re.I)
|
||
_WEB_MARKERS = (
|
||
"journal help", "user username", "remember me login", "current issue",
|
||
"paper submission", "guidelines for authors", "view subscribe",
|
||
)
|
||
_DOCUMENT_MARKERS = ("author:", "authors:", "abstract:", "document type:")
|
||
_GENERIC_PDF_TITLES = {
|
||
"untitled", "microsoft word", "document", "article", "full text", "pdf",
|
||
}
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class TitleAssessment:
|
||
title: str
|
||
plain_length: int
|
||
word_count: int
|
||
sentence_count: int
|
||
suspect: bool
|
||
reasons: tuple[str, ...]
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class FulltextTitleCandidate:
|
||
title: str
|
||
source: str
|
||
confidence: str
|
||
evidence: str
|
||
|
||
|
||
def clean_title(value: Optional[str]) -> str:
|
||
"""去标签、解实体并合并空白,保留公式中的可见文本。"""
|
||
if not value:
|
||
return ""
|
||
text = unescape(_TAG_RE.sub(" ", str(value)))
|
||
text = _SPACE_RE.sub(" ", text).strip(" \t\r\n\ufeff")
|
||
if text.lower().startswith("title:"):
|
||
text = text[6:].strip()
|
||
return text
|
||
|
||
|
||
def normalize_doi(value: Optional[str]) -> str:
|
||
if not value:
|
||
return ""
|
||
value = unescape(str(value)).strip().lower()
|
||
value = re.sub(r"^(?:doi:\s*|https?://(?:dx\.)?doi\.org/)", "", value)
|
||
match = _DOI_RE.search(value)
|
||
return match.group(0).rstrip(".,;)").lower() if match else value.rstrip(".,;)")
|
||
|
||
|
||
def assess_title(value: Optional[str]) -> TitleAssessment:
|
||
plain = clean_title(value)
|
||
lower = plain.lower()
|
||
words = len(_WORD_RE.findall(plain))
|
||
sentences = len(re.findall(r"[.!?](?:\s|$)", plain))
|
||
reasons = []
|
||
|
||
# 去标签后的长度才用于判断,避免 MathML 本身把正常标题撑到几千字符。
|
||
if len(plain) >= 300 and words >= 50 and sentences >= 3:
|
||
reasons.append("abstract_like_prose")
|
||
web_hits = sum(marker in lower for marker in _WEB_MARKERS)
|
||
if len(plain) >= 200 and web_hits >= 2:
|
||
reasons.append("webpage_text")
|
||
doc_hits = sum(marker in lower for marker in _DOCUMENT_MARKERS)
|
||
if len(plain) >= 300 and doc_hits >= 2:
|
||
reasons.append("document_metadata")
|
||
if len(plain) >= 800 and words >= 80:
|
||
reasons.append("extreme_length")
|
||
|
||
return TitleAssessment(
|
||
title=plain,
|
||
plain_length=len(plain),
|
||
word_count=words,
|
||
sentence_count=sentences,
|
||
suspect=bool(reasons),
|
||
reasons=tuple(reasons),
|
||
)
|
||
|
||
|
||
def is_usable_title(value: Optional[str]) -> bool:
|
||
assessment = assess_title(value)
|
||
if not assessment.title or assessment.suspect:
|
||
return False
|
||
if assessment.plain_length < 5 or assessment.plain_length > 500:
|
||
return False
|
||
if assessment.word_count > 70 or assessment.sentence_count > 2:
|
||
return False
|
||
return True
|
||
|
||
|
||
def is_better_title(current: Optional[str], candidate: Optional[str]) -> bool:
|
||
old = assess_title(current)
|
||
new = assess_title(candidate)
|
||
if not old.suspect or not is_usable_title(new.title):
|
||
return False
|
||
normalized_old = re.sub(r"\W+", "", old.title).casefold()
|
||
normalized_new = re.sub(r"\W+", "", new.title).casefold()
|
||
if not normalized_new or normalized_new == normalized_old:
|
||
return False
|
||
return new.plain_length < old.plain_length * 0.75
|
||
|
||
|
||
def _element_text(element) -> str:
|
||
return clean_title(" ".join(element.itertext()))
|
||
|
||
|
||
def extract_title_from_xml(
|
||
path: str,
|
||
expected_doi: Optional[str] = None,
|
||
) -> Optional[FulltextTitleCandidate]:
|
||
"""从 Elsevier/JATS/常见全文 XML 中提取文章标题。
|
||
|
||
XML 内存在 DOI 时必须与目标论文一致;不一致直接拒绝,防止历史文件错配。
|
||
"""
|
||
if not os.path.exists(path):
|
||
return None
|
||
try:
|
||
from lxml import etree
|
||
|
||
parser = etree.XMLParser(recover=True, huge_tree=True, resolve_entities=False)
|
||
root = etree.parse(path, parser).getroot()
|
||
except Exception:
|
||
return None
|
||
|
||
expected = normalize_doi(expected_doi)
|
||
doi_values = []
|
||
for node in root.xpath(
|
||
"//*[local-name()='doi' or local-name()='identifier' "
|
||
"or (local-name()='article-id' and "
|
||
"translate(@pub-id-type, 'DOI', 'doi')='doi')]"
|
||
):
|
||
value = normalize_doi(_element_text(node))
|
||
if value.startswith("10."):
|
||
doi_values.append(value)
|
||
if expected and doi_values and expected not in doi_values:
|
||
return None
|
||
|
||
title_xpaths = (
|
||
"//*[local-name()='coredata']/*[local-name()='title']",
|
||
"//*[local-name()='article-meta']/*[local-name()='title-group']/*[local-name()='article-title']",
|
||
"//*[local-name()='article-title']",
|
||
"/*/*[local-name()='coredata']/*[local-name()='title']",
|
||
)
|
||
for xpath in title_xpaths:
|
||
for node in root.xpath(xpath):
|
||
title = _element_text(node)
|
||
if not is_usable_title(title):
|
||
continue
|
||
doi_matched = bool(expected and expected in doi_values)
|
||
return FulltextTitleCandidate(
|
||
title=title,
|
||
source="fulltext_xml",
|
||
confidence="high" if doi_matched else "medium",
|
||
evidence="xml_title_node+doi" if doi_matched else "xml_title_node",
|
||
)
|
||
return None
|
||
|
||
|
||
def _identity_evidence(text: str, expected_doi: str, first_author: str) -> str:
|
||
if expected_doi:
|
||
found = {normalize_doi(v) for v in _DOI_RE.findall(text)}
|
||
if normalize_doi(expected_doi) in found:
|
||
return "doi"
|
||
if first_author:
|
||
surname = clean_title(first_author).split()[-1].casefold()
|
||
if len(surname) >= 3 and surname in text.casefold():
|
||
return "author"
|
||
return ""
|
||
|
||
|
||
def _title_from_first_page(text: str, first_author: str) -> str:
|
||
"""保守地从首页文本猜标题,仅返回低/中置信候选。"""
|
||
lines = [clean_title(line) for line in text.splitlines()]
|
||
lines = [line for line in lines if line]
|
||
if not lines:
|
||
return ""
|
||
|
||
surname = clean_title(first_author).split()[-1].casefold() if first_author else ""
|
||
author_index = next(
|
||
(i for i, line in enumerate(lines[:30]) if len(surname) >= 3 and surname in line.casefold()),
|
||
None,
|
||
)
|
||
if author_index is None or author_index == 0:
|
||
return ""
|
||
|
||
candidates = []
|
||
for line in lines[max(0, author_index - 4):author_index]:
|
||
lower = line.casefold()
|
||
if _DOI_RE.search(line) or "journal" in lower or "issn" in lower:
|
||
continue
|
||
if re.fullmatch(r"[\d\s|:/.-]+", line):
|
||
continue
|
||
candidates.append(line)
|
||
title = clean_title(" ".join(candidates))
|
||
return title if is_usable_title(title) else ""
|
||
|
||
|
||
def extract_title_from_pdf(
|
||
path: str,
|
||
expected_doi: Optional[str] = None,
|
||
first_author: Optional[str] = None,
|
||
) -> Optional[FulltextTitleCandidate]:
|
||
"""从 PDF metadata 或首页文本提取标题候选。
|
||
|
||
PDF metadata 只有在首页 DOI/作者能确认文件身份时才达到 medium;首页版式推断
|
||
即使 DOI 匹配也保持 medium,默认不会被管理命令自动写库。
|
||
"""
|
||
if not os.path.exists(path):
|
||
return None
|
||
try:
|
||
logging.getLogger("pypdf").setLevel(logging.CRITICAL)
|
||
from pypdf import PdfReader
|
||
|
||
reader = PdfReader(path, strict=False)
|
||
page_text = reader.pages[0].extract_text() or "" if reader.pages else ""
|
||
metadata = reader.metadata
|
||
except Exception:
|
||
return None
|
||
|
||
identity = _identity_evidence(page_text, expected_doi or "", first_author or "")
|
||
meta_title = clean_title(getattr(metadata, "title", "") if metadata else "")
|
||
if meta_title and meta_title.casefold() not in _GENERIC_PDF_TITLES and is_usable_title(meta_title):
|
||
return FulltextTitleCandidate(
|
||
title=meta_title,
|
||
source="pdf_metadata",
|
||
confidence="medium" if identity else "low",
|
||
evidence=f"pdf_metadata+{identity}" if identity else "pdf_metadata_only",
|
||
)
|
||
|
||
page_title = _title_from_first_page(page_text, first_author or "")
|
||
if page_title:
|
||
return FulltextTitleCandidate(
|
||
title=page_title,
|
||
source="pdf_first_page",
|
||
confidence="medium" if identity else "low",
|
||
evidence=f"pdf_first_page+{identity}" if identity else "pdf_first_page_only",
|
||
)
|
||
return None
|