diff --git a/CLAUDE.md b/CLAUDE.md index 646aab2..e6a3451 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -174,6 +174,13 @@ The paper fetch pipeline in `apps/resm/tasks.py` currently includes: - PDF fetch from Elsevier - Sci-Hub fallback - task fan-out and stuck-download release +- suspicious-title detection, targeted OpenAlex title refresh, and high-confidence + correction from fulltext XML when the XML DOI matches the paper + +Title quality is tracked on `Paper` with the original value retained only when a correction +is applied. `python manage.py audit_paper_titles` performs a dry-run audit by default; +`--apply` writes only high-confidence XML corrections unless `--allow-medium` is explicitly +provided for reviewed PDF candidates. Download behavior is stateful: diff --git a/apps/resm/management/commands/audit_paper_titles.py b/apps/resm/management/commands/audit_paper_titles.py new file mode 100644 index 0000000..0d1fe49 --- /dev/null +++ b/apps/resm/management/commands/audit_paper_titles.py @@ -0,0 +1,176 @@ +"""审计异常标题,并可用全文中的标题安全矫正。""" +import csv +import os + +from django.conf import settings +from django.core.management.base import BaseCommand, CommandError +from django.db.models.functions import Length +from django.utils import timezone + +from apps.resm.models import Paper +from apps.resm.title_utils import ( + FulltextTitleCandidate, + assess_title, + clean_title, + extract_title_from_pdf, + extract_title_from_xml, + is_better_title, + is_usable_title, +) + + +def _paper_path(paper, ext): + safe_doi = paper.doi.replace("/", "_") + if paper.publication_date is None: + directory = os.path.join(settings.BASE_DIR, "media", "papers", "unknown") + else: + d = paper.publication_date + directory = os.path.join( + settings.BASE_DIR, "media", "papers", str(d.year), str(d.month), str(d.day) + ) + return os.path.join(directory, f"{safe_doi}.{ext}") + + +def _fulltext_candidate(paper): + candidates = [] + if paper.has_fulltext_xml: + candidate = extract_title_from_xml(_paper_path(paper, "xml"), paper.doi) + if candidate: + candidates.append(candidate) + if paper.has_fulltext_pdf: + candidate = extract_title_from_pdf( + _paper_path(paper, "pdf"), paper.doi, paper.first_author or "" + ) + if candidate: + candidates.append(candidate) + rank = {"high": 3, "medium": 2, "low": 1} + return max(candidates, key=lambda item: rank[item.confidence], default=None) + + +class Command(BaseCommand): + help = "审计摘要化/网页化标题,并可用 fulltext XML/PDF 标题矫正" + + def add_arguments(self, parser): + parser.add_argument("--doi", action="append", default=[], help="只处理指定 DOI,可重复") + parser.add_argument("--min-length", type=int, default=300, help="初筛标题最小长度") + parser.add_argument("--limit", type=int, default=0, help="最多检查多少条,0=不限") + parser.add_argument("--apply", action="store_true", help="实际写库;默认仅审计") + parser.add_argument( + "--allow-medium", action="store_true", + help="允许采用 medium 置信候选;默认仅 XML+DOI 的 high 候选自动矫正", + ) + parser.add_argument("--csv", default="", help="将审计结果写入 CSV") + parser.add_argument("--title", default="", help="单个 --doi 的人工确认标题") + parser.add_argument("--source", default="manual", help="人工标题来源标记") + + def handle(self, *args, **opts): + dois = [doi.strip().lower() for doi in opts["doi"] if doi.strip()] + manual_title = clean_title(opts["title"]) + if manual_title and len(dois) != 1: + raise CommandError("--title 必须与且只能与一个 --doi 一起使用") + if manual_title and not is_usable_title(manual_title): + raise CommandError("--title 不是可接受的标题") + + qs = Paper.objects.all().order_by("id") + if dois: + qs = qs.filter(doi__in=dois) + else: + qs = qs.annotate(title_len=Length("title")).filter( + title_len__gte=max(1, opts["min_length"]) + ) + if opts["limit"]: + qs = qs[:opts["limit"]] + + csv_file = None + writer = None + if opts["csv"]: + csv_file = open(opts["csv"], "w", newline="", encoding="utf-8-sig") + writer = csv.DictWriter(csv_file, fieldnames=( + "doi", "openalex_id", "old_title", "reasons", "candidate_title", + "candidate_source", "confidence", "evidence", "action", + )) + writer.writeheader() + + checked = suspect_count = corrected = candidates = 0 + try: + for paper in qs.iterator(chunk_size=500): + checked += 1 + assessment = assess_title(paper.title) + if not assessment.suspect and not manual_title: + continue + suspect_count += int(assessment.suspect) + + if manual_title: + candidate = FulltextTitleCandidate( + title=manual_title, + source=opts["source"][:30], + confidence="high", + evidence="manual_confirmed", + ) + else: + candidate = _fulltext_candidate(paper) + if candidate: + candidates += 1 + + allowed = candidate and ( + candidate.confidence == "high" + or (opts["allow_medium"] and candidate.confidence == "medium") + ) + should_correct = bool( + candidate and allowed + and (manual_title or is_better_title(paper.title, candidate.title)) + ) + action = "candidate" + if not candidate: + action = "no_candidate" + elif not allowed: + action = "needs_review" + elif not should_correct: + action = "rejected" + + if opts["apply"]: + if should_correct: + if not paper.title_raw: + paper.title_raw = paper.title + paper.title = candidate.title + paper.title_source = candidate.source + paper.title_quality_status = Paper.TITLE_CORRECTED + paper.title_verified_at = timezone.now() + paper.save(update_fields=[ + "title", "title_raw", "title_source", "title_quality_status", + "title_verified_at", "update_time", + ]) + corrected += 1 + action = "corrected" + elif assessment.suspect and paper.title_quality_status == Paper.TITLE_UNCHECKED: + paper.title_quality_status = Paper.TITLE_SUSPECT + paper.save(update_fields=["title_quality_status", "update_time"]) + action = "marked_suspect" + + row = { + "doi": paper.doi, + "openalex_id": paper.openalex_id or "", + "old_title": paper.title_raw if action == "corrected" else paper.title, + "reasons": ",".join(assessment.reasons), + "candidate_title": candidate.title if candidate else "", + "candidate_source": candidate.source if candidate else "", + "confidence": candidate.confidence if candidate else "", + "evidence": candidate.evidence if candidate else "", + "action": action, + } + if writer: + writer.writerow(row) + self.stdout.write( + f"[{action}] {paper.doi} len={assessment.plain_length} " + f"reasons={row['reasons']} candidate={row['candidate_source'] or '-'} " + f"confidence={row['confidence'] or '-'}" + ) + finally: + if csv_file: + csv_file.close() + + mode = "apply" if opts["apply"] else "dry-run" + self.stdout.write(self.style.SUCCESS( + f"完成({mode}) checked={checked} suspect={suspect_count} " + f"fulltext_candidates={candidates} corrected={corrected}" + )) diff --git a/apps/resm/migrations/0012_paper_title_quality.py b/apps/resm/migrations/0012_paper_title_quality.py new file mode 100644 index 0000000..42fdac5 --- /dev/null +++ b/apps/resm/migrations/0012_paper_title_quality.py @@ -0,0 +1,75 @@ +from django.db import migrations, models + + +class Migration(migrations.Migration): + + # resm_paper 现有数百万行;质量状态索引用 CONCURRENTLY 创建,避免长时间阻塞写入。 + atomic = False + + dependencies = [ + ("resm", "0011_seed_briefing_monitors"), + ] + + operations = [ + migrations.AddField( + model_name="paper", + name="title_raw", + field=models.TextField(blank=True, null=True), + ), + migrations.AddField( + model_name="paper", + name="title_source", + field=models.CharField(blank=True, max_length=30, null=True), + ), + migrations.SeparateDatabaseAndState( + database_operations=[ + migrations.AddField( + model_name="paper", + name="title_quality_status", + field=models.CharField( + choices=[ + ("unchecked", "未检查"), + ("suspect", "疑似异常"), + ("verified", "已验证"), + ("corrected", "已矫正"), + ], + default="unchecked", + max_length=20, + ), + ), + ], + state_operations=[ + migrations.AddField( + model_name="paper", + name="title_quality_status", + field=models.CharField( + choices=[ + ("unchecked", "未检查"), + ("suspect", "疑似异常"), + ("verified", "已验证"), + ("corrected", "已矫正"), + ], + db_index=True, + default="unchecked", + max_length=20, + ), + ), + ], + ), + migrations.RunSQL( + sql=( + "CREATE INDEX CONCURRENTLY IF NOT EXISTS " + "resm_paper_title_quality_status_idx " + "ON resm_paper (title_quality_status);" + ), + reverse_sql=( + "DROP INDEX CONCURRENTLY IF EXISTS " + "resm_paper_title_quality_status_idx;" + ), + ), + migrations.AddField( + model_name="paper", + name="title_verified_at", + field=models.DateTimeField(blank=True, null=True), + ), + ] diff --git a/apps/resm/models.py b/apps/resm/models.py index 1853ad2..61eaa51 100644 --- a/apps/resm/models.py +++ b/apps/resm/models.py @@ -5,12 +5,30 @@ import os # Create your models here. class Paper(BaseModel): + TITLE_UNCHECKED = "unchecked" + TITLE_SUSPECT = "suspect" + TITLE_VERIFIED = "verified" + TITLE_CORRECTED = "corrected" + TITLE_QUALITY_CHOICES = ( + (TITLE_UNCHECKED, "未检查"), + (TITLE_SUSPECT, "疑似异常"), + (TITLE_VERIFIED, "已验证"), + (TITLE_CORRECTED, "已矫正"), + ) # ===== 全局唯一标识 ===== openalex_id = models.TextField(unique=True, verbose_name="OpenAlex ID", null=True, blank=True) doi = models.TextField(unique=True, verbose_name="DOI") # ===== 基本信息 ===== type = models.CharField(max_length=20, db_index=True) title = models.TextField() + # title_raw 仅在矫正时保存旧值,避免为全部历史数据复制一份大字段。 + title_raw = models.TextField(null=True, blank=True) + title_source = models.CharField(max_length=30, null=True, blank=True) + title_quality_status = models.CharField( + max_length=20, choices=TITLE_QUALITY_CHOICES, + default=TITLE_UNCHECKED, db_index=True, + ) + title_verified_at = models.DateTimeField(null=True, blank=True) publication_date = models.DateField(null=True, blank=True) publication_year = models.IntegerField(db_index=True) # ===== 作者(最小可用集)===== diff --git a/apps/resm/tasks.py b/apps/resm/tasks.py index 77c7582..e24f5ba 100644 --- a/apps/resm/tasks.py +++ b/apps/resm/tasks.py @@ -12,6 +12,7 @@ from celery import current_app from datetime import datetime, timedelta import random from .pdf_utils import _is_elsevier_preview_pdf +from .title_utils import assess_title, extract_title_from_xml, is_better_title from uuid import uuid4 import asyncio import sys @@ -93,7 +94,10 @@ def _build_paper_from_record(record, keywords: str, search: str) -> Paper: paper.type = (record.get("type") or "article")[:20] paper.openalex_id = record["id"].split("/")[-1] paper.doi = record["doi"].replace("https://doi.org/", "") - paper.title = record["display_name"] + paper.title = record.get("title") or record["display_name"] + paper.title_source = "openalex" + if assess_title(paper.title).suspect: + paper.title_quality_status = Paper.TITLE_SUSPECT paper.publication_date = record["publication_date"] paper.publication_year = record["publication_year"] if record["open_access"]: @@ -226,6 +230,42 @@ def update_paper_meta_from_openalex(days: int = 30, per_combo_max: int = None): return f"openalex update: combos={n_combos}, new_papers={new_papers}, since={from_publication_date}" +@shared_task(base=CustomTask) +def refresh_suspect_titles_from_openalex(limit: int = 100): + """定向刷新疑似异常标题,吸收 OpenAlex/Crossref 后续元数据修正。 + + 常规抓取使用 bulk_create(ignore_conflicts=True),不会更新已有 DOI;本任务只检查 + 已标记 suspect 的记录,并且仅在新标题通过质量检查且明显优于旧值时覆盖。 + """ + qs = Paper.objects.filter( + title_quality_status=Paper.TITLE_SUSPECT, + openalex_id__isnull=False, + ).exclude(openalex_id="").order_by("id")[:max(1, limit)] + checked = corrected = failed = 0 + for paper in qs: + checked += 1 + try: + record = Works()[paper.openalex_id] + except Exception: + failed += 1 + continue + candidate = (record or {}).get("title") or (record or {}).get("display_name") + if not is_better_title(paper.title, candidate): + continue + if not paper.title_raw: + paper.title_raw = paper.title + paper.title = assess_title(candidate).title + paper.title_source = "openalex_refresh" + paper.title_quality_status = Paper.TITLE_CORRECTED + paper.title_verified_at = timezone.now() + paper.save(update_fields=[ + "title", "title_raw", "title_source", "title_quality_status", + "title_verified_at", "update_time", + ]) + corrected += 1 + return f"title refresh: checked={checked}, corrected={corrected}, failed={failed}" + + BACKFILL_STOP_KEY = "backfill_paper_meta_stop" @@ -374,6 +414,9 @@ def _build_paper_from_sd_result(r, qs_text: str): paper.o_search = qs_text paper.doi = str(doi).replace("https://doi.org/", "") paper.title = title + paper.title_source = "elsevier" + if assess_title(paper.title).suspect: + paper.title_quality_status = Paper.TITLE_SUSPECT paper.publication_date = pub_date if len(str(pub_date)) == 10 else None paper.publication_year = year paper.publication_name = r.get("sourceTitle") @@ -625,8 +668,24 @@ def _elsevier_fetch_xml(req, paper): paper.has_fulltext_xml = True paper.save_file_xml(xml_str) - paper.save(update_fields=["has_abstract", "has_abstract_xml", - "has_fulltext", "has_fulltext_xml", "update_time"]) + update_fields = ["has_abstract", "has_abstract_xml", + "has_fulltext", "has_fulltext_xml", "update_time"] + # XML 标题节点且 DOI 匹配才是 high;仅在旧标题已呈摘要化/网页化时自动覆盖。 + if paper.title_quality_status == Paper.TITLE_SUSPECT or assess_title(paper.title).suspect: + candidate = extract_title_from_xml(paper.init_paper_path("xml"), paper.doi) + if (candidate and candidate.confidence == "high" + and is_better_title(paper.title, candidate.title)): + if not paper.title_raw: + paper.title_raw = paper.title + paper.title = candidate.title + paper.title_source = candidate.source + paper.title_quality_status = Paper.TITLE_CORRECTED + paper.title_verified_at = timezone.now() + update_fields.extend([ + "title", "title_raw", "title_source", "title_quality_status", + "title_verified_at", + ]) + paper.save(update_fields=update_fields) return True, has_fulltext, None @@ -942,4 +1001,4 @@ def save_pdf_from_scihub(paper:Paper): else: paper.save_fail_reason(err_msg) return err_msg -# https://sci.bban.top/pdf/10.1016/j.conbuildmat.2020.121016.pdf?download=true \ No newline at end of file +# https://sci.bban.top/pdf/10.1016/j.conbuildmat.2020.121016.pdf?download=true diff --git a/apps/resm/test_title_utils.py b/apps/resm/test_title_utils.py new file mode 100644 index 0000000..2066291 --- /dev/null +++ b/apps/resm/test_title_utils.py @@ -0,0 +1,102 @@ +import os +import tempfile +import types +import unittest +from unittest.mock import patch + +from apps.resm.title_utils import ( + assess_title, + extract_title_from_pdf, + extract_title_from_xml, + is_better_title, +) + + +class TitleAssessmentTests(unittest.TestCase): + def test_normal_title_is_not_suspect(self): + result = assess_title("Optimal Nonlinear PID Speed Control for an Electric Vehicle") + self.assertFalse(result.suspect) + + def test_abstract_stored_as_title_is_suspect(self): + text = ( + "In many situations, a beam opening is necessary near a plastic hinge. " + "However, few studies investigated the resulting behavior. " + "In this study, nine full-scale joints were tested with several configurations. " + "The results show that additional reinforcement improves strength and ductility. " + ) * 2 + result = assess_title(text) + self.assertTrue(result.suspect) + self.assertIn("abstract_like_prose", result.reasons) + + def test_mathml_markup_does_not_trigger_raw_length_false_positive(self): + markup = "" + "x" * 100 + "" + result = assess_title(f"Behavior of {markup} under pressure") + self.assertFalse(result.suspect) + + def test_short_candidate_is_better_than_abstract(self): + old = ("This study investigates reinforced concrete joints. " * 20).strip() + new = "Cyclic Behavior of Reinforced Concrete Beam-Column Joints" + self.assertTrue(is_better_title(old, new)) + + +class XmlTitleExtractionTests(unittest.TestCase): + def _xml_file(self, body): + handle = tempfile.NamedTemporaryFile("w", suffix=".xml", delete=False, encoding="utf-8") + self.addCleanup(lambda: os.path.exists(handle.name) and os.unlink(handle.name)) + handle.write(body) + handle.close() + return handle.name + + def test_extracts_elsevier_title_and_verifies_doi(self): + path = self._xml_file(""" + + + 10.1234/example.1 + Correct & Verified Article Title + + """) + candidate = extract_title_from_xml(path, "10.1234/example.1") + self.assertIsNotNone(candidate) + self.assertEqual(candidate.title, "Correct & Verified Article Title") + self.assertEqual(candidate.confidence, "high") + + def test_rejects_xml_for_another_doi(self): + path = self._xml_file("""
+ 10.1234/wrong + A Correct Looking Title +
""") + candidate = extract_title_from_xml(path, "10.1234/expected") + self.assertIsNone(candidate) + + +class PdfTitleExtractionTests(unittest.TestCase): + def test_pdf_metadata_requires_identity_evidence_and_stays_medium(self): + handle = tempfile.NamedTemporaryFile("wb", suffix=".pdf", delete=False) + self.addCleanup(lambda: os.path.exists(handle.name) and os.unlink(handle.name)) + handle.write(b"%PDF-placeholder") + handle.close() + + page = types.SimpleNamespace( + extract_text=lambda: ( + "Correct Article Title\nMohamed Shamseldin\n" + "https://doi.org/10.1234/example.1\nAbstract" + ) + ) + reader = types.SimpleNamespace( + pages=[page], + metadata=types.SimpleNamespace(title="Correct Article Title"), + ) + module = types.SimpleNamespace(PdfReader=lambda *args, **kwargs: reader) + with patch.dict("sys.modules", {"pypdf": module}): + candidate = extract_title_from_pdf( + handle.name, "10.1234/example.1", "Mohamed Shamseldin" + ) + self.assertIsNotNone(candidate) + self.assertEqual(candidate.title, "Correct Article Title") + self.assertEqual(candidate.confidence, "medium") + self.assertEqual(candidate.evidence, "pdf_metadata+doi") + + +if __name__ == "__main__": + unittest.main() diff --git a/apps/resm/title_utils.py b/apps/resm/title_utils.py new file mode 100644 index 0000000..d7200dd --- /dev/null +++ b/apps/resm/title_utils.py @@ -0,0 +1,256 @@ +"""论文标题质量检测与全文标题提取工具。 + +本模块不依赖 Django,管理命令、抓取任务和单元测试都可以复用。全文来源的 +可信度按 XML(标题节点且 DOI 匹配) > PDF 元数据(首页身份匹配) > PDF 首页版式 +推断排序;调用方默认只应自动采用 high 置信度候选。 +""" +from __future__ import annotations + +from dataclasses import dataclass +from html import unescape +import logging +import os +import re +from typing import Optional + + +_TAG_RE = re.compile(r"<[^>]+>") +_SPACE_RE = re.compile(r"\s+") +_WORD_RE = re.compile(r"[\w'-]+", re.UNICODE) +_DOI_RE = re.compile(r"10\.\d{4,9}/[-._;()/:A-Z0-9]+", re.I) +_WEB_MARKERS = ( + "journal help", "user username", "remember me login", "current issue", + "paper submission", "guidelines for authors", "view subscribe", +) +_DOCUMENT_MARKERS = ("author:", "authors:", "abstract:", "document type:") +_GENERIC_PDF_TITLES = { + "untitled", "microsoft word", "document", "article", "full text", "pdf", +} + + +@dataclass(frozen=True) +class TitleAssessment: + title: str + plain_length: int + word_count: int + sentence_count: int + suspect: bool + reasons: tuple[str, ...] + + +@dataclass(frozen=True) +class FulltextTitleCandidate: + title: str + source: str + confidence: str + evidence: str + + +def clean_title(value: Optional[str]) -> str: + """去标签、解实体并合并空白,保留公式中的可见文本。""" + if not value: + return "" + text = unescape(_TAG_RE.sub(" ", str(value))) + text = _SPACE_RE.sub(" ", text).strip(" \t\r\n\ufeff") + if text.lower().startswith("title:"): + text = text[6:].strip() + return text + + +def normalize_doi(value: Optional[str]) -> str: + if not value: + return "" + value = unescape(str(value)).strip().lower() + value = re.sub(r"^(?:doi:\s*|https?://(?:dx\.)?doi\.org/)", "", value) + match = _DOI_RE.search(value) + return match.group(0).rstrip(".,;)").lower() if match else value.rstrip(".,;)") + + +def assess_title(value: Optional[str]) -> TitleAssessment: + plain = clean_title(value) + lower = plain.lower() + words = len(_WORD_RE.findall(plain)) + sentences = len(re.findall(r"[.!?](?:\s|$)", plain)) + reasons = [] + + # 去标签后的长度才用于判断,避免 MathML 本身把正常标题撑到几千字符。 + if len(plain) >= 300 and words >= 50 and sentences >= 3: + reasons.append("abstract_like_prose") + web_hits = sum(marker in lower for marker in _WEB_MARKERS) + if len(plain) >= 200 and web_hits >= 2: + reasons.append("webpage_text") + doc_hits = sum(marker in lower for marker in _DOCUMENT_MARKERS) + if len(plain) >= 300 and doc_hits >= 2: + reasons.append("document_metadata") + if len(plain) >= 800 and words >= 80: + reasons.append("extreme_length") + + return TitleAssessment( + title=plain, + plain_length=len(plain), + word_count=words, + sentence_count=sentences, + suspect=bool(reasons), + reasons=tuple(reasons), + ) + + +def is_usable_title(value: Optional[str]) -> bool: + assessment = assess_title(value) + if not assessment.title or assessment.suspect: + return False + if assessment.plain_length < 5 or assessment.plain_length > 500: + return False + if assessment.word_count > 70 or assessment.sentence_count > 2: + return False + return True + + +def is_better_title(current: Optional[str], candidate: Optional[str]) -> bool: + old = assess_title(current) + new = assess_title(candidate) + if not old.suspect or not is_usable_title(new.title): + return False + normalized_old = re.sub(r"\W+", "", old.title).casefold() + normalized_new = re.sub(r"\W+", "", new.title).casefold() + if not normalized_new or normalized_new == normalized_old: + return False + return new.plain_length < old.plain_length * 0.75 + + +def _element_text(element) -> str: + return clean_title(" ".join(element.itertext())) + + +def extract_title_from_xml( + path: str, + expected_doi: Optional[str] = None, +) -> Optional[FulltextTitleCandidate]: + """从 Elsevier/JATS/常见全文 XML 中提取文章标题。 + + XML 内存在 DOI 时必须与目标论文一致;不一致直接拒绝,防止历史文件错配。 + """ + if not os.path.exists(path): + return None + try: + from lxml import etree + + parser = etree.XMLParser(recover=True, huge_tree=True, resolve_entities=False) + root = etree.parse(path, parser).getroot() + except Exception: + return None + + expected = normalize_doi(expected_doi) + doi_values = [] + for node in root.xpath( + "//*[local-name()='doi' or local-name()='identifier' " + "or (local-name()='article-id' and " + "translate(@pub-id-type, 'DOI', 'doi')='doi')]" + ): + value = normalize_doi(_element_text(node)) + if value.startswith("10."): + doi_values.append(value) + if expected and doi_values and expected not in doi_values: + return None + + title_xpaths = ( + "//*[local-name()='coredata']/*[local-name()='title']", + "//*[local-name()='article-meta']/*[local-name()='title-group']/*[local-name()='article-title']", + "//*[local-name()='article-title']", + "/*/*[local-name()='coredata']/*[local-name()='title']", + ) + for xpath in title_xpaths: + for node in root.xpath(xpath): + title = _element_text(node) + if not is_usable_title(title): + continue + doi_matched = bool(expected and expected in doi_values) + return FulltextTitleCandidate( + title=title, + source="fulltext_xml", + confidence="high" if doi_matched else "medium", + evidence="xml_title_node+doi" if doi_matched else "xml_title_node", + ) + return None + + +def _identity_evidence(text: str, expected_doi: str, first_author: str) -> str: + if expected_doi: + found = {normalize_doi(v) for v in _DOI_RE.findall(text)} + if normalize_doi(expected_doi) in found: + return "doi" + if first_author: + surname = clean_title(first_author).split()[-1].casefold() + if len(surname) >= 3 and surname in text.casefold(): + return "author" + return "" + + +def _title_from_first_page(text: str, first_author: str) -> str: + """保守地从首页文本猜标题,仅返回低/中置信候选。""" + lines = [clean_title(line) for line in text.splitlines()] + lines = [line for line in lines if line] + if not lines: + return "" + + surname = clean_title(first_author).split()[-1].casefold() if first_author else "" + author_index = next( + (i for i, line in enumerate(lines[:30]) if len(surname) >= 3 and surname in line.casefold()), + None, + ) + if author_index is None or author_index == 0: + return "" + + candidates = [] + for line in lines[max(0, author_index - 4):author_index]: + lower = line.casefold() + if _DOI_RE.search(line) or "journal" in lower or "issn" in lower: + continue + if re.fullmatch(r"[\d\s|:/.-]+", line): + continue + candidates.append(line) + title = clean_title(" ".join(candidates)) + return title if is_usable_title(title) else "" + + +def extract_title_from_pdf( + path: str, + expected_doi: Optional[str] = None, + first_author: Optional[str] = None, +) -> Optional[FulltextTitleCandidate]: + """从 PDF metadata 或首页文本提取标题候选。 + + PDF metadata 只有在首页 DOI/作者能确认文件身份时才达到 medium;首页版式推断 + 即使 DOI 匹配也保持 medium,默认不会被管理命令自动写库。 + """ + if not os.path.exists(path): + return None + try: + logging.getLogger("pypdf").setLevel(logging.CRITICAL) + from pypdf import PdfReader + + reader = PdfReader(path, strict=False) + page_text = reader.pages[0].extract_text() or "" if reader.pages else "" + metadata = reader.metadata + except Exception: + return None + + identity = _identity_evidence(page_text, expected_doi or "", first_author or "") + meta_title = clean_title(getattr(metadata, "title", "") if metadata else "") + if meta_title and meta_title.casefold() not in _GENERIC_PDF_TITLES and is_usable_title(meta_title): + return FulltextTitleCandidate( + title=meta_title, + source="pdf_metadata", + confidence="medium" if identity else "low", + evidence=f"pdf_metadata+{identity}" if identity else "pdf_metadata_only", + ) + + page_title = _title_from_first_page(page_text, first_author or "") + if page_title: + return FulltextTitleCandidate( + title=page_title, + source="pdf_first_page", + confidence="medium" if identity else "low", + evidence=f"pdf_first_page+{identity}" if identity else "pdf_first_page_only", + ) + return None