init
This commit is contained in:
@@ -0,0 +1,165 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Archive a finished literature-search-verify session into a permanent,
|
||||
project-level folder instead of leaving results sitting in the skill's
|
||||
own scratch output/ directory (which is easy to lose track of across
|
||||
sessions and isn't meant to be a durable deliverable location).
|
||||
|
||||
Bundles the verified BibTeX file -- and, if given, any downloaded PDFs --
|
||||
into <project-root>/references/<topic-slug>/, and writes a README.md
|
||||
index (entry list, suspect/unverified entries flagged separately, free-
|
||||
text coverage notes) so a future session or a human can find and trust
|
||||
what's there without re-reading the conversation that produced it.
|
||||
|
||||
No third-party dependencies; uses only the standard library.
|
||||
|
||||
CLI usage:
|
||||
python3 archive_references.py "UAV aeromagnetic compensation" \\
|
||||
--bib output/uav_aeromagnetic_compensation_final.bib \\
|
||||
--project-root . \\
|
||||
--pdfs-dir output/pdfs \\
|
||||
--suspect "Some fabricated-looking title|DOI resolves but venue is topically unrelated" \\
|
||||
--notes "Kalman-filter and GA/PSO angles searched, no on-topic hits found."
|
||||
|
||||
Output: prints the path of the archive directory that was created/updated.
|
||||
"""
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import sys
|
||||
from datetime import date
|
||||
|
||||
|
||||
def slugify(text):
|
||||
text = text.strip().lower()
|
||||
text = re.sub(r"[^a-z0-9]+", "_", text)
|
||||
return text.strip("_")[:60] or "references"
|
||||
|
||||
|
||||
def parse_bib_entries(bib_path):
|
||||
"""Minimal BibTeX parser -- just enough to pull key/title/year/venue/doi/note
|
||||
(plus the raw entry text, for reordering) for the README index. Not a
|
||||
general-purpose BibTeX parser."""
|
||||
with open(bib_path, encoding="utf-8") as f:
|
||||
content = f.read()
|
||||
|
||||
entries = []
|
||||
for m in re.finditer(r"@(\w+)\{([^,\n]+),(.*?)\n\}", content, re.S):
|
||||
entry_type, key, body = m.groups()
|
||||
fields = {}
|
||||
for fm in re.finditer(r"(\w+)\s*=\s*\{(.*?)\}\s*,?\s*(?=\n\s*\w+\s*=|\n\Z|\Z)", body, re.S):
|
||||
fields[fm.group(1).lower()] = re.sub(r"\s+", " ", fm.group(2)).strip()
|
||||
entries.append({"type": entry_type, "key": key.strip(), "raw": m.group(0).strip(), **fields})
|
||||
return entries
|
||||
|
||||
|
||||
def year_sort_key(entry):
|
||||
"""Chronological order, oldest first; entries with no parseable year sort last."""
|
||||
year_str = re.sub(r"[^0-9]", "", entry.get("year", "") or "")
|
||||
year = int(year_str) if year_str else 9999
|
||||
return (year, entry.get("key", ""))
|
||||
|
||||
|
||||
def build_readme(topic, entries, pdf_count, suspect, notes):
|
||||
lines = []
|
||||
lines.append(f"# {topic} — literature archive")
|
||||
lines.append("")
|
||||
lines.append(f"Archived: {date.today().isoformat()}")
|
||||
lines.append(f"Verified entries: {len(entries)}")
|
||||
lines.append(f"PDFs bundled: {pdf_count}")
|
||||
lines.append("")
|
||||
lines.append(
|
||||
"Every entry in `references.bib` passed independent verification "
|
||||
"(arXiv ID / DOI resolution and/or cross-source title match, "
|
||||
"similarity >= 0.9) via the literature-search-verify skill before "
|
||||
"being archived here. Citation keys follow the surname+year "
|
||||
"convention and are stable -- the paper-writing-grounded skill's "
|
||||
"`\\cite{}` calls should match these keys directly."
|
||||
)
|
||||
lines.append("")
|
||||
lines.append("## Entries (chronological, oldest first)")
|
||||
lines.append("")
|
||||
for e in entries:
|
||||
title = e.get("title", "?")
|
||||
year = e.get("year", "?")
|
||||
venue = e.get("journal") or e.get("booktitle") or e.get("school") or ""
|
||||
doi = e.get("doi", "")
|
||||
note = e.get("note", "")
|
||||
line = f"- **{e['key']}** ({year}) — {title}"
|
||||
if venue:
|
||||
line += f". *{venue}*"
|
||||
if doi:
|
||||
line += f". DOI: {doi}"
|
||||
lines.append(line)
|
||||
if note:
|
||||
lines.append(f" - Note: {note}")
|
||||
|
||||
if suspect:
|
||||
lines.append("")
|
||||
lines.append("## Flagged during search — NOT included above, do not cite")
|
||||
lines.append("")
|
||||
for s in suspect:
|
||||
parts = s.split("|", 1)
|
||||
title = parts[0].strip()
|
||||
reason = parts[1].strip() if len(parts) > 1 else ""
|
||||
lines.append(f"- {title}" + (f" — {reason}" if reason else ""))
|
||||
|
||||
if notes:
|
||||
lines.append("")
|
||||
lines.append("## Search coverage notes")
|
||||
lines.append("")
|
||||
lines.append(notes)
|
||||
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
ap.add_argument("topic", help="Human-readable topic name, e.g. \"UAV aeromagnetic compensation\"")
|
||||
ap.add_argument("--bib", required=True, help="path to the curated/verified .bib file to archive")
|
||||
ap.add_argument("--project-root", default=".", help="project root; archive is written under <root>/references/<slug>/")
|
||||
ap.add_argument("--pdfs-dir", default=None, help="optional folder of open-access PDFs to copy alongside the bib")
|
||||
ap.add_argument("--suspect", action="append", default=[], help="title|reason of a suspect/unverified entry to log; repeatable")
|
||||
ap.add_argument("--notes", default=None, help="free-text notes on search coverage/gaps for the README")
|
||||
args = ap.parse_args()
|
||||
|
||||
if not os.path.isfile(args.bib):
|
||||
print(f"error: bib file not found: {args.bib}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
slug = slugify(args.topic)
|
||||
archive_dir = os.path.join(args.project_root, "references", slug)
|
||||
os.makedirs(archive_dir, exist_ok=True)
|
||||
|
||||
bib_dest = os.path.join(archive_dir, "references.bib")
|
||||
shutil.copyfile(args.bib, bib_dest)
|
||||
entries = parse_bib_entries(bib_dest)
|
||||
entries.sort(key=year_sort_key)
|
||||
|
||||
# Rewrite the archived .bib in chronological order (oldest first) so the
|
||||
# file itself, not just the README, reads as a timeline.
|
||||
header = f"% {args.topic} -- verified references, chronological order\n% Archived {date.today().isoformat()}\n\n"
|
||||
with open(bib_dest, "w", encoding="utf-8") as f:
|
||||
f.write(header)
|
||||
f.write("\n\n".join(e["raw"] for e in entries))
|
||||
f.write("\n")
|
||||
|
||||
pdf_count = 0
|
||||
if args.pdfs_dir and os.path.isdir(args.pdfs_dir):
|
||||
pdf_dest_dir = os.path.join(archive_dir, "pdfs")
|
||||
os.makedirs(pdf_dest_dir, exist_ok=True)
|
||||
for fn in sorted(os.listdir(args.pdfs_dir)):
|
||||
if fn.lower().endswith(".pdf"):
|
||||
shutil.copyfile(os.path.join(args.pdfs_dir, fn), os.path.join(pdf_dest_dir, fn))
|
||||
pdf_count += 1
|
||||
|
||||
readme = build_readme(args.topic, entries, pdf_count, args.suspect, args.notes)
|
||||
with open(os.path.join(archive_dir, "README.md"), "w", encoding="utf-8") as f:
|
||||
f.write(readme)
|
||||
|
||||
print(archive_dir)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,160 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
End-to-end literature search: query arXiv + Semantic Scholar + Crossref,
|
||||
merge/dedupe candidates, independently verify each one, and emit both a
|
||||
human-readable report and BibTeX for the entries that passed verification.
|
||||
|
||||
This is the one script Claude should actually call for a normal literature
|
||||
search -- the individual search_*.py / verify_citation.py scripts exist
|
||||
mainly as building blocks it can reuse for one-off / follow-up lookups.
|
||||
|
||||
CLI usage:
|
||||
python3 literature_search.py "UAV magnetic compensation Tolles-Lawson" \\
|
||||
--max-per-source 8 --bib-out refs.bib
|
||||
|
||||
Output: prints a JSON report to stdout (one entry per merged candidate,
|
||||
with its verdict), and if --bib-out is given, writes BibTeX for every
|
||||
"verified" entry to that file (never for "suspect" or "unverified" ones).
|
||||
"""
|
||||
import sys
|
||||
import os
|
||||
import json
|
||||
import argparse
|
||||
import difflib
|
||||
import re
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
from search_arxiv import search_arxiv
|
||||
from search_semantic_scholar import search_s2
|
||||
from search_crossref import search_crossref
|
||||
from verify_citation import verify
|
||||
|
||||
|
||||
def _similar(a, b, threshold=0.88):
|
||||
if not a or not b:
|
||||
return False
|
||||
return difflib.SequenceMatcher(None, a.lower().strip(), b.lower().strip()).ratio() >= threshold
|
||||
|
||||
|
||||
def merge_candidates(all_results):
|
||||
"""Dedupe candidates that are the same paper found via multiple sources,
|
||||
merging their metadata (preferring whichever source has an identifier)."""
|
||||
merged = []
|
||||
for item in all_results:
|
||||
placed = False
|
||||
for m in merged:
|
||||
if _similar(item.get("title"), m.get("title")):
|
||||
# merge: fill in any missing fields, keep track of all sources
|
||||
for key in ("doi", "arxiv_id", "abstract", "venue", "year", "citation_count", "pdf_url"):
|
||||
if not m.get(key) and item.get(key):
|
||||
m[key] = item[key]
|
||||
m["sources"] = sorted(set(m.get("sources", [m.get("source")]) + [item.get("source")]))
|
||||
placed = True
|
||||
break
|
||||
if not placed:
|
||||
item = dict(item)
|
||||
item["sources"] = [item.get("source")]
|
||||
merged.append(item)
|
||||
return merged
|
||||
|
||||
|
||||
def make_bibtex_key(candidate, used_keys):
|
||||
authors = candidate.get("authors") or []
|
||||
surname = "unknown"
|
||||
if authors:
|
||||
first_author = authors[0]
|
||||
surname = first_author.strip().split()[-1].lower()
|
||||
surname = re.sub(r"[^a-z]", "", surname) or "unknown"
|
||||
year = str(candidate.get("year") or "nd")
|
||||
base = f"{surname}{year}"
|
||||
key = base
|
||||
suffix = ord("a")
|
||||
while key in used_keys:
|
||||
key = f"{base}{chr(suffix)}"
|
||||
suffix += 1
|
||||
used_keys.add(key)
|
||||
return key
|
||||
|
||||
|
||||
def to_bibtex(candidate, key):
|
||||
authors = candidate.get("authors") or []
|
||||
author_str = " and ".join(authors) if authors else "Unknown"
|
||||
title = candidate.get("title") or ""
|
||||
year = candidate.get("year") or ""
|
||||
venue = candidate.get("venue") or ""
|
||||
doi = candidate.get("doi") or ""
|
||||
arxiv_id = candidate.get("arxiv_id") or ""
|
||||
|
||||
if arxiv_id and not venue:
|
||||
entry_type = "misc"
|
||||
fields = [
|
||||
("author", author_str),
|
||||
("title", title),
|
||||
("year", str(year)),
|
||||
("eprint", arxiv_id),
|
||||
("archivePrefix", "arXiv"),
|
||||
]
|
||||
else:
|
||||
entry_type = "article"
|
||||
fields = [
|
||||
("author", author_str),
|
||||
("title", title),
|
||||
("journal", venue),
|
||||
("year", str(year)),
|
||||
]
|
||||
if doi:
|
||||
fields.append(("doi", doi))
|
||||
|
||||
lines = [f"@{entry_type}{{{key},"]
|
||||
for k, v in fields:
|
||||
if v:
|
||||
lines.append(f" {k} = {{{v}}},")
|
||||
lines.append("}")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def run(query, max_per_source=8):
|
||||
all_results = []
|
||||
errors = {}
|
||||
for name, fn in (("arxiv", search_arxiv), ("semantic_scholar", search_s2), ("crossref", search_crossref)):
|
||||
try:
|
||||
all_results.extend(fn(query, max_per_source))
|
||||
except Exception as e:
|
||||
errors[name] = str(e)
|
||||
|
||||
merged = merge_candidates(all_results)
|
||||
|
||||
used_keys = set()
|
||||
for cand in merged:
|
||||
result = verify(title=cand.get("title"), arxiv_id=cand.get("arxiv_id"), doi=cand.get("doi"))
|
||||
cand["verdict"] = result["verdict"]
|
||||
cand["verification_checks"] = result["checks"]
|
||||
if result["verdict"] == "verified":
|
||||
cand["bibtex_key"] = make_bibtex_key(cand, used_keys)
|
||||
|
||||
return {"query": query, "search_errors": errors, "candidates": merged}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
ap = argparse.ArgumentParser(description=__doc__)
|
||||
ap.add_argument("query")
|
||||
ap.add_argument("--max-per-source", type=int, default=8)
|
||||
ap.add_argument("--bib-out", default=None, help="path to write BibTeX for verified entries")
|
||||
args = ap.parse_args()
|
||||
|
||||
try:
|
||||
report = run(args.query, args.max_per_source)
|
||||
except Exception as e:
|
||||
print(json.dumps({"error": str(e)}, ensure_ascii=False))
|
||||
sys.exit(1)
|
||||
|
||||
print(json.dumps(report, ensure_ascii=False, indent=2))
|
||||
|
||||
if args.bib_out:
|
||||
verified = [c for c in report["candidates"] if c["verdict"] == "verified"]
|
||||
with open(args.bib_out, "w", encoding="utf-8") as f:
|
||||
for cand in verified:
|
||||
f.write(to_bibtex(cand, cand["bibtex_key"]))
|
||||
f.write("\n\n")
|
||||
sys.stderr.write(f"Wrote {len(verified)} verified BibTeX entries to {args.bib_out}\n")
|
||||
@@ -0,0 +1,80 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Search arXiv via its public Atom API. No API key required.
|
||||
|
||||
CLI usage:
|
||||
python3 search_arxiv.py "UAV magnetic compensation" --max 10
|
||||
|
||||
Importable:
|
||||
from search_arxiv import search_arxiv
|
||||
"""
|
||||
import sys
|
||||
import json
|
||||
import argparse
|
||||
import urllib.request
|
||||
import urllib.parse
|
||||
import xml.etree.ElementTree as ET
|
||||
|
||||
ARXIV_API = "http://export.arxiv.org/api/query"
|
||||
NS = {"atom": "http://www.w3.org/2005/Atom"}
|
||||
|
||||
|
||||
def search_arxiv(query, max_results=10, timeout=20):
|
||||
params = {
|
||||
"search_query": f"all:{query}",
|
||||
"start": 0,
|
||||
"max_results": max_results,
|
||||
"sortBy": "relevance",
|
||||
"sortOrder": "descending",
|
||||
}
|
||||
url = f"{ARXIV_API}?{urllib.parse.urlencode(params)}"
|
||||
with urllib.request.urlopen(url, timeout=timeout) as resp:
|
||||
data = resp.read()
|
||||
root = ET.fromstring(data)
|
||||
results = []
|
||||
for entry in root.findall("atom:entry", NS):
|
||||
id_el = entry.find("atom:id", NS)
|
||||
title_el = entry.find("atom:title", NS)
|
||||
summary_el = entry.find("atom:summary", NS)
|
||||
published_el = entry.find("atom:published", NS)
|
||||
if id_el is None or title_el is None:
|
||||
continue
|
||||
arxiv_id_full = id_el.text.strip()
|
||||
arxiv_id = arxiv_id_full.rsplit("/", 1)[-1]
|
||||
title = " ".join(title_el.text.split())
|
||||
summary = " ".join(summary_el.text.split()) if summary_el is not None else ""
|
||||
authors = [
|
||||
a.find("atom:name", NS).text
|
||||
for a in entry.findall("atom:author", NS)
|
||||
if a.find("atom:name", NS) is not None
|
||||
]
|
||||
published = published_el.text[:10] if published_el is not None else None
|
||||
pdf_url = None
|
||||
for link in entry.findall("atom:link", NS):
|
||||
if link.attrib.get("title") == "pdf":
|
||||
pdf_url = link.attrib.get("href")
|
||||
results.append({
|
||||
"source": "arxiv",
|
||||
"arxiv_id": arxiv_id,
|
||||
"title": title,
|
||||
"authors": authors,
|
||||
"year": published[:4] if published else None,
|
||||
"published": published,
|
||||
"abstract": summary,
|
||||
"pdf_url": pdf_url,
|
||||
"doi": None,
|
||||
})
|
||||
return results
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
ap = argparse.ArgumentParser(description=__doc__)
|
||||
ap.add_argument("query")
|
||||
ap.add_argument("--max", type=int, default=10)
|
||||
args = ap.parse_args()
|
||||
try:
|
||||
out = search_arxiv(args.query, args.max)
|
||||
print(json.dumps(out, ensure_ascii=False, indent=2))
|
||||
except Exception as e:
|
||||
print(json.dumps({"error": str(e)}, ensure_ascii=False))
|
||||
sys.exit(1)
|
||||
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Search the Crossref works API. No API key required.
|
||||
Good for journal articles / DOIs that arXiv and Semantic Scholar might miss.
|
||||
|
||||
CLI usage:
|
||||
python3 search_crossref.py "UAV magnetic compensation" --max 10
|
||||
|
||||
Importable:
|
||||
from search_crossref import search_crossref
|
||||
"""
|
||||
import sys
|
||||
import json
|
||||
import argparse
|
||||
import urllib.request
|
||||
import urllib.parse
|
||||
|
||||
CROSSREF_API = "https://api.crossref.org/works"
|
||||
UA = "literature-search-verify-skill/1.0 (mailto:research-assistant@example.com)"
|
||||
|
||||
|
||||
def search_crossref(query, max_results=10, timeout=20):
|
||||
params = {"query": query, "rows": max_results}
|
||||
url = f"{CROSSREF_API}?{urllib.parse.urlencode(params)}"
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
data = json.loads(resp.read())
|
||||
results = []
|
||||
for item in data.get("message", {}).get("items", []) or []:
|
||||
titles = item.get("title") or []
|
||||
title = titles[0] if titles else ""
|
||||
authors = []
|
||||
for a in item.get("author", []) or []:
|
||||
name = " ".join(filter(None, [a.get("given"), a.get("family")]))
|
||||
if name:
|
||||
authors.append(name)
|
||||
year = None
|
||||
date_parts = (item.get("issued", {}) or {}).get("date-parts")
|
||||
if date_parts and date_parts[0]:
|
||||
year = date_parts[0][0]
|
||||
containers = item.get("container-title") or []
|
||||
results.append({
|
||||
"source": "crossref",
|
||||
"title": title,
|
||||
"authors": authors,
|
||||
"year": year,
|
||||
"venue": containers[0] if containers else None,
|
||||
"doi": item.get("DOI"),
|
||||
"arxiv_id": None,
|
||||
"abstract": None,
|
||||
})
|
||||
return results
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
ap = argparse.ArgumentParser(description=__doc__)
|
||||
ap.add_argument("query")
|
||||
ap.add_argument("--max", type=int, default=10)
|
||||
args = ap.parse_args()
|
||||
try:
|
||||
out = search_crossref(args.query, args.max)
|
||||
print(json.dumps(out, ensure_ascii=False, indent=2))
|
||||
except Exception as e:
|
||||
print(json.dumps({"error": str(e)}, ensure_ascii=False))
|
||||
sys.exit(1)
|
||||
@@ -0,0 +1,59 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Search the Semantic Scholar Graph API. No API key required for light use;
|
||||
set the S2_API_KEY environment variable for higher rate limits.
|
||||
|
||||
CLI usage:
|
||||
python3 search_semantic_scholar.py "UAV magnetic compensation" --max 10
|
||||
|
||||
Importable:
|
||||
from search_semantic_scholar import search_s2
|
||||
"""
|
||||
import sys
|
||||
import os
|
||||
import json
|
||||
import argparse
|
||||
import urllib.request
|
||||
import urllib.parse
|
||||
|
||||
S2_API = "https://api.semanticscholar.org/graph/v1/paper/search"
|
||||
FIELDS = "title,authors,year,venue,externalIds,abstract,citationCount"
|
||||
|
||||
|
||||
def search_s2(query, max_results=10, timeout=20):
|
||||
params = {"query": query, "limit": max_results, "fields": FIELDS}
|
||||
url = f"{S2_API}?{urllib.parse.urlencode(params)}"
|
||||
req = urllib.request.Request(url)
|
||||
api_key = os.environ.get("S2_API_KEY")
|
||||
if api_key:
|
||||
req.add_header("x-api-key", api_key)
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
data = json.loads(resp.read())
|
||||
results = []
|
||||
for p in data.get("data", []) or []:
|
||||
ext = p.get("externalIds") or {}
|
||||
results.append({
|
||||
"source": "semantic_scholar",
|
||||
"title": p.get("title"),
|
||||
"authors": [a.get("name") for a in (p.get("authors") or [])],
|
||||
"year": p.get("year"),
|
||||
"venue": p.get("venue"),
|
||||
"doi": ext.get("DOI"),
|
||||
"arxiv_id": ext.get("ArXiv"),
|
||||
"abstract": p.get("abstract"),
|
||||
"citation_count": p.get("citationCount"),
|
||||
})
|
||||
return results
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
ap = argparse.ArgumentParser(description=__doc__)
|
||||
ap.add_argument("query")
|
||||
ap.add_argument("--max", type=int, default=10)
|
||||
args = ap.parse_args()
|
||||
try:
|
||||
out = search_s2(args.query, args.max)
|
||||
print(json.dumps(out, ensure_ascii=False, indent=2))
|
||||
except Exception as e:
|
||||
print(json.dumps({"error": str(e)}, ensure_ascii=False))
|
||||
sys.exit(1)
|
||||
@@ -0,0 +1,122 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Independently cross-verify a single candidate citation. This is the anti-
|
||||
hallucination check: it never trusts a single source. If a check cannot be
|
||||
run at all (e.g. no network), that check is reported as "skipped" -- never
|
||||
silently counted as a pass.
|
||||
|
||||
CLI usage:
|
||||
python3 verify_citation.py --title "Compensation of magnetic ..." \\
|
||||
--arxiv-id 2401.12345 --doi 10.1109/TGRS.2024.1234567
|
||||
|
||||
Importable:
|
||||
from verify_citation import verify
|
||||
"""
|
||||
import sys
|
||||
import json
|
||||
import argparse
|
||||
import difflib
|
||||
import urllib.request
|
||||
import urllib.parse
|
||||
import urllib.error
|
||||
import xml.etree.ElementTree as ET
|
||||
|
||||
ATOM_NS = {"atom": "http://www.w3.org/2005/Atom"}
|
||||
|
||||
|
||||
def title_similarity(a, b):
|
||||
if not a or not b:
|
||||
return 0.0
|
||||
return difflib.SequenceMatcher(None, a.lower().strip(), b.lower().strip()).ratio()
|
||||
|
||||
|
||||
def check_arxiv_id(arxiv_id, timeout=20):
|
||||
"""Confirm an arXiv ID actually resolves to a real paper."""
|
||||
try:
|
||||
url = f"http://export.arxiv.org/api/query?id_list={urllib.parse.quote(arxiv_id)}"
|
||||
with urllib.request.urlopen(url, timeout=timeout) as resp:
|
||||
data = resp.read()
|
||||
root = ET.fromstring(data)
|
||||
entry = root.find("atom:entry", ATOM_NS)
|
||||
if entry is None:
|
||||
return {"status": "fail", "reason": "arXiv ID not found"}
|
||||
title_el = entry.find("atom:title", ATOM_NS)
|
||||
title = " ".join(title_el.text.split()) if title_el is not None else None
|
||||
return {"status": "pass", "canonical_title": title}
|
||||
except Exception as e:
|
||||
return {"status": "skipped", "reason": str(e)}
|
||||
|
||||
|
||||
def check_doi(doi, timeout=20):
|
||||
"""Confirm a DOI actually resolves via Crossref."""
|
||||
try:
|
||||
url = f"https://api.crossref.org/works/{urllib.parse.quote(doi)}"
|
||||
req = urllib.request.Request(
|
||||
url, headers={"User-Agent": "literature-search-verify-skill/1.0"}
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
data = json.loads(resp.read())
|
||||
titles = data.get("message", {}).get("title") or []
|
||||
return {"status": "pass", "canonical_title": titles[0] if titles else None}
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code == 404:
|
||||
return {"status": "fail", "reason": "DOI not found in Crossref"}
|
||||
return {"status": "skipped", "reason": f"HTTP {e.code}"}
|
||||
except Exception as e:
|
||||
return {"status": "skipped", "reason": str(e)}
|
||||
|
||||
|
||||
def check_title_cross_source(title, timeout=20):
|
||||
"""Independently re-search by title on a different source (Semantic
|
||||
Scholar) and require a near-exact title match. This is what catches a
|
||||
plausible-sounding but entirely invented title/author combination."""
|
||||
try:
|
||||
params = {"query": title, "limit": 3, "fields": "title"}
|
||||
url = f"https://api.semanticscholar.org/graph/v1/paper/search?{urllib.parse.urlencode(params)}"
|
||||
with urllib.request.urlopen(url, timeout=timeout) as resp:
|
||||
data = json.loads(resp.read())
|
||||
candidates = data.get("data", []) or []
|
||||
if not candidates:
|
||||
return {"status": "fail", "reason": "no matching title found on Semantic Scholar"}
|
||||
best = max(candidates, key=lambda p: title_similarity(title, p.get("title", "")))
|
||||
sim = title_similarity(title, best.get("title", ""))
|
||||
if sim >= 0.9:
|
||||
return {"status": "pass", "similarity": round(sim, 3), "matched_title": best.get("title")}
|
||||
return {"status": "fail", "similarity": round(sim, 3), "matched_title": best.get("title")}
|
||||
except Exception as e:
|
||||
return {"status": "skipped", "reason": str(e)}
|
||||
|
||||
|
||||
def verify(title=None, arxiv_id=None, doi=None):
|
||||
checks = {}
|
||||
if arxiv_id:
|
||||
checks["arxiv_id_check"] = check_arxiv_id(arxiv_id)
|
||||
if doi:
|
||||
checks["doi_check"] = check_doi(doi)
|
||||
if title:
|
||||
checks["title_cross_source_check"] = check_title_cross_source(title)
|
||||
|
||||
passed = [c for c in checks.values() if c["status"] == "pass"]
|
||||
failed = [c for c in checks.values() if c["status"] == "fail"]
|
||||
|
||||
if failed:
|
||||
verdict = "suspect" # something actively contradicted it
|
||||
elif passed:
|
||||
verdict = "verified" # at least one independent check passed
|
||||
else:
|
||||
verdict = "unverified" # everything skipped (e.g. no network) -- NOT the same as verified
|
||||
|
||||
return {"title": title, "arxiv_id": arxiv_id, "doi": doi, "verdict": verdict, "checks": checks}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
ap = argparse.ArgumentParser(description=__doc__)
|
||||
ap.add_argument("--title", default=None)
|
||||
ap.add_argument("--arxiv-id", default=None)
|
||||
ap.add_argument("--doi", default=None)
|
||||
args = ap.parse_args()
|
||||
if not any([args.title, args.arxiv_id, args.doi]):
|
||||
print(json.dumps({"error": "provide at least one of --title/--arxiv-id/--doi"}))
|
||||
sys.exit(1)
|
||||
result = verify(args.title, args.arxiv_id, args.doi)
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2))
|
||||
Reference in New Issue
Block a user