This commit is contained in:
2026-07-20 03:58:38 -10:00
commit c3e37f642a
15 changed files with 1104 additions and 0 deletions
@@ -0,0 +1,165 @@
#!/usr/bin/env python3
"""
Archive a finished literature-search-verify session into a permanent,
project-level folder instead of leaving results sitting in the skill's
own scratch output/ directory (which is easy to lose track of across
sessions and isn't meant to be a durable deliverable location).
Bundles the verified BibTeX file -- and, if given, any downloaded PDFs --
into <project-root>/references/<topic-slug>/, and writes a README.md
index (entry list, suspect/unverified entries flagged separately, free-
text coverage notes) so a future session or a human can find and trust
what's there without re-reading the conversation that produced it.
No third-party dependencies; uses only the standard library.
CLI usage:
python3 archive_references.py "UAV aeromagnetic compensation" \\
--bib output/uav_aeromagnetic_compensation_final.bib \\
--project-root . \\
--pdfs-dir output/pdfs \\
--suspect "Some fabricated-looking title|DOI resolves but venue is topically unrelated" \\
--notes "Kalman-filter and GA/PSO angles searched, no on-topic hits found."
Output: prints the path of the archive directory that was created/updated.
"""
import argparse
import os
import re
import shutil
import sys
from datetime import date
def slugify(text):
text = text.strip().lower()
text = re.sub(r"[^a-z0-9]+", "_", text)
return text.strip("_")[:60] or "references"
def parse_bib_entries(bib_path):
"""Minimal BibTeX parser -- just enough to pull key/title/year/venue/doi/note
(plus the raw entry text, for reordering) for the README index. Not a
general-purpose BibTeX parser."""
with open(bib_path, encoding="utf-8") as f:
content = f.read()
entries = []
for m in re.finditer(r"@(\w+)\{([^,\n]+),(.*?)\n\}", content, re.S):
entry_type, key, body = m.groups()
fields = {}
for fm in re.finditer(r"(\w+)\s*=\s*\{(.*?)\}\s*,?\s*(?=\n\s*\w+\s*=|\n\Z|\Z)", body, re.S):
fields[fm.group(1).lower()] = re.sub(r"\s+", " ", fm.group(2)).strip()
entries.append({"type": entry_type, "key": key.strip(), "raw": m.group(0).strip(), **fields})
return entries
def year_sort_key(entry):
"""Chronological order, oldest first; entries with no parseable year sort last."""
year_str = re.sub(r"[^0-9]", "", entry.get("year", "") or "")
year = int(year_str) if year_str else 9999
return (year, entry.get("key", ""))
def build_readme(topic, entries, pdf_count, suspect, notes):
lines = []
lines.append(f"# {topic} — literature archive")
lines.append("")
lines.append(f"Archived: {date.today().isoformat()}")
lines.append(f"Verified entries: {len(entries)}")
lines.append(f"PDFs bundled: {pdf_count}")
lines.append("")
lines.append(
"Every entry in `references.bib` passed independent verification "
"(arXiv ID / DOI resolution and/or cross-source title match, "
"similarity >= 0.9) via the literature-search-verify skill before "
"being archived here. Citation keys follow the surname+year "
"convention and are stable -- the paper-writing-grounded skill's "
"`\\cite{}` calls should match these keys directly."
)
lines.append("")
lines.append("## Entries (chronological, oldest first)")
lines.append("")
for e in entries:
title = e.get("title", "?")
year = e.get("year", "?")
venue = e.get("journal") or e.get("booktitle") or e.get("school") or ""
doi = e.get("doi", "")
note = e.get("note", "")
line = f"- **{e['key']}** ({year}) — {title}"
if venue:
line += f". *{venue}*"
if doi:
line += f". DOI: {doi}"
lines.append(line)
if note:
lines.append(f" - Note: {note}")
if suspect:
lines.append("")
lines.append("## Flagged during search — NOT included above, do not cite")
lines.append("")
for s in suspect:
parts = s.split("|", 1)
title = parts[0].strip()
reason = parts[1].strip() if len(parts) > 1 else ""
lines.append(f"- {title}" + (f" — {reason}" if reason else ""))
if notes:
lines.append("")
lines.append("## Search coverage notes")
lines.append("")
lines.append(notes)
return "\n".join(lines) + "\n"
def main():
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("topic", help="Human-readable topic name, e.g. \"UAV aeromagnetic compensation\"")
ap.add_argument("--bib", required=True, help="path to the curated/verified .bib file to archive")
ap.add_argument("--project-root", default=".", help="project root; archive is written under <root>/references/<slug>/")
ap.add_argument("--pdfs-dir", default=None, help="optional folder of open-access PDFs to copy alongside the bib")
ap.add_argument("--suspect", action="append", default=[], help="title|reason of a suspect/unverified entry to log; repeatable")
ap.add_argument("--notes", default=None, help="free-text notes on search coverage/gaps for the README")
args = ap.parse_args()
if not os.path.isfile(args.bib):
print(f"error: bib file not found: {args.bib}", file=sys.stderr)
sys.exit(1)
slug = slugify(args.topic)
archive_dir = os.path.join(args.project_root, "references", slug)
os.makedirs(archive_dir, exist_ok=True)
bib_dest = os.path.join(archive_dir, "references.bib")
shutil.copyfile(args.bib, bib_dest)
entries = parse_bib_entries(bib_dest)
entries.sort(key=year_sort_key)
# Rewrite the archived .bib in chronological order (oldest first) so the
# file itself, not just the README, reads as a timeline.
header = f"% {args.topic} -- verified references, chronological order\n% Archived {date.today().isoformat()}\n\n"
with open(bib_dest, "w", encoding="utf-8") as f:
f.write(header)
f.write("\n\n".join(e["raw"] for e in entries))
f.write("\n")
pdf_count = 0
if args.pdfs_dir and os.path.isdir(args.pdfs_dir):
pdf_dest_dir = os.path.join(archive_dir, "pdfs")
os.makedirs(pdf_dest_dir, exist_ok=True)
for fn in sorted(os.listdir(args.pdfs_dir)):
if fn.lower().endswith(".pdf"):
shutil.copyfile(os.path.join(args.pdfs_dir, fn), os.path.join(pdf_dest_dir, fn))
pdf_count += 1
readme = build_readme(args.topic, entries, pdf_count, args.suspect, args.notes)
with open(os.path.join(archive_dir, "README.md"), "w", encoding="utf-8") as f:
f.write(readme)
print(archive_dir)
if __name__ == "__main__":
main()
@@ -0,0 +1,160 @@
#!/usr/bin/env python3
"""
End-to-end literature search: query arXiv + Semantic Scholar + Crossref,
merge/dedupe candidates, independently verify each one, and emit both a
human-readable report and BibTeX for the entries that passed verification.
This is the one script Claude should actually call for a normal literature
search -- the individual search_*.py / verify_citation.py scripts exist
mainly as building blocks it can reuse for one-off / follow-up lookups.
CLI usage:
python3 literature_search.py "UAV magnetic compensation Tolles-Lawson" \\
--max-per-source 8 --bib-out refs.bib
Output: prints a JSON report to stdout (one entry per merged candidate,
with its verdict), and if --bib-out is given, writes BibTeX for every
"verified" entry to that file (never for "suspect" or "unverified" ones).
"""
import sys
import os
import json
import argparse
import difflib
import re
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from search_arxiv import search_arxiv
from search_semantic_scholar import search_s2
from search_crossref import search_crossref
from verify_citation import verify
def _similar(a, b, threshold=0.88):
if not a or not b:
return False
return difflib.SequenceMatcher(None, a.lower().strip(), b.lower().strip()).ratio() >= threshold
def merge_candidates(all_results):
"""Dedupe candidates that are the same paper found via multiple sources,
merging their metadata (preferring whichever source has an identifier)."""
merged = []
for item in all_results:
placed = False
for m in merged:
if _similar(item.get("title"), m.get("title")):
# merge: fill in any missing fields, keep track of all sources
for key in ("doi", "arxiv_id", "abstract", "venue", "year", "citation_count", "pdf_url"):
if not m.get(key) and item.get(key):
m[key] = item[key]
m["sources"] = sorted(set(m.get("sources", [m.get("source")]) + [item.get("source")]))
placed = True
break
if not placed:
item = dict(item)
item["sources"] = [item.get("source")]
merged.append(item)
return merged
def make_bibtex_key(candidate, used_keys):
authors = candidate.get("authors") or []
surname = "unknown"
if authors:
first_author = authors[0]
surname = first_author.strip().split()[-1].lower()
surname = re.sub(r"[^a-z]", "", surname) or "unknown"
year = str(candidate.get("year") or "nd")
base = f"{surname}{year}"
key = base
suffix = ord("a")
while key in used_keys:
key = f"{base}{chr(suffix)}"
suffix += 1
used_keys.add(key)
return key
def to_bibtex(candidate, key):
authors = candidate.get("authors") or []
author_str = " and ".join(authors) if authors else "Unknown"
title = candidate.get("title") or ""
year = candidate.get("year") or ""
venue = candidate.get("venue") or ""
doi = candidate.get("doi") or ""
arxiv_id = candidate.get("arxiv_id") or ""
if arxiv_id and not venue:
entry_type = "misc"
fields = [
("author", author_str),
("title", title),
("year", str(year)),
("eprint", arxiv_id),
("archivePrefix", "arXiv"),
]
else:
entry_type = "article"
fields = [
("author", author_str),
("title", title),
("journal", venue),
("year", str(year)),
]
if doi:
fields.append(("doi", doi))
lines = [f"@{entry_type}{{{key},"]
for k, v in fields:
if v:
lines.append(f" {k} = {{{v}}},")
lines.append("}")
return "\n".join(lines)
def run(query, max_per_source=8):
all_results = []
errors = {}
for name, fn in (("arxiv", search_arxiv), ("semantic_scholar", search_s2), ("crossref", search_crossref)):
try:
all_results.extend(fn(query, max_per_source))
except Exception as e:
errors[name] = str(e)
merged = merge_candidates(all_results)
used_keys = set()
for cand in merged:
result = verify(title=cand.get("title"), arxiv_id=cand.get("arxiv_id"), doi=cand.get("doi"))
cand["verdict"] = result["verdict"]
cand["verification_checks"] = result["checks"]
if result["verdict"] == "verified":
cand["bibtex_key"] = make_bibtex_key(cand, used_keys)
return {"query": query, "search_errors": errors, "candidates": merged}
if __name__ == "__main__":
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("query")
ap.add_argument("--max-per-source", type=int, default=8)
ap.add_argument("--bib-out", default=None, help="path to write BibTeX for verified entries")
args = ap.parse_args()
try:
report = run(args.query, args.max_per_source)
except Exception as e:
print(json.dumps({"error": str(e)}, ensure_ascii=False))
sys.exit(1)
print(json.dumps(report, ensure_ascii=False, indent=2))
if args.bib_out:
verified = [c for c in report["candidates"] if c["verdict"] == "verified"]
with open(args.bib_out, "w", encoding="utf-8") as f:
for cand in verified:
f.write(to_bibtex(cand, cand["bibtex_key"]))
f.write("\n\n")
sys.stderr.write(f"Wrote {len(verified)} verified BibTeX entries to {args.bib_out}\n")
@@ -0,0 +1,80 @@
#!/usr/bin/env python3
"""
Search arXiv via its public Atom API. No API key required.
CLI usage:
python3 search_arxiv.py "UAV magnetic compensation" --max 10
Importable:
from search_arxiv import search_arxiv
"""
import sys
import json
import argparse
import urllib.request
import urllib.parse
import xml.etree.ElementTree as ET
ARXIV_API = "http://export.arxiv.org/api/query"
NS = {"atom": "http://www.w3.org/2005/Atom"}
def search_arxiv(query, max_results=10, timeout=20):
params = {
"search_query": f"all:{query}",
"start": 0,
"max_results": max_results,
"sortBy": "relevance",
"sortOrder": "descending",
}
url = f"{ARXIV_API}?{urllib.parse.urlencode(params)}"
with urllib.request.urlopen(url, timeout=timeout) as resp:
data = resp.read()
root = ET.fromstring(data)
results = []
for entry in root.findall("atom:entry", NS):
id_el = entry.find("atom:id", NS)
title_el = entry.find("atom:title", NS)
summary_el = entry.find("atom:summary", NS)
published_el = entry.find("atom:published", NS)
if id_el is None or title_el is None:
continue
arxiv_id_full = id_el.text.strip()
arxiv_id = arxiv_id_full.rsplit("/", 1)[-1]
title = " ".join(title_el.text.split())
summary = " ".join(summary_el.text.split()) if summary_el is not None else ""
authors = [
a.find("atom:name", NS).text
for a in entry.findall("atom:author", NS)
if a.find("atom:name", NS) is not None
]
published = published_el.text[:10] if published_el is not None else None
pdf_url = None
for link in entry.findall("atom:link", NS):
if link.attrib.get("title") == "pdf":
pdf_url = link.attrib.get("href")
results.append({
"source": "arxiv",
"arxiv_id": arxiv_id,
"title": title,
"authors": authors,
"year": published[:4] if published else None,
"published": published,
"abstract": summary,
"pdf_url": pdf_url,
"doi": None,
})
return results
if __name__ == "__main__":
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("query")
ap.add_argument("--max", type=int, default=10)
args = ap.parse_args()
try:
out = search_arxiv(args.query, args.max)
print(json.dumps(out, ensure_ascii=False, indent=2))
except Exception as e:
print(json.dumps({"error": str(e)}, ensure_ascii=False))
sys.exit(1)
@@ -0,0 +1,65 @@
#!/usr/bin/env python3
"""
Search the Crossref works API. No API key required.
Good for journal articles / DOIs that arXiv and Semantic Scholar might miss.
CLI usage:
python3 search_crossref.py "UAV magnetic compensation" --max 10
Importable:
from search_crossref import search_crossref
"""
import sys
import json
import argparse
import urllib.request
import urllib.parse
CROSSREF_API = "https://api.crossref.org/works"
UA = "literature-search-verify-skill/1.0 (mailto:research-assistant@example.com)"
def search_crossref(query, max_results=10, timeout=20):
params = {"query": query, "rows": max_results}
url = f"{CROSSREF_API}?{urllib.parse.urlencode(params)}"
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=timeout) as resp:
data = json.loads(resp.read())
results = []
for item in data.get("message", {}).get("items", []) or []:
titles = item.get("title") or []
title = titles[0] if titles else ""
authors = []
for a in item.get("author", []) or []:
name = " ".join(filter(None, [a.get("given"), a.get("family")]))
if name:
authors.append(name)
year = None
date_parts = (item.get("issued", {}) or {}).get("date-parts")
if date_parts and date_parts[0]:
year = date_parts[0][0]
containers = item.get("container-title") or []
results.append({
"source": "crossref",
"title": title,
"authors": authors,
"year": year,
"venue": containers[0] if containers else None,
"doi": item.get("DOI"),
"arxiv_id": None,
"abstract": None,
})
return results
if __name__ == "__main__":
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("query")
ap.add_argument("--max", type=int, default=10)
args = ap.parse_args()
try:
out = search_crossref(args.query, args.max)
print(json.dumps(out, ensure_ascii=False, indent=2))
except Exception as e:
print(json.dumps({"error": str(e)}, ensure_ascii=False))
sys.exit(1)
@@ -0,0 +1,59 @@
#!/usr/bin/env python3
"""
Search the Semantic Scholar Graph API. No API key required for light use;
set the S2_API_KEY environment variable for higher rate limits.
CLI usage:
python3 search_semantic_scholar.py "UAV magnetic compensation" --max 10
Importable:
from search_semantic_scholar import search_s2
"""
import sys
import os
import json
import argparse
import urllib.request
import urllib.parse
S2_API = "https://api.semanticscholar.org/graph/v1/paper/search"
FIELDS = "title,authors,year,venue,externalIds,abstract,citationCount"
def search_s2(query, max_results=10, timeout=20):
params = {"query": query, "limit": max_results, "fields": FIELDS}
url = f"{S2_API}?{urllib.parse.urlencode(params)}"
req = urllib.request.Request(url)
api_key = os.environ.get("S2_API_KEY")
if api_key:
req.add_header("x-api-key", api_key)
with urllib.request.urlopen(req, timeout=timeout) as resp:
data = json.loads(resp.read())
results = []
for p in data.get("data", []) or []:
ext = p.get("externalIds") or {}
results.append({
"source": "semantic_scholar",
"title": p.get("title"),
"authors": [a.get("name") for a in (p.get("authors") or [])],
"year": p.get("year"),
"venue": p.get("venue"),
"doi": ext.get("DOI"),
"arxiv_id": ext.get("ArXiv"),
"abstract": p.get("abstract"),
"citation_count": p.get("citationCount"),
})
return results
if __name__ == "__main__":
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("query")
ap.add_argument("--max", type=int, default=10)
args = ap.parse_args()
try:
out = search_s2(args.query, args.max)
print(json.dumps(out, ensure_ascii=False, indent=2))
except Exception as e:
print(json.dumps({"error": str(e)}, ensure_ascii=False))
sys.exit(1)
@@ -0,0 +1,122 @@
#!/usr/bin/env python3
"""
Independently cross-verify a single candidate citation. This is the anti-
hallucination check: it never trusts a single source. If a check cannot be
run at all (e.g. no network), that check is reported as "skipped" -- never
silently counted as a pass.
CLI usage:
python3 verify_citation.py --title "Compensation of magnetic ..." \\
--arxiv-id 2401.12345 --doi 10.1109/TGRS.2024.1234567
Importable:
from verify_citation import verify
"""
import sys
import json
import argparse
import difflib
import urllib.request
import urllib.parse
import urllib.error
import xml.etree.ElementTree as ET
ATOM_NS = {"atom": "http://www.w3.org/2005/Atom"}
def title_similarity(a, b):
if not a or not b:
return 0.0
return difflib.SequenceMatcher(None, a.lower().strip(), b.lower().strip()).ratio()
def check_arxiv_id(arxiv_id, timeout=20):
"""Confirm an arXiv ID actually resolves to a real paper."""
try:
url = f"http://export.arxiv.org/api/query?id_list={urllib.parse.quote(arxiv_id)}"
with urllib.request.urlopen(url, timeout=timeout) as resp:
data = resp.read()
root = ET.fromstring(data)
entry = root.find("atom:entry", ATOM_NS)
if entry is None:
return {"status": "fail", "reason": "arXiv ID not found"}
title_el = entry.find("atom:title", ATOM_NS)
title = " ".join(title_el.text.split()) if title_el is not None else None
return {"status": "pass", "canonical_title": title}
except Exception as e:
return {"status": "skipped", "reason": str(e)}
def check_doi(doi, timeout=20):
"""Confirm a DOI actually resolves via Crossref."""
try:
url = f"https://api.crossref.org/works/{urllib.parse.quote(doi)}"
req = urllib.request.Request(
url, headers={"User-Agent": "literature-search-verify-skill/1.0"}
)
with urllib.request.urlopen(req, timeout=timeout) as resp:
data = json.loads(resp.read())
titles = data.get("message", {}).get("title") or []
return {"status": "pass", "canonical_title": titles[0] if titles else None}
except urllib.error.HTTPError as e:
if e.code == 404:
return {"status": "fail", "reason": "DOI not found in Crossref"}
return {"status": "skipped", "reason": f"HTTP {e.code}"}
except Exception as e:
return {"status": "skipped", "reason": str(e)}
def check_title_cross_source(title, timeout=20):
"""Independently re-search by title on a different source (Semantic
Scholar) and require a near-exact title match. This is what catches a
plausible-sounding but entirely invented title/author combination."""
try:
params = {"query": title, "limit": 3, "fields": "title"}
url = f"https://api.semanticscholar.org/graph/v1/paper/search?{urllib.parse.urlencode(params)}"
with urllib.request.urlopen(url, timeout=timeout) as resp:
data = json.loads(resp.read())
candidates = data.get("data", []) or []
if not candidates:
return {"status": "fail", "reason": "no matching title found on Semantic Scholar"}
best = max(candidates, key=lambda p: title_similarity(title, p.get("title", "")))
sim = title_similarity(title, best.get("title", ""))
if sim >= 0.9:
return {"status": "pass", "similarity": round(sim, 3), "matched_title": best.get("title")}
return {"status": "fail", "similarity": round(sim, 3), "matched_title": best.get("title")}
except Exception as e:
return {"status": "skipped", "reason": str(e)}
def verify(title=None, arxiv_id=None, doi=None):
checks = {}
if arxiv_id:
checks["arxiv_id_check"] = check_arxiv_id(arxiv_id)
if doi:
checks["doi_check"] = check_doi(doi)
if title:
checks["title_cross_source_check"] = check_title_cross_source(title)
passed = [c for c in checks.values() if c["status"] == "pass"]
failed = [c for c in checks.values() if c["status"] == "fail"]
if failed:
verdict = "suspect" # something actively contradicted it
elif passed:
verdict = "verified" # at least one independent check passed
else:
verdict = "unverified" # everything skipped (e.g. no network) -- NOT the same as verified
return {"title": title, "arxiv_id": arxiv_id, "doi": doi, "verdict": verdict, "checks": checks}
if __name__ == "__main__":
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("--title", default=None)
ap.add_argument("--arxiv-id", default=None)
ap.add_argument("--doi", default=None)
args = ap.parse_args()
if not any([args.title, args.arxiv_id, args.doi]):
print(json.dumps({"error": "provide at least one of --title/--arxiv-id/--doi"}))
sys.exit(1)
result = verify(args.title, args.arxiv_id, args.doi)
print(json.dumps(result, ensure_ascii=False, indent=2))