Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 5 additions & 5 deletions docs/examples/scholar-metrics-analysis.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,11 +17,11 @@ The SearchAPI-backed utilities cover the following workflows:

| Task | Function | Result |
| --- | --- | --- |
| Search Scholar | `get_scholar_papers` | Paper metadata as a DataFrame |
| Count citations | `get_citations_from_title` | Google Scholar citation count |
| Export a citation | `get_bibtex_entry`, `get_endnote_entry` | BibTeX or EndNote text |
| Find citing papers | `get_citing_papers_from_title` | A list of `Paper` objects |
| Find an author's papers | `get_scholar_author_papers` | Author-paper metadata as a DataFrame |
| Search Scholar | [`get_scholar_papers`][paperscraper.scholar.get_scholar_papers] | Paper metadata as a DataFrame |
| Count citations | [`get_citations_from_title`][paperscraper.citations.get_citations_from_title] | Google Scholar citation count |
| Export a citation | [`get_bibtex_entry`][paperscraper.citations.get_bibtex_entry], [`get_endnote_entry`][paperscraper.citations.get_endnote_entry] | BibTeX or EndNote text |
| Find citing papers | [`get_citing_papers_from_title`][paperscraper.citations.get_citing_papers_from_title] | A list of `Paper` objects |
| Find an author's papers | [`get_scholar_author_papers`][paperscraper.scholar.get_scholar_author_papers] | Author-paper metadata as a DataFrame |

SearchAPI calls are retried with bounded exponential backoff. The complete
outputs below were captured from live calls on 23 September 2026. They reflect
Expand Down
241 changes: 61 additions & 180 deletions paperscraper/citations/citations.py
Original file line number Diff line number Diff line change
@@ -1,24 +1,18 @@
import logging
import re
import sys
from typing import Iterable, Literal, Optional

from scholarly import scholarly
from semanticscholar import SemanticScholarException

from ..utils import retry_with_exponential_backoff
from ..searchapi.core import SEARCH_API_KEY, SearchAPICitations, SearchAPIClient
from ..utils import _resolve_backend
from .entity import Paper
from .searchapi import (
SEARCH_API_CACHE,
_cache_search_api_citation,
_get_citation_count_from_searchapi_html,
_get_citations_from_title_scholarly,
_get_citations_from_title_semantic_scholar,
_get_searchapi_citation_entry,
_get_searchapi_citing_papers,
_resolve_citation_backend,
search_api_requests_get,
)
from .utils import (
DOI_PATTERN,
PAPER_URL,
SS_API_KEY,
_semantic_scholar_requests_get_with_backoff,
get_doi_from_title,
)
Expand Down Expand Up @@ -73,7 +67,20 @@ def get_citation_entry(
if format not in {"endnote", "bibtex"}:
raise ValueError("format must be 'endnote' or 'bibtex'")

return _get_searchapi_citation_entry(title_or_doi.strip(), format, api_key)
title_or_doi = title_or_doi.strip()
doi = re.search(DOI_PATTERN, title_or_doi, re.IGNORECASE)
search_title = None
if doi:
response = _semantic_scholar_requests_get_with_backoff(
f"{PAPER_URL}DOI:{doi.group(0)}",
params={"fields": "title"},
base_delay=5.0,
factor=2.0,
)
search_title = response.json().get("title") or ""
return SearchAPICitations(SearchAPIClient(api_key)).get_citation_entry(
title_or_doi, format, search_title=search_title
)


def get_bibtex_entry(title_or_doi: str, *, api_key: Optional[str] = None) -> str:
Expand Down Expand Up @@ -125,8 +132,8 @@ def get_citing_papers_from_title(
if not isinstance(full_info, bool):
raise TypeError(f"Pass bool not {type(full_info)}")

results = _get_searchapi_citing_papers(
title.strip(), api_key, max_results=max_results
results = SearchAPICitations(SearchAPIClient(api_key)).get_citing_papers(
title.strip(), max_results=max_results
)
papers = []
for result in results:
Expand Down Expand Up @@ -190,175 +197,49 @@ def get_citation_count_from_searchapi_author(
author_names: Optional[Iterable[str]] = None,
) -> Optional[int]:
"""Retrieve a canonical count through a matching Scholar author profile."""
normalized_title = " ".join(title.casefold().split())
candidate_authors = set(author_names or ())

# Discover authors when the initial paper search did not provide them.
@retry_with_exponential_backoff(retry_if=lambda found: not found, base_delay=0)
def discover_authors() -> bool:
response = search_api_requests_get(
api_key=api_key,
params={
"engine": "google_scholar",
"q": f"allintitle: {title}",
"hl": "en",
"num": 20,
},
)
candidate_authors.update(
author["name"]
for paper in response.json().get("organic_results", [])
if " ".join(paper.get("title", "").casefold().split()).startswith(
normalized_title
)
for author in paper.get("authors", [])
if author.get("name")
)
return bool(candidate_authors)

if not candidate_authors:
discover_authors()

# Resolve the most specific candidate names to Scholar profiles.
for author_name in sorted(candidate_authors, key=len, reverse=True)[:3]:
response = search_api_requests_get(
api_key=api_key,
params={
"engine": "google_scholar",
"q": f"author:{author_name}",
"hl": "en",
"num": 20,
},
)
for profile in response.json().get("profiles", [])[:3]:
author_id = profile.get("author_id")
if not author_id:
continue

# Only accept a citation_id from an exact article-title match.
response = search_api_requests_get(
api_key=api_key,
params={
"engine": "google_scholar_author",
"author_id": author_id,
},
)
article = next(
(
article
for article in response.json().get("articles", [])
if " ".join(article.get("title", "").casefold().split())
== normalized_title
),
None,
)
if article is None or not article.get("citation_id"):
continue

# Fetch the paper-level count and verify the title once more.
response = search_api_requests_get(
api_key=api_key,
params={
"engine": "google_scholar_author",
"view_op": "view_citation",
"citation_id": article["citation_id"],
},
)
scholar_article = (
response.json().get("citation", {}).get("scholar_articles", {})
)
if (
" ".join(scholar_article.get("title", "").casefold().split())
!= normalized_title
):
continue
total = scholar_article.get("cited_by", {}).get("total")
if total is not None:
return int(total)
return None
return SearchAPICitations(SearchAPIClient(api_key)).get_citation_count_from_author(
title, author_names=author_names
)


def get_citations_from_title_searchapi(title: str, api_key: Optional[str]) -> int:
"""Retrieve a Google Scholar citation count through SearchApi."""
normalized_title = " ".join(title.casefold().split())
citation_cache = SEARCH_API_CACHE["citations"]
if title in citation_cache:
return citation_cache[title]

search_ids = []
author_names = set()

@retry_with_exponential_backoff(retry_if=lambda count: count is None, base_delay=0)
def search() -> Optional[int]:
response = search_api_requests_get(
api_key=api_key,
params={
"engine": "google_scholar",
"q": f'"{title}"',
"hl": "en",
"num": 20,
},
)
data = response.json()
metadata = data.get("search_metadata", {})
search_ids.append(metadata.get("id", "unknown"))
author_names.update(
author["name"]
for paper in data.get("organic_results", [])
if " ".join(paper.get("title", "").casefold().split()).startswith(
normalized_title
)
for author in paper.get("authors", [])
if author.get("name")
)
exact_matches = [
paper
for paper in data.get("organic_results", [])
if " ".join(paper.get("title", "").casefold().split()) == normalized_title
]
if not exact_matches:
return None

preferred_matches = [
paper for paper in exact_matches if paper.get("type") != "CITATION"
] or exact_matches
counts = {
int(cited_by["total"])
for paper in preferred_matches
if (cited_by := paper.get("inline_links", {}).get("cited_by", {})).get(
"total"
)
is not None
}
if len(counts) == 1:
count = counts.pop()
return _cache_search_api_citation(title, count)
if len(counts) > 1:
raise RuntimeError(f"SearchApi returned conflicting counts for {title!r}.")

data_cids = {
paper["data_cid"] for paper in preferred_matches if paper.get("data_cid")
}
html_url = metadata.get("html_url")
if html_url and data_cids:
count = _get_citation_count_from_searchapi_html(
html_url, data_cids, api_key
)
if count is not None:
return _cache_search_api_citation(title, count)
return None

count = search()
if count is not None:
return count

count = get_citation_count_from_searchapi_author(
title, api_key, author_names=author_names
return SearchAPICitations(SearchAPIClient(api_key)).get_citations_from_title(title)


def _get_citations_from_title_scholarly(title: str) -> int:
"""Retrieve a Google Scholar citation count through scholarly."""
matches = scholarly.search_pubs(f'"{title}"')
counts = [int(paper["num_citations"]) for paper in matches]
if len(counts) == 0:
logger.warning(f"Found no match for {title}.")
return 0
if len(counts) > 1:
logger.warning(f"Found {len(counts)} matches for {title}, returning first one.")
return counts[0]


def _get_citations_from_title_semantic_scholar(
title: str, api_key: Optional[str]
) -> int:
"""Retrieve a Semantic Scholar citation count."""
response = _semantic_scholar_requests_get_with_backoff(
f"{PAPER_URL}search",
params={"query": title, "fields": "citationCount", "limit": 1},
api_key=api_key,
)
if count is not None:
return _cache_search_api_citation(title, count)
matches = response.json().get("data", [])
if not matches:
logger.warning(f"Found no match for {title}.")
return 0
return int(matches[0].get("citationCount") or 0)


raise RuntimeError(
f"SearchApi returned no complete exact match for {title!r} "
f"(search IDs: {', '.join(search_ids)})."
def _resolve_citation_backend(backend: str, api_key: Optional[str]) -> str:
"""Resolve the citation backend."""
return _resolve_backend(
backend,
api_key,
(("searchapi", SEARCH_API_KEY), ("semantic_scholar", SS_API_KEY)),
"scholarly",
)
Loading
Loading