Skip to content

paperscraper.scholar

paperscraper.scholar

search_api_requests_get(*, api_key: Optional[str] = None, url: str = SEARCH_API_URL, **kwargs) -> requests.Response

Perform an authenticated SearchApi request.

Parameters:

Name Type Description Default
api_key Optional[str]

Explicit API key.

None
url str

SearchApi URL.

SEARCH_API_URL

Returns:

Type Description
Response

requests.Response: API response.

Source code in paperscraper/citations/utils.py
def search_api_requests_get(
    *, api_key: Optional[str] = None, url: str = SEARCH_API_URL, **kwargs
) -> requests.Response:
    """
    Perform an authenticated SearchApi request.

    Args:
        api_key: Explicit API key.
        url: SearchApi URL.

    Returns:
        requests.Response: API response.
    """
    resolved_api_key = api_key if api_key is not None else SEARCH_API_KEY
    if not resolved_api_key:
        raise ValueError(
            "SearchApi requires api_key or the SEARCH_API_KEY environment variable."
        )

    response = requests.get(
        url,
        headers={"Authorization": f"Bearer {resolved_api_key}"},
        timeout=90,
        **kwargs,
    )
    response.raise_for_status()
    return response

dump_papers(papers: pd.DataFrame, filepath: str) -> None

Receives a pd.DataFrame, one paper per row and dumps it into a .jsonl file with one paper per line.

Parameters:

Name Type Description Default
papers DataFrame

A dataframe of paper metadata, one paper per row.

required
filepath str

Path to dump the papers, has to end with .jsonl.

required
Source code in paperscraper/utils.py
def dump_papers(papers: pd.DataFrame, filepath: str) -> None:
    """
    Receives a pd.DataFrame, one paper per row and dumps it into a .jsonl
    file with one paper per line.

    Args:
        papers (pd.DataFrame): A dataframe of paper metadata, one paper per row.
        filepath (str): Path to dump the papers, has to end with `.jsonl`.
    """
    if not isinstance(filepath, str):
        raise TypeError(f"filepath must be a string, not {type(filepath)}")
    if not filepath.endswith(".jsonl"):
        raise ValueError("Please provide a filepath with .jsonl extension")

    if isinstance(papers, List) and all([isinstance(p, Dict) for p in papers]):
        papers = pd.DataFrame(papers)
        logger.warning(
            "Preferably pass a pd.DataFrame, not a list of dictionaries. "
            "Passing a list is a legacy functionality that might become deprecated."
        )

    if not isinstance(papers, pd.DataFrame):
        raise TypeError(f"papers must be a pd.DataFrame, not {type(papers)}")

    paper_list = list(papers.T.to_dict().values())

    with open(filepath, "w") as f:
        for paper in paper_list:
            f.write(json.dumps(paper) + "\n")

retry_with_exponential_backoff(*, max_attempts: int = 3, retry_if: Callable[[T], bool] = lambda result: False, exceptions: Tuple[Type[BaseException], ...] = (), base_delay: float = 1) -> Callable[[Callable[..., T]], Callable[..., T]]

Retry a function after failures, waiting base_delay * 2**attempt.

Source code in paperscraper/utils.py
def retry_with_exponential_backoff(
    *,
    max_attempts: int = 3,
    retry_if: Callable[[T], bool] = lambda result: False,
    exceptions: Tuple[Type[BaseException], ...] = (),
    base_delay: float = 1,
) -> Callable[[Callable[..., T]], Callable[..., T]]:
    """Retry a function after failures, waiting ``base_delay * 2**attempt``."""

    def decorator(func: Callable[..., T]) -> Callable[..., T]:
        @wraps(func)
        def wrapper(*args, **kwargs) -> T:
            for attempt in range(max_attempts):
                try:
                    result = func(*args, **kwargs)
                except exceptions:
                    if attempt == max_attempts - 1:
                        raise
                else:
                    if not retry_if(result) or attempt == max_attempts - 1:
                        return result
                if base_delay:
                    time.sleep(base_delay * 2**attempt)
            raise RuntimeError("unreachable")

        return wrapper

    return decorator

get_scholar_papers(title: str, fields: List = ['title', 'authors', 'year', 'abstract', 'journal', 'citations'], backend: Literal['auto', 'scholarly', 'searchapi'] = 'auto', *, api_key: Optional[str] = None, search_api_kwargs: Optional[dict] = None) -> pd.DataFrame

Performs Google Scholar API request of a given title and returns list of papers with fields as desired.

Parameters:

Name Type Description Default
title str

Google Scholar search query.

required
fields List

List of strings with fields to keep in output.

['title', 'authors', 'year', 'abstract', 'journal', 'citations']
backend Literal['auto', 'scholarly', 'searchapi']

Scholar backend. auto uses SearchApi when configured.

'auto'
api_key Optional[str]

Explicit SearchApi key.

None
search_api_kwargs Optional[dict]

SearchApi-specific keyword arguments.

None

Returns:

Type Description
DataFrame

pd.DataFrame. One paper per row.

Source code in paperscraper/scholar/scholar.py
def get_scholar_papers(
    title: str,
    fields: List = ["title", "authors", "year", "abstract", "journal", "citations"],
    backend: Literal["auto", "scholarly", "searchapi"] = "auto",
    *,
    api_key: Optional[str] = None,
    search_api_kwargs: Optional[dict] = None,
) -> pd.DataFrame:
    """
    Performs Google Scholar API request of a given title and returns list of papers with
    fields as desired.

    Args:
        title: Google Scholar search query.
        fields: List of strings with fields to keep in output.
        backend: Scholar backend. ``auto`` uses SearchApi when configured.
        api_key: Explicit SearchApi key.
        search_api_kwargs: SearchApi-specific keyword arguments.

    Returns:
        pd.DataFrame. One paper per row.

    """
    if not isinstance(title, str):
        raise TypeError(f"Pass str not {type(title)}")

    if re.search(r"\b(?:AND|OR)\b", title):
        logger.info(
            "NOTE: Scholar API cannot be used with Boolean logic in keywords."
            " Query should be a single string to be entered in the Scholar search field."
        )

    resolved_backend = _resolve_backend(
        backend, api_key, (("searchapi", SEARCH_API_KEY),), "scholarly"
    )
    if resolved_backend == "searchapi":
        return get_scholar_papers_searchapi(
            title,
            fields,
            api_key,
            search_api_kwargs=search_api_kwargs,
        )
    if resolved_backend != "scholarly":
        raise ValueError(f"Unknown backend: {backend}")
    if api_key is not None:
        raise ValueError("api_key is not supported by backend='scholarly'")

    matches = scholarly.search_pubs(title)

    processed = []
    for paper in matches:
        # Extracts title, author, year, journal, abstract
        entry = {
            scholar_field_mapper.get(key, key): process_fields.get(
                scholar_field_mapper.get(key, key), lambda x: x
            )(value)
            for key, value in paper["bib"].items()
            if scholar_field_mapper.get(key, key) in fields
        }

        entry["citations"] = paper["num_citations"]
        processed.append(entry)

    return pd.DataFrame(processed)

get_scholar_papers_searchapi(title: str, fields: List = ['title', 'authors', 'year', 'abstract', 'journal', 'citations'], api_key: Optional[str] = None, search_api_kwargs: Optional[dict] = None) -> pd.DataFrame

Retrieve Google Scholar paper metadata through SearchApi.

Parameters:

Name Type Description Default
title str

Google Scholar search query.

required
fields List

List of strings with fields to keep in output.

['title', 'authors', 'year', 'abstract', 'journal', 'citations']
api_key Optional[str]

Explicit SearchApi key.

None
search_api_kwargs Optional[dict]

Supports top_k, num_enrich and max_author_requests.

None

Returns:

Type Description
DataFrame

pd.DataFrame. One paper per row.

Source code in paperscraper/scholar/scholar.py
def get_scholar_papers_searchapi(
    title: str,
    fields: List = ["title", "authors", "year", "abstract", "journal", "citations"],
    api_key: Optional[str] = None,
    search_api_kwargs: Optional[dict] = None,
) -> pd.DataFrame:
    """
    Retrieve Google Scholar paper metadata through SearchApi.

    Args:
        title: Google Scholar search query.
        fields: List of strings with fields to keep in output.
        api_key: Explicit SearchApi key.
        search_api_kwargs: Supports ``top_k``, ``num_enrich`` and
            ``max_author_requests``.

    Returns:
        pd.DataFrame. One paper per row.
    """
    resolved_kwargs = _resolve_search_api_kwargs(search_api_kwargs)
    exact_title = title[1:-1] if re.fullmatch(r'"[^"]+"', title) else None

    @retry_with_exponential_backoff(
        retry_if=lambda papers: exact_title is not None and not papers,
        exceptions=(requests.exceptions.RequestException,),
    )
    def search() -> list[dict]:
        response = search_api_requests_get(
            api_key=api_key,
            params={
                "engine": "google_scholar",
                "q": f"allintitle: {exact_title}" if exact_title else title,
                "hl": "en",
                "num": 20 if exact_title else resolved_kwargs["top_k"],
            },
        )
        papers = response.json().get("organic_results", [])
        if exact_title:
            normalized_title = _normalize_searchapi_title(exact_title)
            papers = [
                paper
                for paper in papers
                if _normalize_searchapi_title(paper.get("title", ""))
                == normalized_title
            ]
        return papers

    papers = search()

    processed = []
    for index, paper in enumerate(papers[: resolved_kwargs["top_k"]]):
        # Search result snippets are not abstracts; only the citation view exposes one.
        citation = (
            get_searchapi_scholar_citation(
                paper,
                api_key,
                search_api_kwargs=resolved_kwargs,
            )
            if index < resolved_kwargs["num_enrich"]
            else {}
        )
        entry = _parse_searchapi_scholar_result(paper, citation)
        if "citations" in fields and entry["citations"] < 0:
            from ..citations.citations import get_citations_from_title_searchapi

            entry["citations"] = get_citations_from_title_searchapi(
                paper["title"], api_key
            )
        processed.append({key: value for key, value in entry.items() if key in fields})

    return pd.DataFrame(processed, columns=fields)

get_scholar_author_papers(author: str, max_results: int = 30, *, author_id: Optional[str] = None, api_key: Optional[str] = None, full_info: bool = False) -> pd.DataFrame

Return papers by a researcher from Google Scholar through SearchApi.

An exact Google Scholar profile is preferred. If none exists, results from an author:"name" Scholar query are returned and may mix namesakes. The default limit is 30 papers; explicit integer limits are respected. full_info=True requires a profile and consumes one extra SearchApi request per paper.

Parameters:

Name Type Description Default
author str

Researcher name.

required
max_results int

Maximum papers to return. Defaults to 30.

30
author_id Optional[str]

Optional Google Scholar author ID for exact identification.

None
api_key Optional[str]

Explicit SearchApi key.

None
full_info bool

Add journal, date, volume, issue, pages, publisher, and description from each paper's citation detail.

False

Returns:

Type Description
DataFrame

pd.DataFrame. One paper per row.

Source code in paperscraper/scholar/scholar.py
def get_scholar_author_papers(
    author: str,
    max_results: int = 30,
    *,
    author_id: Optional[str] = None,
    api_key: Optional[str] = None,
    full_info: bool = False,
) -> pd.DataFrame:
    """Return papers by a researcher from Google Scholar through SearchApi.

    An exact Google Scholar profile is preferred. If none exists, results from
    an ``author:\"name\"`` Scholar query are returned and may mix namesakes.
    The default limit is 30 papers; explicit integer limits are respected.
    ``full_info=True`` requires a profile and consumes one extra SearchApi
    request per paper.

    Args:
        author: Researcher name.
        max_results: Maximum papers to return. Defaults to 30.
        author_id: Optional Google Scholar author ID for exact identification.
        api_key: Explicit SearchApi key.
        full_info: Add journal, date, volume, issue, pages, publisher, and
            description from each paper's citation detail.

    Returns:
        pd.DataFrame. One paper per row.
    """
    if not isinstance(author, str):
        raise TypeError(f"Pass str not {type(author)}")
    author = author.strip()
    if not author:
        raise ValueError("author must not be empty")
    if not isinstance(max_results, int) or isinstance(max_results, bool):
        raise TypeError(f"Pass int not {type(max_results)}")
    if max_results < 0:
        raise ValueError("max_results must be non-negative")
    if author_id is not None and (not isinstance(author_id, str) or not author_id):
        raise ValueError("author_id must be a non-empty string")
    if not isinstance(full_info, bool):
        raise TypeError(f"Pass bool not {type(full_info)}")

    fields = _AUTHOR_PAPER_FIELDS + (_AUTHOR_DETAIL_FIELDS if full_info else [])
    if max_results == 0:
        return pd.DataFrame(columns=fields)

    search = None
    if author_id is None:
        search_params = {
            "engine": "google_scholar",
            "q": f'author:"{author}"',
            "hl": "en",
            "num": _AUTHOR_PAGE_SIZE,
        }
        search = _get_searchapi_data(search_params, api_key)
        exact_profiles = [
            profile
            for profile in search.get("profiles", [])
            if _normalize_searchapi_author(profile.get("name", ""))
            == _normalize_searchapi_author(author)
            and profile.get("author_id")
        ]
        if len(exact_profiles) > 1:
            candidates = ", ".join(
                f"{profile['author_id']} ({profile.get('affiliations', 'unknown')})"
                for profile in exact_profiles
            )
            raise RuntimeError(
                f"Multiple exact Google Scholar profiles found for {author!r}: "
                f"{candidates}. Pass author_id to disambiguate."
            )
        author_id = exact_profiles[0]["author_id"] if exact_profiles else None

    if author_id is not None:
        profile_params = {
            "engine": "google_scholar_author",
            "author_id": author_id,
            "hl": "en",
        }
        profile = _get_searchapi_data(profile_params, api_key)
        if not profile.get("author"):
            raise RuntimeError(
                f"SearchApi returned no author profile for {author_id!r}."
            )
        articles = _get_searchapi_pages(
            profile_params,
            "articles",
            "citation_id",
            max_results,
            api_key,
            first_page=profile,
        )
        return pd.DataFrame(
            [
                _parse_searchapi_author_article(
                    article,
                    api_key,
                    full_info=full_info,
                )
                for article in articles
            ],
            columns=fields,
        )

    if full_info:
        raise RuntimeError(
            f"full_info requires a Google Scholar profile for {author!r}."
        )
    logger.warning(
        "No exact Google Scholar profile found for %r; returning name-query "
        "results that may mix namesakes.",
        author,
    )
    papers = _get_searchapi_pages(
        search_params,
        "organic_results",
        "data_cid",
        max_results,
        api_key,
        first_page=search,
    )
    from ..citations.citations import get_citations_from_title_searchapi

    processed = []
    for paper in papers:
        entry = _parse_searchapi_scholar_result(paper, {})
        if entry["citations"] < 0:
            entry["citations"] = get_citations_from_title_searchapi(
                paper["title"], api_key
            )
        entry["publication"] = _normalize_searchapi_text(paper.get("publication", ""))
        processed.append({field: entry[field] for field in fields})
    return pd.DataFrame(processed, columns=fields)

get_searchapi_scholar_citation(paper: dict, api_key: Optional[str] = None, search_api_kwargs: Optional[dict] = None) -> dict

Retrieve citation details for a SearchApi Scholar result when available.

Parameters:

Name Type Description Default
paper dict

SearchApi google_scholar organic result.

required
api_key Optional[str]

Explicit SearchApi key.

None
search_api_kwargs Optional[dict]

Supports max_author_requests.

None

Returns:

Type Description
dict

SearchApi google_scholar_author citation dict. Common entries are

dict

title, link, resources, description, authors,

dict

publication_date, journal, volume, issue, pages,

dict

publisher, cited_by, cites_histogram, and

dict

scholar_articles. Returns an empty dict if no exact title match is found.

Source code in paperscraper/scholar/scholar.py
def get_searchapi_scholar_citation(
    paper: dict,
    api_key: Optional[str] = None,
    search_api_kwargs: Optional[dict] = None,
) -> dict:
    """
    Retrieve citation details for a SearchApi Scholar result when available.

    Args:
        paper: SearchApi ``google_scholar`` organic result.
        api_key: Explicit SearchApi key.
        search_api_kwargs: Supports ``max_author_requests``.

    Returns:
        SearchApi ``google_scholar_author`` citation dict. Common entries are
        ``title``, ``link``, ``resources``, ``description``, ``authors``,
        ``publication_date``, ``journal``, ``volume``, ``issue``, ``pages``,
        ``publisher``, ``cited_by``, ``cites_histogram``, and
        ``scholar_articles``. Returns an empty dict if no exact title match is found.
    """
    resolved_kwargs = _resolve_search_api_kwargs(search_api_kwargs)
    title = paper.get("title", "")
    citations = []
    remaining_requests = max(0, resolved_kwargs["max_author_requests"])
    authors = [author for author in paper.get("authors", [])[:3] if author.get("id")]
    for index, author in enumerate(authors):
        if remaining_requests < 2:
            break
        author_id = author.get("id")
        author_count = len(authors) - index
        reserved_requests = 2 * (author_count - 1) + 1
        page_requests = max(1, remaining_requests - reserved_requests)
        citation_id, requests_used = _find_searchapi_citation_id(
            title, author_id, api_key, page_requests
        )
        remaining_requests -= requests_used
        if not citation_id:
            continue
        if remaining_requests < 1:
            break
        remaining_requests -= 1
        try:
            citation = (
                search_api_requests_get(
                    api_key=api_key,
                    params={
                        "engine": "google_scholar_author",
                        "view_op": "view_citation",
                        "citation_id": citation_id,
                        "hl": "en",
                    },
                )
                .json()
                .get("citation", {})
            )
        except requests.exceptions.RequestException:
            continue
        citations.append(citation)
    return max(
        citations,
        key=lambda citation: int(citation.get("cited_by", {}).get("total") or -1),
        default={},
    )

get_and_dump_scholar_papers(title: str, output_filepath: str, fields: List = ['title', 'authors', 'year', 'abstract', 'journal', 'citations'], backend: Literal['auto', 'scholarly', 'searchapi'] = 'auto', *, api_key: Optional[str] = None, search_api_kwargs: Optional[dict] = None) -> None

Combines get_scholar_papers and dump_papers.

Parameters:

Name Type Description Default
title str

Paper to search for on Google Scholar.

required
output_filepath str

Path where the dump will be saved.

required
fields List

List of strings with fields to keep in output.

['title', 'authors', 'year', 'abstract', 'journal', 'citations']
backend Literal['auto', 'scholarly', 'searchapi']

Scholar backend.

'auto'
api_key Optional[str]

Explicit SearchApi key.

None
search_api_kwargs Optional[dict]

SearchApi-specific keyword arguments.

None
Source code in paperscraper/scholar/scholar.py
def get_and_dump_scholar_papers(
    title: str,
    output_filepath: str,
    fields: List = ["title", "authors", "year", "abstract", "journal", "citations"],
    backend: Literal["auto", "scholarly", "searchapi"] = "auto",
    *,
    api_key: Optional[str] = None,
    search_api_kwargs: Optional[dict] = None,
) -> None:
    """
    Combines get_scholar_papers and dump_papers.

    Args:
        title: Paper to search for on Google Scholar.
        output_filepath: Path where the dump will be saved.
        fields: List of strings with fields to keep in output.
        backend: Scholar backend.
        api_key: Explicit SearchApi key.
        search_api_kwargs: SearchApi-specific keyword arguments.
    """
    papers = get_scholar_papers(
        title,
        fields,
        backend,
        api_key=api_key,
        search_api_kwargs=search_api_kwargs,
    )
    dump_papers(papers, output_filepath)

scholar

get_scholar_papers(title: str, fields: List = ['title', 'authors', 'year', 'abstract', 'journal', 'citations'], backend: Literal['auto', 'scholarly', 'searchapi'] = 'auto', *, api_key: Optional[str] = None, search_api_kwargs: Optional[dict] = None) -> pd.DataFrame

Performs Google Scholar API request of a given title and returns list of papers with fields as desired.

Parameters:

Name Type Description Default
title str

Google Scholar search query.

required
fields List

List of strings with fields to keep in output.

['title', 'authors', 'year', 'abstract', 'journal', 'citations']
backend Literal['auto', 'scholarly', 'searchapi']

Scholar backend. auto uses SearchApi when configured.

'auto'
api_key Optional[str]

Explicit SearchApi key.

None
search_api_kwargs Optional[dict]

SearchApi-specific keyword arguments.

None

Returns:

Type Description
DataFrame

pd.DataFrame. One paper per row.

Source code in paperscraper/scholar/scholar.py
def get_scholar_papers(
    title: str,
    fields: List = ["title", "authors", "year", "abstract", "journal", "citations"],
    backend: Literal["auto", "scholarly", "searchapi"] = "auto",
    *,
    api_key: Optional[str] = None,
    search_api_kwargs: Optional[dict] = None,
) -> pd.DataFrame:
    """
    Performs Google Scholar API request of a given title and returns list of papers with
    fields as desired.

    Args:
        title: Google Scholar search query.
        fields: List of strings with fields to keep in output.
        backend: Scholar backend. ``auto`` uses SearchApi when configured.
        api_key: Explicit SearchApi key.
        search_api_kwargs: SearchApi-specific keyword arguments.

    Returns:
        pd.DataFrame. One paper per row.

    """
    if not isinstance(title, str):
        raise TypeError(f"Pass str not {type(title)}")

    if re.search(r"\b(?:AND|OR)\b", title):
        logger.info(
            "NOTE: Scholar API cannot be used with Boolean logic in keywords."
            " Query should be a single string to be entered in the Scholar search field."
        )

    resolved_backend = _resolve_backend(
        backend, api_key, (("searchapi", SEARCH_API_KEY),), "scholarly"
    )
    if resolved_backend == "searchapi":
        return get_scholar_papers_searchapi(
            title,
            fields,
            api_key,
            search_api_kwargs=search_api_kwargs,
        )
    if resolved_backend != "scholarly":
        raise ValueError(f"Unknown backend: {backend}")
    if api_key is not None:
        raise ValueError("api_key is not supported by backend='scholarly'")

    matches = scholarly.search_pubs(title)

    processed = []
    for paper in matches:
        # Extracts title, author, year, journal, abstract
        entry = {
            scholar_field_mapper.get(key, key): process_fields.get(
                scholar_field_mapper.get(key, key), lambda x: x
            )(value)
            for key, value in paper["bib"].items()
            if scholar_field_mapper.get(key, key) in fields
        }

        entry["citations"] = paper["num_citations"]
        processed.append(entry)

    return pd.DataFrame(processed)

get_scholar_papers_searchapi(title: str, fields: List = ['title', 'authors', 'year', 'abstract', 'journal', 'citations'], api_key: Optional[str] = None, search_api_kwargs: Optional[dict] = None) -> pd.DataFrame

Retrieve Google Scholar paper metadata through SearchApi.

Parameters:

Name Type Description Default
title str

Google Scholar search query.

required
fields List

List of strings with fields to keep in output.

['title', 'authors', 'year', 'abstract', 'journal', 'citations']
api_key Optional[str]

Explicit SearchApi key.

None
search_api_kwargs Optional[dict]

Supports top_k, num_enrich and max_author_requests.

None

Returns:

Type Description
DataFrame

pd.DataFrame. One paper per row.

Source code in paperscraper/scholar/scholar.py
def get_scholar_papers_searchapi(
    title: str,
    fields: List = ["title", "authors", "year", "abstract", "journal", "citations"],
    api_key: Optional[str] = None,
    search_api_kwargs: Optional[dict] = None,
) -> pd.DataFrame:
    """
    Retrieve Google Scholar paper metadata through SearchApi.

    Args:
        title: Google Scholar search query.
        fields: List of strings with fields to keep in output.
        api_key: Explicit SearchApi key.
        search_api_kwargs: Supports ``top_k``, ``num_enrich`` and
            ``max_author_requests``.

    Returns:
        pd.DataFrame. One paper per row.
    """
    resolved_kwargs = _resolve_search_api_kwargs(search_api_kwargs)
    exact_title = title[1:-1] if re.fullmatch(r'"[^"]+"', title) else None

    @retry_with_exponential_backoff(
        retry_if=lambda papers: exact_title is not None and not papers,
        exceptions=(requests.exceptions.RequestException,),
    )
    def search() -> list[dict]:
        response = search_api_requests_get(
            api_key=api_key,
            params={
                "engine": "google_scholar",
                "q": f"allintitle: {exact_title}" if exact_title else title,
                "hl": "en",
                "num": 20 if exact_title else resolved_kwargs["top_k"],
            },
        )
        papers = response.json().get("organic_results", [])
        if exact_title:
            normalized_title = _normalize_searchapi_title(exact_title)
            papers = [
                paper
                for paper in papers
                if _normalize_searchapi_title(paper.get("title", ""))
                == normalized_title
            ]
        return papers

    papers = search()

    processed = []
    for index, paper in enumerate(papers[: resolved_kwargs["top_k"]]):
        # Search result snippets are not abstracts; only the citation view exposes one.
        citation = (
            get_searchapi_scholar_citation(
                paper,
                api_key,
                search_api_kwargs=resolved_kwargs,
            )
            if index < resolved_kwargs["num_enrich"]
            else {}
        )
        entry = _parse_searchapi_scholar_result(paper, citation)
        if "citations" in fields and entry["citations"] < 0:
            from ..citations.citations import get_citations_from_title_searchapi

            entry["citations"] = get_citations_from_title_searchapi(
                paper["title"], api_key
            )
        processed.append({key: value for key, value in entry.items() if key in fields})

    return pd.DataFrame(processed, columns=fields)

get_scholar_author_papers(author: str, max_results: int = 30, *, author_id: Optional[str] = None, api_key: Optional[str] = None, full_info: bool = False) -> pd.DataFrame

Return papers by a researcher from Google Scholar through SearchApi.

An exact Google Scholar profile is preferred. If none exists, results from an author:"name" Scholar query are returned and may mix namesakes. The default limit is 30 papers; explicit integer limits are respected. full_info=True requires a profile and consumes one extra SearchApi request per paper.

Parameters:

Name Type Description Default
author str

Researcher name.

required
max_results int

Maximum papers to return. Defaults to 30.

30
author_id Optional[str]

Optional Google Scholar author ID for exact identification.

None
api_key Optional[str]

Explicit SearchApi key.

None
full_info bool

Add journal, date, volume, issue, pages, publisher, and description from each paper's citation detail.

False

Returns:

Type Description
DataFrame

pd.DataFrame. One paper per row.

Source code in paperscraper/scholar/scholar.py
def get_scholar_author_papers(
    author: str,
    max_results: int = 30,
    *,
    author_id: Optional[str] = None,
    api_key: Optional[str] = None,
    full_info: bool = False,
) -> pd.DataFrame:
    """Return papers by a researcher from Google Scholar through SearchApi.

    An exact Google Scholar profile is preferred. If none exists, results from
    an ``author:\"name\"`` Scholar query are returned and may mix namesakes.
    The default limit is 30 papers; explicit integer limits are respected.
    ``full_info=True`` requires a profile and consumes one extra SearchApi
    request per paper.

    Args:
        author: Researcher name.
        max_results: Maximum papers to return. Defaults to 30.
        author_id: Optional Google Scholar author ID for exact identification.
        api_key: Explicit SearchApi key.
        full_info: Add journal, date, volume, issue, pages, publisher, and
            description from each paper's citation detail.

    Returns:
        pd.DataFrame. One paper per row.
    """
    if not isinstance(author, str):
        raise TypeError(f"Pass str not {type(author)}")
    author = author.strip()
    if not author:
        raise ValueError("author must not be empty")
    if not isinstance(max_results, int) or isinstance(max_results, bool):
        raise TypeError(f"Pass int not {type(max_results)}")
    if max_results < 0:
        raise ValueError("max_results must be non-negative")
    if author_id is not None and (not isinstance(author_id, str) or not author_id):
        raise ValueError("author_id must be a non-empty string")
    if not isinstance(full_info, bool):
        raise TypeError(f"Pass bool not {type(full_info)}")

    fields = _AUTHOR_PAPER_FIELDS + (_AUTHOR_DETAIL_FIELDS if full_info else [])
    if max_results == 0:
        return pd.DataFrame(columns=fields)

    search = None
    if author_id is None:
        search_params = {
            "engine": "google_scholar",
            "q": f'author:"{author}"',
            "hl": "en",
            "num": _AUTHOR_PAGE_SIZE,
        }
        search = _get_searchapi_data(search_params, api_key)
        exact_profiles = [
            profile
            for profile in search.get("profiles", [])
            if _normalize_searchapi_author(profile.get("name", ""))
            == _normalize_searchapi_author(author)
            and profile.get("author_id")
        ]
        if len(exact_profiles) > 1:
            candidates = ", ".join(
                f"{profile['author_id']} ({profile.get('affiliations', 'unknown')})"
                for profile in exact_profiles
            )
            raise RuntimeError(
                f"Multiple exact Google Scholar profiles found for {author!r}: "
                f"{candidates}. Pass author_id to disambiguate."
            )
        author_id = exact_profiles[0]["author_id"] if exact_profiles else None

    if author_id is not None:
        profile_params = {
            "engine": "google_scholar_author",
            "author_id": author_id,
            "hl": "en",
        }
        profile = _get_searchapi_data(profile_params, api_key)
        if not profile.get("author"):
            raise RuntimeError(
                f"SearchApi returned no author profile for {author_id!r}."
            )
        articles = _get_searchapi_pages(
            profile_params,
            "articles",
            "citation_id",
            max_results,
            api_key,
            first_page=profile,
        )
        return pd.DataFrame(
            [
                _parse_searchapi_author_article(
                    article,
                    api_key,
                    full_info=full_info,
                )
                for article in articles
            ],
            columns=fields,
        )

    if full_info:
        raise RuntimeError(
            f"full_info requires a Google Scholar profile for {author!r}."
        )
    logger.warning(
        "No exact Google Scholar profile found for %r; returning name-query "
        "results that may mix namesakes.",
        author,
    )
    papers = _get_searchapi_pages(
        search_params,
        "organic_results",
        "data_cid",
        max_results,
        api_key,
        first_page=search,
    )
    from ..citations.citations import get_citations_from_title_searchapi

    processed = []
    for paper in papers:
        entry = _parse_searchapi_scholar_result(paper, {})
        if entry["citations"] < 0:
            entry["citations"] = get_citations_from_title_searchapi(
                paper["title"], api_key
            )
        entry["publication"] = _normalize_searchapi_text(paper.get("publication", ""))
        processed.append({field: entry[field] for field in fields})
    return pd.DataFrame(processed, columns=fields)

get_searchapi_scholar_citation(paper: dict, api_key: Optional[str] = None, search_api_kwargs: Optional[dict] = None) -> dict

Retrieve citation details for a SearchApi Scholar result when available.

Parameters:

Name Type Description Default
paper dict

SearchApi google_scholar organic result.

required
api_key Optional[str]

Explicit SearchApi key.

None
search_api_kwargs Optional[dict]

Supports max_author_requests.

None

Returns:

Type Description
dict

SearchApi google_scholar_author citation dict. Common entries are

dict

title, link, resources, description, authors,

dict

publication_date, journal, volume, issue, pages,

dict

publisher, cited_by, cites_histogram, and

dict

scholar_articles. Returns an empty dict if no exact title match is found.

Source code in paperscraper/scholar/scholar.py
def get_searchapi_scholar_citation(
    paper: dict,
    api_key: Optional[str] = None,
    search_api_kwargs: Optional[dict] = None,
) -> dict:
    """
    Retrieve citation details for a SearchApi Scholar result when available.

    Args:
        paper: SearchApi ``google_scholar`` organic result.
        api_key: Explicit SearchApi key.
        search_api_kwargs: Supports ``max_author_requests``.

    Returns:
        SearchApi ``google_scholar_author`` citation dict. Common entries are
        ``title``, ``link``, ``resources``, ``description``, ``authors``,
        ``publication_date``, ``journal``, ``volume``, ``issue``, ``pages``,
        ``publisher``, ``cited_by``, ``cites_histogram``, and
        ``scholar_articles``. Returns an empty dict if no exact title match is found.
    """
    resolved_kwargs = _resolve_search_api_kwargs(search_api_kwargs)
    title = paper.get("title", "")
    citations = []
    remaining_requests = max(0, resolved_kwargs["max_author_requests"])
    authors = [author for author in paper.get("authors", [])[:3] if author.get("id")]
    for index, author in enumerate(authors):
        if remaining_requests < 2:
            break
        author_id = author.get("id")
        author_count = len(authors) - index
        reserved_requests = 2 * (author_count - 1) + 1
        page_requests = max(1, remaining_requests - reserved_requests)
        citation_id, requests_used = _find_searchapi_citation_id(
            title, author_id, api_key, page_requests
        )
        remaining_requests -= requests_used
        if not citation_id:
            continue
        if remaining_requests < 1:
            break
        remaining_requests -= 1
        try:
            citation = (
                search_api_requests_get(
                    api_key=api_key,
                    params={
                        "engine": "google_scholar_author",
                        "view_op": "view_citation",
                        "citation_id": citation_id,
                        "hl": "en",
                    },
                )
                .json()
                .get("citation", {})
            )
        except requests.exceptions.RequestException:
            continue
        citations.append(citation)
    return max(
        citations,
        key=lambda citation: int(citation.get("cited_by", {}).get("total") or -1),
        default={},
    )

get_and_dump_scholar_papers(title: str, output_filepath: str, fields: List = ['title', 'authors', 'year', 'abstract', 'journal', 'citations'], backend: Literal['auto', 'scholarly', 'searchapi'] = 'auto', *, api_key: Optional[str] = None, search_api_kwargs: Optional[dict] = None) -> None

Combines get_scholar_papers and dump_papers.

Parameters:

Name Type Description Default
title str

Paper to search for on Google Scholar.

required
output_filepath str

Path where the dump will be saved.

required
fields List

List of strings with fields to keep in output.

['title', 'authors', 'year', 'abstract', 'journal', 'citations']
backend Literal['auto', 'scholarly', 'searchapi']

Scholar backend.

'auto'
api_key Optional[str]

Explicit SearchApi key.

None
search_api_kwargs Optional[dict]

SearchApi-specific keyword arguments.

None
Source code in paperscraper/scholar/scholar.py
def get_and_dump_scholar_papers(
    title: str,
    output_filepath: str,
    fields: List = ["title", "authors", "year", "abstract", "journal", "citations"],
    backend: Literal["auto", "scholarly", "searchapi"] = "auto",
    *,
    api_key: Optional[str] = None,
    search_api_kwargs: Optional[dict] = None,
) -> None:
    """
    Combines get_scholar_papers and dump_papers.

    Args:
        title: Paper to search for on Google Scholar.
        output_filepath: Path where the dump will be saved.
        fields: List of strings with fields to keep in output.
        backend: Scholar backend.
        api_key: Explicit SearchApi key.
        search_api_kwargs: SearchApi-specific keyword arguments.
    """
    papers = get_scholar_papers(
        title,
        fields,
        backend,
        api_key=api_key,
        search_api_kwargs=search_api_kwargs,
    )
    dump_papers(papers, output_filepath)