arxiv-mcp-server
Search arXiv, fetch paper metadata, and read full-text content.
사용해야 할까요
품질 및 안전성
도구 정의와 프로토콜 준수에 대한 자동 분석을 기반으로 합니다.
컨텍스트 비용
이는 서버의 도구가 모델의 컨텍스트에 로드될 때마다 소비되는 대략적인 토큰 수입니다. 수치가 높을수록 다른 작업에 사용할 수 있는 주의가 줄어듭니다.
설치
원클릭 설치
`claude_desktop_config.json` 파일에 다음을 추가하세요:
{
"mcpServers": {
"arxiv-mcp-server": {
"command": "bun",
"args": [
"@cyanheads/arxiv-mcp-server"
]
}
}
}실행 가능한 패키지
1.5.3streamable-http원격 엔드포인트
https://arxiv.caseyjhand.com/mcpstreamable-http할 수 있는 일
도구 목록
도구 (4)
🟢arxiv_search(query, category, max_results, sort_by, sort_order, ...)
Search arXiv papers by query with category and sort filters. Returns paper metadata including title, authors, abstract, categories, and links.
입력 스키마
{
"type": "object",
"properties": {
"query": {
"type": "string",
"minLength": 1,
"maxLength": 1000,
"pattern": "^[^\\x00-\\x08\\x0B\\x0C\\x0E-\\x1F]*$",
"description": "Search query. Field prefixes: ti: (title), au: (author — token-based; quote multi-token names like au:\"hinton g\" or pair with a topical clause to disambiguate common surnames), abs: (abstract), cat: (category — a leaf code matches exactly, a bare archive code such as cat:astro-ph matches its whole subtree), co: (comment), jr: (journal ref), all: (all fields). Boolean operators: AND, OR, ANDNOT. Examples: \"au:bengio AND ti:attention\", \"all:transformer AND cat:cs.CL\"."
},
"category": {
"description": "Restrict results to an arXiv category. A leaf code (\"cs.CL\", \"math.AG\") matches exactly. A bare archive code (\"astro-ph\", \"cond-mat\", \"cs\", \"math\") matches the whole archive — its subject classes plus the legacy flat papers filed before the archive was subdivided. Note \"physics\" is the general-physics archive (physics.*), not the wider physics group: astro-ph, cond-mat, hep-*, quant-ph and the rest are separate archive codes. Use arxiv_list_categories to discover subject classes.",
"type": "string"
},
"max_results": {
"default": 10,
"description": "Maximum results to return (1-50). Default 10. Each result includes title, authors, abstract, and metadata — keep low to limit response size.",
"type": "integer",
"minimum": 1,
"maximum": 50
},
"sort_by": {
"default": "relevance",
"description": "Sort criterion. Use \"submitted\" for newest papers, \"relevance\" for best query matches.",
"type": "string",
"enum": [
"relevance",
"submitted",
"updated"
]
},
"sort_order": {
"default": "descending",
"description": "Sort direction. \"descending\" returns newest/most relevant first.",
"type": "string",
"enum": [
"ascending",
"descending"
]
},
"start": {
"default": 0,
"description": "Pagination offset (0-10000). Use with max_results to page through results. E.g., start=10 with max_results=10 returns results 11-20. Matches beyond offset 10000 + max_results are unreachable by paging — carve the search into submitted_from/submitted_to windows and page within each.",
"type": "integer",
"minimum": 0,
"maximum": 10000
},
"submitted_from": {
"description": "Earliest submission date to include, inclusive, as a UTC YYYY-MM-DD date. Omit for no lower bound.",
"type": "string",
"pattern": "^(\\d{4}-\\d{2}-\\d{2})?$"
},
"submitted_to": {
"description": "Latest submission date to include, inclusive, as a UTC YYYY-MM-DD date. Omit for no upper bound. Both bounds are inclusive, so consecutive windows (\"2024-01-01\"..\"2024-01-15\" then \"2024-01-16\"..\"2024-01-31\") cover the matches with no gap; a paper submitted at exactly the midnight seam between two windows appears in both, so de-duplicate collected results by paper id. That is the way to reach matches past the start ceiling: split the date range, then page within each window.",
"type": "string",
"pattern": "^(\\d{4}-\\d{2}-\\d{2})?$"
}
},
"required": [
"query"
],
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false
}출력 스키마
{
"type": "object",
"properties": {
"papers": {
"type": "array",
"items": {
"type": "object",
"properties": {
"id": {
"type": "string",
"description": "arXiv paper ID (e.g., \"2401.12345v1\")."
},
"title": {
"type": "string",
"description": "Paper title."
},
"authors": {
"type": "array",
"items": {
"type": "string"
},
"description": "Author names."
},
"abstract": {
"type": "string",
"description": "Full abstract text."
},
"primary_category": {
"type": "string",
"description": "Primary arXiv category (e.g., \"cs.CL\")."
},
"categories": {
"type": "array",
"items": {
"type": "string"
},
"description": "All arXiv categories assigned to this paper."
},
"published": {
"type": "string",
"description": "Original submission date (ISO 8601)."
},
"updated": {
"type": "string",
"description": "Last update date (ISO 8601)."
},
"comment": {
"description": "Author comment (e.g., page count, conference).",
"type": "string"
},
"journal_ref": {
"description": "Journal reference if published.",
"type": "string"
},
"doi": {
"description": "DOI if available.",
"type": "string"
},
"pdf_url": {
"type": "string",
"description": "Direct PDF download URL."
},
"abstract_url": {
"type": "string",
"description": "arXiv abstract page URL."
}
},
"required": [
"id",
"title",
"authors",
"abstract",
"primary_category",
"categories",
"published",
"updated",
"pdf_url",
"abstract_url"
],
"additionalProperties": false,
"description": "arXiv paper metadata — identifier, title, authors, abstract, categories, and links."
},
"description": "Matching papers with full metadata."
},
"effectiveQuery": {
"type": "string",
"description": "The query as actually searched, carrying every filter applied — the category subtree and submitted-date window folded into arXiv syntax alongside the supplied terms. Replaying it as `query` with no other filters reproduces this exact result set."
},
"totalFound": {
"type": "number",
"description": "Total matching papers reported by arXiv (before pagination)."
},
"pageStart": {
"type": "number",
"description": "Pagination offset of this result page."
},
"truncated": {
"description": "True when more matching papers exist beyond this page (totalFound > start + shown).",
"type": "boolean"
},
"shown": {
"description": "Papers returned on this page.",
"type": "number"
},
"cap": {
"description": "The max_results limit applied to this page.",
"type": "number"
},
"notice": {
"description": "Recovery guidance when results are empty or paging overshot. Absent on successful pages.",
"type": "string"
},
"error": {
"description": "Present when the call failed. Absent on success.",
"type": "object",
"properties": {
"code": {
"type": "integer",
"minimum": -9007199254740991,
"maximum": 9007199254740991,
"description": "JSON-RPC error code for this failure."
},
"message": {
"type": "string",
"description": "Human-readable description of what went wrong."
},
"data": {
"type": "object",
"properties": {
"reason": {
"type": "string",
"description": "Machine-readable failure mode. Declared by this tool: `unknown_category`: Provided category code is not part of the arXiv taxonomy. `rate_limited`: arXiv has throttled requests (HTTP 429 or \"Rate exceeded.\" body). `invalid_request`: arXiv rejected the request (HTTP 4xx other than 429), typically malformed query syntax. `unsupported_query_syntax`: Query translates to a mirror FTS5 expression the search engine cannot parse, typically two operands juxtaposed across a parenthesized group without an explicit operator. `invalid_date_range`: submitted_from or submitted_to is not a real UTC calendar date, or the window starts after it ends. Other values are possible when a failure originates below the handler.",
"examples": [
"unknown_category",
"rate_limited",
"invalid_request",
"unsupported_query_syntax",
"invalid_date_range"
]
},
"recovery": {
"description": "Actionable next step for the caller.",
"type": "object",
"properties": {
"hint": {
"type": "string"
}
},
"required": [
"hint"
],
"additionalProperties": {}
},
"retryable": {
"description": "Whether retrying may succeed.",
"type": "boolean"
}
},
"additionalProperties": {}
}
},
"required": [
"code",
"message"
],
"additionalProperties": {}
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"anyOf": [
{
"not": {
"required": [
"error"
]
},
"required": [
"papers",
"effectiveQuery",
"totalFound",
"pageStart"
]
},
{
"required": [
"error"
]
}
]
}🟢arxiv_get_metadata(paper_ids)
Get full metadata for one or more arXiv papers by ID. Use when you have known IDs from citations, prior search results, or memory.
입력 스키마
{
"type": "object",
"properties": {
"paper_ids": {
"anyOf": [
{
"type": "string",
"minLength": 1,
"description": "Single arXiv paper ID (e.g., \"2401.12345\" or \"2401.12345v2\")."
},
{
"minItems": 1,
"maxItems": 10,
"type": "array",
"items": {
"type": "string",
"minLength": 1
},
"description": "Array of up to 10 arXiv paper IDs for batch lookup."
}
],
"description": "arXiv paper ID or array of up to 10 IDs. Format: \"2401.12345\" or \"2401.12345v2\" (with version). Also accepts legacy IDs like \"hep-th/9901001\"."
}
},
"required": [
"paper_ids"
],
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false
}출력 스키마
{
"type": "object",
"properties": {
"papers": {
"type": "array",
"items": {
"type": "object",
"properties": {
"id": {
"type": "string",
"description": "arXiv paper ID (e.g., \"2401.12345v1\")."
},
"title": {
"type": "string",
"description": "Paper title."
},
"authors": {
"type": "array",
"items": {
"type": "string"
},
"description": "Author names."
},
"abstract": {
"type": "string",
"description": "Full abstract text."
},
"primary_category": {
"type": "string",
"description": "Primary arXiv category (e.g., \"cs.CL\")."
},
"categories": {
"type": "array",
"items": {
"type": "string"
},
"description": "All arXiv categories assigned to this paper."
},
"published": {
"type": "string",
"description": "Original submission date (ISO 8601)."
},
"updated": {
"type": "string",
"description": "Last update date (ISO 8601)."
},
"comment": {
"description": "Author comment (e.g., page count, conference).",
"type": "string"
},
"journal_ref": {
"description": "Journal reference if published.",
"type": "string"
},
"doi": {
"description": "DOI if available.",
"type": "string"
},
"pdf_url": {
"type": "string",
"description": "Direct PDF download URL."
},
"abstract_url": {
"type": "string",
"description": "arXiv abstract page URL."
}
},
"required": [
"id",
"title",
"authors",
"abstract",
"primary_category",
"categories",
"published",
"updated",
"pdf_url",
"abstract_url"
],
"additionalProperties": false,
"description": "arXiv paper metadata — identifier, title, authors, abstract, categories, and links."
},
"description": "Papers found. May be fewer than requested if some IDs are invalid."
},
"totalSucceeded": {
"type": "integer",
"minimum": 0,
"maximum": 9007199254740991,
"description": "Number of successful items in 'papers'"
},
"not_found": {
"description": "Per-input explanations for inputs that could not be returned. Absent when nothing failed.",
"type": "array",
"items": {
"type": "object",
"properties": {
"id": {
"type": "string",
"description": "arXiv ID that returned no data."
},
"reason": {
"type": "string",
"enum": [
"not_in_arxiv",
"version_not_in_mirror"
],
"description": "Why the paper ID could not be returned."
},
"detail": {
"description": "Additional human-readable context, when available",
"type": "string"
}
},
"required": [
"id",
"reason"
],
"additionalProperties": false,
"description": "A requested ID that could not be returned, with the reason it was missed."
}
},
"error": {
"description": "Present when the call failed. Absent on success.",
"type": "object",
"properties": {
"code": {
"type": "integer",
"minimum": -9007199254740991,
"maximum": 9007199254740991,
"description": "JSON-RPC error code for this failure."
},
"message": {
"type": "string",
"description": "Human-readable description of what went wrong."
},
"data": {
"type": "object",
"properties": {
"reason": {
"type": "string",
"description": "Machine-readable failure mode. Declared by this tool: `no_match`: None of the requested IDs returned data from arXiv. `version_unavailable`: Every requested ID pinned a version the local mirror does not hold, and live arXiv fallback is disabled. `rate_limited`: arXiv has throttled requests (HTTP 429 or \"Rate exceeded.\" body). `invalid_request`: arXiv rejected the request (HTTP 4xx other than 429), e.g. malformed ID syntax. Other values are possible when a failure originates below the handler.",
"examples": [
"no_match",
"version_unavailable",
"rate_limited",
"invalid_request"
]
},
"recovery": {
"description": "Actionable next step for the caller.",
"type": "object",
"properties": {
"hint": {
"type": "string"
}
},
"required": [
"hint"
],
"additionalProperties": {}
},
"retryable": {
"description": "Whether retrying may succeed.",
"type": "boolean"
}
},
"additionalProperties": {}
}
},
"required": [
"code",
"message"
],
"additionalProperties": {}
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"anyOf": [
{
"not": {
"required": [
"error"
]
},
"required": [
"papers",
"totalSucceeded"
]
},
{
"required": [
"error"
]
}
]
}🟢arxiv_read_paper(paper_id, max_characters, start)
Fetch the full text of an arXiv paper. Tries arxiv.org/html first, falls back to ar5iv.labs.arxiv.org, and falls back again to text extracted from the PDF when neither has an HTML render — check the source field to know which one answered. Page through long papers with start and max_characters, or pass max_characters null to get the entire body in one call.
입력 스키마
{
"type": "object",
"properties": {
"paper_id": {
"type": "string",
"minLength": 1,
"description": "arXiv paper ID (e.g., \"2401.12345\" or \"2401.12345v2\")."
},
"max_characters": {
"default": 100000,
"description": "Maximum characters of paper body to return, counted after boilerplate stripping. Defaults to 100,000; pass null to return the entire body in one call. Whole-paper reads can exceed a client tool-result size cap — math-heavy bodies run 300KB-1MB+ — so prefer the default plus start-based paging unless the full text is needed. When truncated, a notice and the total character count are included.",
"anyOf": [
{
"type": "integer",
"minimum": 1,
"maximum": 9007199254740991
},
{
"type": "null"
}
]
},
"start": {
"default": 0,
"description": "Character offset into the cleaned body to begin reading from. Defaults to 0. Use with max_characters to page through long papers — e.g., start=100000 with max_characters=100000 returns chars 100,000–199,999. The total length is reported as body_characters in the response.",
"type": "integer",
"minimum": 0,
"maximum": 9007199254740991
}
},
"required": [
"paper_id"
],
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false
}출력 스키마
{
"type": "object",
"properties": {
"paper_id": {
"type": "string",
"description": "arXiv paper ID."
},
"title": {
"type": "string",
"description": "Paper title (from metadata, not parsed from HTML)."
},
"content": {
"type": "string",
"description": "Paper body for the requested slice — cleaned HTML when source is arxiv_html or ar5iv, plain text when source is pdf_text. Empty when start is past body_characters."
},
"source": {
"type": "string",
"enum": [
"arxiv_html",
"ar5iv",
"pdf_text"
],
"description": "Which upstream artifact the body was read from. arxiv_html and ar5iv are HTML renders; pdf_text is text extracted from the PDF, where prose is reliable but math, tables, and heading structure are flattened."
},
"truncated": {
"type": "boolean",
"description": "True when more body content exists past this slice (start + content.length < body_characters)."
},
"start": {
"type": "number",
"description": "Character offset of the first character in content within the cleaned body."
},
"total_characters": {
"type": "number",
"description": "Character count of the body before cleaning — the unprocessed HTML body for arxiv_html and ar5iv, and equal to body_characters for pdf_text, which needs no cleaning."
},
"body_characters": {
"type": "number",
"description": "Character count of the full cleaned body. Use with start and max_characters to page. Typically 3-4× smaller than total_characters for math-heavy HTML papers."
},
"pdf_url": {
"type": "string",
"description": "Direct PDF download URL."
},
"abstract_url": {
"type": "string",
"description": "arXiv abstract page URL for attribution."
},
"error": {
"description": "Present when the call failed. Absent on success.",
"type": "object",
"properties": {
"code": {
"type": "integer",
"minimum": -9007199254740991,
"maximum": 9007199254740991,
"description": "JSON-RPC error code for this failure."
},
"message": {
"type": "string",
"description": "Human-readable description of what went wrong."
},
"data": {
"type": "object",
"properties": {
"reason": {
"type": "string",
"description": "Machine-readable failure mode. Declared by this tool: `no_match`: Paper ID is not present in the arXiv index. `content_unavailable`: Paper exists but neither arxiv.org/html nor ar5iv has an HTML rendering and arXiv served no PDF either. `pdf_extraction_failed`: Paper has no HTML rendering and its PDF carries no text layer — an image-only or scanned submission. `version_unavailable`: A version-pinned paper_id was requested, arXiv is unreachable, and the local mirror holds only a different version — per-version reads require the live API. `rate_limited`: arXiv has throttled requests (HTTP 429 or \"Rate exceeded.\" body). `invalid_request`: arXiv rejected the metadata lookup (HTTP 4xx other than 429), e.g. malformed ID syntax. Other values are possible when a failure originates below the handler.",
"examples": [
"no_match",
"content_unavailable",
"pdf_extraction_failed",
"version_unavailable",
"rate_limited",
"invalid_request"
]
},
"recovery": {
"description": "Actionable next step for the caller.",
"type": "object",
"properties": {
"hint": {
"type": "string"
}
},
"required": [
"hint"
],
"additionalProperties": {}
},
"retryable": {
"description": "Whether retrying may succeed.",
"type": "boolean"
}
},
"additionalProperties": {}
}
},
"required": [
"code",
"message"
],
"additionalProperties": {}
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"anyOf": [
{
"not": {
"required": [
"error"
]
},
"required": [
"paper_id",
"title",
"content",
"source",
"truncated",
"start",
"total_characters",
"body_characters",
"pdf_url",
"abstract_url"
]
},
{
"required": [
"error"
]
}
]
}🟢arxiv_list_categories(group)
List arXiv category codes and names. Useful for discovering valid category filters for arxiv_search. Lists subject classes only; arxiv_search also accepts a bare archive code (the part before the dot, e.g. "astro-ph" or "cs") to search a whole archive at once.
입력 스키마
{
"type": "object",
"properties": {
"group": {
"description": "Filter by top-level group (e.g., \"cs\", \"math\", \"physics\"). Returns all categories if omitted.",
"type": "string",
"enum": [
"cs",
"econ",
"eess",
"math",
"physics",
"q-bio",
"q-fin",
"stat"
]
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false
}출력 스키마
{
"type": "object",
"properties": {
"categories": {
"type": "array",
"items": {
"type": "object",
"properties": {
"code": {
"type": "string",
"description": "Category code (e.g., \"cs.AI\")."
},
"name": {
"type": "string",
"description": "Full name (e.g., \"Artificial Intelligence\")."
},
"group": {
"type": "string",
"description": "Top-level group (e.g., \"cs\")."
}
},
"required": [
"code",
"name",
"group"
],
"additionalProperties": false,
"description": "arXiv category — subject code, full name, and top-level group."
},
"description": "arXiv categories matching the filter."
},
"totalCount": {
"type": "number",
"description": "Total number of categories returned."
},
"notice": {
"description": "Guidance when the group filter returns no categories.",
"type": "string"
},
"error": {
"description": "Present when the call failed. Absent on success.",
"type": "object",
"properties": {
"code": {
"type": "integer",
"minimum": -9007199254740991,
"maximum": 9007199254740991,
"description": "JSON-RPC error code for this failure."
},
"message": {
"type": "string",
"description": "Human-readable description of what went wrong."
},
"data": {
"type": "object",
"properties": {
"reason": {
"type": "string",
"description": "Machine-readable failure mode."
},
"recovery": {
"description": "Actionable next step for the caller.",
"type": "object",
"properties": {
"hint": {
"type": "string"
}
},
"required": [
"hint"
],
"additionalProperties": {}
},
"retryable": {
"description": "Whether retrying may succeed.",
"type": "boolean"
}
},
"additionalProperties": {}
}
},
"required": [
"code",
"message"
],
"additionalProperties": {}
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"anyOf": [
{
"not": {
"required": [
"error"
]
},
"required": [
"categories",
"totalCount"
]
},
{
"required": [
"error"
]
}
]
}커뮤니티
증거