← Files Life Sciences DatabasesARCHIVED FILE
skills/ncbi-datasets-skill/scripts/ncbi_datasets.py
9.24 KB · Sep 30, 2026 · 23:00 UTC
#!/usr/bin/env python3
"""Compact NCBI Datasets v2 helper for imported skills."""
from __future__ import annotations
import json
import sys
from pathlib import Path
from typing import Any
sys.path.insert(0, str(Path(__file__).resolve().parents[3] / "scripts"))
from database_source_contract import apply_source_contract # noqa: E402
try:
import requests
except ImportError as exc: # pragma: no cover
requests = None
REQUESTS_IMPORT_ERROR = exc
else:
REQUESTS_IMPORT_ERROR = None
DATASETS_BASE = "https://api.ncbi.nlm.nih.gov/datasets/v2"
def error(code: str, message: str, warnings: list[str] | None = None) -> dict[str, Any]:
return {
"ok": False,
"error": {"code": code, "message": message},
"warnings": warnings or [],
}
def _require_str(name: str, value: Any, required: bool = False) -> str | None:
if value is None:
if required:
raise ValueError(f"`{name}` is required.")
return None
if not isinstance(value, str) or not value.strip():
raise ValueError(f"`{name}` must be a non-empty string.")
return value.strip()
def _require_int(name: str, value: Any, default: int) -> int:
if value is None:
return default
if not isinstance(value, int) or value <= 0:
raise ValueError(f"`{name}` must be a positive integer.")
return value
def _require_bool(name: str, value: Any, default: bool) -> bool:
if value is None:
return default
if not isinstance(value, bool):
raise ValueError(f"`{name}` must be a boolean.")
return value
def _require_object(name: str, value: Any) -> dict[str, Any]:
if value is None:
return {}
if not isinstance(value, dict):
raise ValueError(f"`{name}` must be an object.")
return value
def _get_by_path(value: Any, path: str) -> Any:
current = value
for part in path.split("."):
if isinstance(current, list):
if not part.isdigit():
raise ValueError(f"`record_path` segment {part!r} must be a list index.")
index = int(part)
if index >= len(current):
raise ValueError(f"`record_path` index {index} is out of range.")
current = current[index]
elif isinstance(current, dict):
if part not in current:
raise ValueError(f"`record_path` key {part!r} was not present in the response.")
current = current[part]
else:
raise ValueError(f"`record_path` segment {part!r} could not be applied.")
return current
def _infer_target(data: Any) -> tuple[str | None, Any]:
if isinstance(data, list):
return "$", data
if isinstance(data, dict):
for key in ("result", "results", "records", "uids", "documents"):
value = data.get(key)
if isinstance(value, list):
return key, value
return None, data
def _compact(value: Any, max_items: int, max_depth: int) -> Any:
if isinstance(value, str):
return value if len(value) <= 240 else value[:240] + "..."
if max_depth <= 0:
if isinstance(value, (dict, list)):
return "..."
return value
if isinstance(value, list):
out = [_compact(item, max_items, max_depth - 1) for item in value[:max_items]]
if len(value) > max_items:
out.append(f"... (+{len(value) - max_items} more)")
return out
if isinstance(value, dict):
out: dict[str, Any] = {}
items = list(value.items())
for key, item in items[:max_items]:
out[str(key)] = _compact(item, max_items, max_depth - 1)
if len(items) > max_items:
out["_truncated_keys"] = len(items) - max_items
return out
return value
def _save_raw_text(raw_text: str | bytes, raw_output_path: str | None, suffix: str) -> str:
path = Path(raw_output_path or f"/tmp/ncbi-datasets.{suffix}")
path.parent.mkdir(parents=True, exist_ok=True)
raw_bytes = raw_text if isinstance(raw_text, bytes) else raw_text.encode("utf-8")
path.write_bytes(raw_bytes)
return str(path)
def parse_input(payload: Any) -> dict[str, Any]:
if not isinstance(payload, dict):
raise ValueError("Input must be one JSON object.")
response_format = (
_require_str("response_format", payload.get("response_format")) or "auto"
).lower()
if response_format not in {"auto", "json", "text"}:
raise ValueError("`response_format` must be auto, json, or text.")
return {
"path": _require_str("path", payload.get("path"), required=True),
"params": _require_object("params", payload.get("params")),
"record_path": _require_str("record_path", payload.get("record_path")),
"response_format": response_format,
"max_items": _require_int("max_items", payload.get("max_items"), 10),
"max_depth": _require_int("max_depth", payload.get("max_depth"), 3),
"timeout_sec": _require_int("timeout_sec", payload.get("timeout_sec"), 30),
"save_raw": _require_bool("save_raw", payload.get("save_raw"), False),
"raw_output_path": _require_str("raw_output_path", payload.get("raw_output_path")),
}
def _json_output(data: Any, config: dict[str, Any], raw_output_path: str | None) -> dict[str, Any]:
record_path = config["record_path"]
path_used, target = (
_infer_target(data)
if record_path is None
else (record_path, _get_by_path(data, record_path))
)
output = {
"ok": True,
"source": "ncbi-datasets",
"path": config["path"],
"record_path": path_used,
"raw_output_path": raw_output_path,
"warnings": [],
}
if isinstance(target, list):
output.update(
{
"record_count_returned": min(len(target), config["max_items"]),
"record_count_available": len(target),
"truncated": len(target) > config["max_items"],
"records": _compact(
target[: config["max_items"]],
config["max_items"],
config["max_depth"],
),
}
)
else:
output["summary"] = _compact(target, config["max_items"], config["max_depth"])
if isinstance(target, dict):
output["top_keys"] = list(target)[: config["max_items"]]
request_url = DATASETS_BASE.rstrip("/") + "/" + config["path"].lstrip("/")
return apply_source_contract(output, "ncbi-datasets-skill", request_url)
def execute(payload: Any) -> dict[str, Any]:
if requests is None:
return error("missing_dependency", f"`requests` is required: {REQUESTS_IMPORT_ERROR}")
config = parse_input(payload)
try:
url = DATASETS_BASE.rstrip("/") + "/" + config["path"].lstrip("/")
response = requests.get(url, params=config["params"], timeout=config["timeout_sec"])
response.raise_for_status()
text = response.text
raw_content = getattr(response, "content", None)
raw_response = (
bytes(raw_content)
if isinstance(raw_content, (bytes, bytearray, memoryview))
else text.encode("utf-8")
)
content_type = (response.headers.get("content-type") or "").lower()
if (
config["response_format"] == "json"
or "json" in content_type
or text.lstrip().startswith("{")
):
data = response.json()
raw_output_path = None
if config["save_raw"]:
raw_output_path = _save_raw_text(raw_response, config["raw_output_path"], "json")
return _json_output(data, config, raw_output_path)
raw_output_path = (
_save_raw_text(raw_response, config["raw_output_path"], "txt")
if config["save_raw"]
else None
)
text_head = None if raw_output_path else text[:800]
return apply_source_contract(
{
"ok": True,
"source": "ncbi-datasets",
"path": config["path"],
"text_head": text_head,
"text_head_truncated": False if raw_output_path else len(text) > 800,
"raw_output_path": raw_output_path,
"warnings": [],
},
"ncbi-datasets-skill",
url,
)
except ValueError as exc:
return error("invalid_response", str(exc))
except requests.RequestException as exc:
response = getattr(exc, "response", None)
status_code = getattr(response, "status_code", None)
message = (
f"NCBI Datasets returned HTTP {status_code}."
if isinstance(status_code, int)
else f"NCBI Datasets request failed ({type(exc).__name__})."
)
return error("network_error", message)
def main() -> int:
try:
payload = json.load(sys.stdin)
except Exception as exc: # noqa: BLE001
sys.stdout.write(json.dumps(error("invalid_json", f"Could not parse JSON input: {exc}")))
return 2
try:
output = execute(payload)
except ValueError as exc:
output = error("invalid_input", str(exc))
code = 2
else:
code = 0 if output.get("ok") else 1
sys.stdout.write(json.dumps(output))
return code
if __name__ == "__main__":
raise SystemExit(main())
SHA-256: 26a7d0979433f17e03b2d5d3fe5619c87258eb8bd2ca2eec3795d49fabaccc1f