Pular para conteúdo

Referência da API

Esta referência é gerada automaticamente a partir das docstrings do código por meio do mkdocstrings. O estilo de docstring configurado é o Google, e o código-fonte de cada membro é exibido junto da assinatura.

Os blocos abaixo correspondem aos módulos públicos do pacote alberto_research. Para o uso pela linha de comando, veja Uso; para a visão geral dos módulos, veja Arquitetura.

Configuração

Carregamento, validação e campos obrigatórios do YAML de projeto.

config

Fluxo de trabalho

Orquestração do fluxo completo: descoberta, deduplicação, filtros, resolução de texto completo, triagem, leitura e busca de citações.

workflow

Texto completo

Cadeia de resolvedores, cache de PDFs e limites de download.

fulltext

Esquemas

Contratos JSON Schema usados para validar a saída do LLM antes de ela ser persistida.

schemas

Leitura

Construção de prompts, normalização da resposta do modelo e template de saída estruturada.

reader

Provedores

Clientes dos provedores de descoberta e dos resolvedores externos.

providers

Banco de dados

Conexão SQLite, migrações e repositórios de acesso a dados.

db

Database helpers for Alberto.

Linha de comando

Definição dos subcomandos e ponto de entrada do script alberto-research.

cli

Command-line interface for Alberto Research.

build_parser()

Build the argument parser for the alberto-research command.

Source code in src/alberto_research/cli.py
def build_parser() -> argparse.ArgumentParser:
    """Build the argument parser for the ``alberto-research`` command."""
    parser = argparse.ArgumentParser(
        prog=PROG,
        description="Discovery, screening, reading, synthesis and digest of scientific literature.",
    )
    parser.add_argument("--version", action="version", version=f"{PROG} {__version__}")
    sub = parser.add_subparsers(dest="command", required=True)

    db = sub.add_parser("db", help="Database maintenance commands.")
    db_sub = db.add_subparsers(dest="db_command", required=True)
    migrate = db_sub.add_parser("migrate", help="Apply pending SQLite migrations.")
    migrate.add_argument("--db")

    config = sub.add_parser("config", help="Project configuration commands.")
    config_sub = config.add_subparsers(dest="config_command", required=True)
    validate = config_sub.add_parser("validate", help="Validate a project YAML file.")
    validate.add_argument("project")

    research = sub.add_parser("research", help="Research workflow commands.")
    research_sub = research.add_subparsers(dest="research_command", required=True)

    run = research_sub.add_parser("run", help="Run discovery and screening.")
    run.add_argument("--project", required=True)
    run.add_argument("--db")
    run.add_argument("--dry-run", action="store_true")

    digest = research_sub.add_parser("digest", help="Generate the daily digest.")
    digest.add_argument("--project", required=True)
    digest.add_argument("--db")
    digest.add_argument("--output-dir")

    feedback = research_sub.add_parser("feedback", help="Record feedback for a digest item.")
    feedback.add_argument("--project", required=True)
    feedback.add_argument("--db")
    feedback.add_argument("--type", required=True, choices=[item.value for item in FeedbackType])
    feedback.add_argument("--digest-item-id")
    feedback.add_argument("--paper-id", type=int)
    feedback.add_argument("--note")

    notion = sub.add_parser("notion", help="Notion archive commands.")
    notion_sub = notion.add_subparsers(dest="notion_command", required=True)
    notion_setup = notion_sub.add_parser("setup", help="Create the Notion archive database.")
    notion_setup.add_argument("--parent-page-id", required=True)
    notion_setup.add_argument("--title", default="Alberto Research Library")
    notion_backfill = notion_sub.add_parser("backfill", help="Backfill readings into Notion.")
    notion_backfill.add_argument("--db")
    notion_backfill.add_argument("--project-id")

    return parser

main(argv=None)

Run the command-line interface and return a process exit code.

Source code in src/alberto_research/cli.py
def main(argv: Sequence[str] | None = None) -> int:
    """Run the command-line interface and return a process exit code."""
    configure_logging()
    parser = build_parser()
    args = parser.parse_args(argv)

    if args.command == "db" and args.db_command == "migrate":
        conn = connect(args.db)
        applied = apply_migrations(conn)
        conn.close()
        print(json.dumps({"applied": applied}))
        return 0

    if args.command == "config" and args.config_command == "validate":
        config_data = load_project_config(args.project)
        print(json.dumps({"ok": True, "project_id": config_data["id"]}))
        return 0

    if args.command == "research" and args.research_command == "run":
        run_id = run_research_workflow(
            project_path=args.project, db_path=args.db, dry_run=args.dry_run
        )
        print(json.dumps({"run_id": run_id}))
        return 0

    if args.command == "research" and args.research_command == "digest":
        digest_id, path = run_digest_workflow(
            project_path=args.project, db_path=args.db, output_dir=args.output_dir
        )
        print(json.dumps({"digest_id": digest_id, "path": str(path)}))
        return 0

    if args.command == "research" and args.research_command == "feedback":
        config_data = load_project_config(args.project)
        conn = connect(args.db)
        apply_migrations(conn)
        repo = AlbertoRepository(conn)
        repo.upsert_project(config_data, args.project)
        feedback_id = store_feedback(
            repo,
            project_id=config_data["id"],
            feedback_type=args.type,
            digest_item_id=args.digest_item_id,
            paper_id=args.paper_id,
            note=args.note,
        )
        conn.close()
        print(json.dumps({"feedback_id": feedback_id}))
        return 0

    if args.command == "notion" and args.notion_command == "setup":
        database = NotionAdapter().create_article_database(
            parent_page_id=args.parent_page_id, title=args.title
        )
        print(
            json.dumps(
                {"database_id": database.database_id, "data_source_id": database.data_source_id}
            )
        )
        return 0

    if args.command == "notion" and args.notion_command == "backfill":
        conn = connect(args.db)
        apply_migrations(conn)
        linked, report = backfill_digest_readings_to_notion(
            AlbertoRepository(conn), project_id=args.project_id
        )
        conn.close()
        print(
            json.dumps(
                {
                    "linked": linked,
                    "status": report.status,
                    "created": report.created,
                    "updated": report.updated,
                    "error": report.error,
                }
            )
        )
        return 0 if report.status not in {"failed", "not_configured"} else 1

    print(f"error: unsupported command: {args.command}", file=sys.stderr)
    return 2

Digest

Montagem do corpo do digest e gravação do Markdown local.

digest

Entrega

Envio do digest pelos canais configurados, incluindo SMTP.

delivery

Zotero

Integração opcional com a API do Zotero.

zotero

ZoteroAdapter

Official Zotero Web API adapter.

SQLite remains operational state; Zotero is synchronized as the user's human research library when credentials are configured.

Source code in src/alberto_research/zotero.py
class ZoteroAdapter:
    """Official Zotero Web API adapter.

    SQLite remains operational state; Zotero is synchronized as the user's human
    research library when credentials are configured.
    """

    def __init__(
        self,
        *,
        api_key: str | None = None,
        library_type: str | None = None,
        library_id: str | None = None,
        base_url: str = "https://api.zotero.org",
    ):
        self.api_key = api_key or os.environ.get("ZOTERO_API_KEY")
        self.library_type = library_type or os.environ.get("ZOTERO_LIBRARY_TYPE", "user")
        self.library_id = library_id or os.environ.get("ZOTERO_LIBRARY_ID")
        self.base_url = base_url.rstrip("/")

    @property
    def configured(self) -> bool:
        return bool(self.api_key and self.library_id)

    def _headers(self) -> dict[str, str]:
        if not self.configured:
            raise RuntimeError("Zotero credentials are not configured")
        return {"Zotero-API-Key": self.api_key or "", "Content-Type": "application/json"}

    def _library_url(self, suffix: str) -> str:
        return f"{self.base_url}/{self.library_type}s/{self.library_id}{suffix}"

    def search(self, query: str) -> list[dict[str, Any]]:
        return cast(
            "list[dict[str, Any]]",
            self._request("GET", "/items", params={"q": query, "format": "json"}),
        )

    def find_by_doi(self, doi: str) -> dict[str, Any] | None:
        normalized = normalize_doi(doi)
        for item in self.search(normalized or doi):
            data = item.get("data", {})
            if normalize_doi(data.get("DOI")) == normalized:
                return item
        return None

    def find_item_by_doi(self, doi: str) -> dict[str, Any] | None:
        return self.find_by_doi(doi)

    def children(self, item_key: str) -> list[dict[str, Any]]:
        return cast(
            "list[dict[str, Any]]",
            self._request("GET", f"/items/{item_key}/children", params={"format": "json"}),
        )

    def pdf_attachments(self, item_key: str) -> list[dict[str, Any]]:
        attachments = []
        for child in self.children(item_key):
            data = child.get("data", {})
            content_type = (data.get("contentType") or "").lower()
            title = (data.get("title") or "").lower()
            filename = (data.get("filename") or "").lower()
            if data.get("itemType") == "attachment" and (
                content_type == "application/pdf"
                or filename.endswith(".pdf")
                or title.endswith(".pdf")
            ):
                attachments.append(child)
        return attachments

    def attachment_fulltext(self, attachment_key: str) -> str | None:
        payload = self._request("GET", f"/items/{attachment_key}/fulltext")
        if isinstance(payload, dict):
            text = payload.get("content") or payload.get("text")
            return text if isinstance(text, str) and text.strip() else None
        return None

    def download_attachment_file(self, attachment_key: str) -> tuple[bytes, str | None]:
        return self._request_bytes("GET", f"/items/{attachment_key}/file")

    def create_item(self, item: dict[str, Any]) -> Any:
        return self._request("POST", "/items", json=[item])

    def update_metadata(self, item_key: str, version: int, data: dict[str, Any]) -> Any:
        headers = self._headers() | {"If-Unmodified-Since-Version": str(version)}
        return self._request("PATCH", f"/items/{item_key}", json=data, headers=headers)

    def set_tags(self, item_key: str, version: int, tags: list[str]) -> Any:
        return self.update_metadata(item_key, version, {"tags": [{"tag": tag} for tag in tags]})

    def add_note(self, parent_key: str, note_html: str) -> Any:
        return self.create_item({"itemType": "note", "parentItem": parent_key, "note": note_html})

    def record_attachment_metadata(self, parent_key: str, title: str, url: str) -> Any:
        return self.create_item(
            {
                "itemType": "attachment",
                "parentItem": parent_key,
                "title": title,
                "url": url,
                "linkMode": "linked_url",
            }
        )

    def dedupe_key(self, item: dict[str, Any]) -> str:
        data = item.get("data", item)
        doi = normalize_doi(data.get("DOI"))
        if doi:
            return f"doi:{doi}"
        return f"title:{(data.get('title') or '').strip().lower()}"

    def _request(self, method: str, suffix: str, **kwargs: Any) -> Any:
        try:
            import requests
        except ModuleNotFoundError as exc:  # pragma: no cover
            raise RuntimeError("requests is required for Zotero API calls") from exc
        headers = kwargs.pop("headers", self._headers())
        response = requests.request(
            method, self._library_url(suffix), headers=headers, timeout=30, **kwargs
        )
        response.raise_for_status()
        if response.content:
            return response.json()
        return None

    def _request_bytes(self, method: str, suffix: str, **kwargs: Any) -> tuple[bytes, str | None]:
        try:
            import requests
        except ModuleNotFoundError as exc:  # pragma: no cover
            raise RuntimeError("requests is required for Zotero API calls") from exc
        headers = kwargs.pop("headers", self._headers())
        response = requests.request(
            method, self._library_url(suffix), headers=headers, timeout=60, **kwargs
        )
        response.raise_for_status()
        return response.content, response.headers.get("Content-Type")

Notion

Integração opcional com a API do Notion, criação do banco e backfill.

notion

NotionAdapter

Official Notion API adapter for validated digest readings.

Source code in src/alberto_research/notion.py
class NotionAdapter:
    """Official Notion API adapter for validated digest readings."""

    def __init__(
        self,
        *,
        api_key: str | None = None,
        data_source_id: str | None = None,
        database_id: str | None = None,
        base_url: str = NOTION_BASE_URL,
    ):
        self.api_key = api_key or os.environ.get("NOTION_API_KEY")
        self.data_source_id = data_source_id or os.environ.get("NOTION_DATA_SOURCE_ID")
        self.database_id = database_id or os.environ.get("NOTION_DATABASE_ID")
        self.base_url = base_url.rstrip("/")

    @classmethod
    def from_project_config(cls, config: dict[str, Any]) -> NotionAdapter:
        settings = config.get("notion") or {}
        if not isinstance(settings, dict):
            settings = {}
        enabled_by_environment = os.environ.get("ALBERTO_NOTION_ENABLED") == "1"
        if settings.get("enabled") is not True and not enabled_by_environment:
            return cls(api_key="", data_source_id="", database_id="")
        return cls(
            data_source_id=settings.get("data_source_id"),
            database_id=settings.get("database_id"),
        )

    @property
    def configured(self) -> bool:
        return bool(self.api_key and (self.data_source_id or self.database_id))

    def create_article_database(
        self, *, parent_page_id: str, title: str = "Alberto Research Library"
    ) -> NotionDatabase:
        if not self.api_key:
            raise RuntimeError("NOTION_API_KEY is not configured")
        payload = {
            "parent": {"type": "page_id", "page_id": parent_page_id},
            "title": text_blocks(title),
            "description": text_blocks(
                "Artigos lidos pelo Alberto Research e incluídos em digests."
            ),
            "initial_data_source": {"properties": article_database_properties()},
        }
        response = self._request("POST", "/databases", json=payload)
        database_id = str(response["id"])
        data_source_id = extract_data_source_id(response)
        if not data_source_id:
            data_source_id = extract_data_source_id(
                self._request("GET", f"/databases/{database_id}")
            )
        if not data_source_id:
            raise RuntimeError("Notion did not return the initial data source id")
        return NotionDatabase(database_id=database_id, data_source_id=data_source_id)

    def resolved_data_source_id(self) -> str:
        if self.data_source_id:
            return self.data_source_id
        if not self.database_id:
            raise RuntimeError("NOTION_DATA_SOURCE_ID or NOTION_DATABASE_ID is not configured")
        response = self._request("GET", f"/databases/{self.database_id}")
        data_source_id = extract_data_source_id(response)
        if not data_source_id:
            raise RuntimeError("Notion database has no accessible data source")
        return data_source_id

    def ensure_article_schema(self, data_source_id: str) -> None:
        """Add Alberto's fields without removing existing user-created fields."""
        source = self._request("GET", f"/data_sources/{data_source_id}")
        existing = source.get("properties") if isinstance(source, dict) else None
        if not isinstance(existing, dict):
            raise RuntimeError("Notion data source did not return its property schema")
        updates: dict[str, Any] = {}
        article = existing.get("Article")
        if article is not None and article.get("type") != "title":
            raise RuntimeError("Notion property 'Article' exists but is not a title property")
        if article is None:
            title_property = next(
                (value for value in existing.values() if value.get("type") == "title"), None
            )
            if title_property and title_property.get("id"):
                updates[str(title_property["id"])] = {"name": "Article"}
            else:
                updates["Article"] = {"title": {}}
        for name, definition in article_database_properties().items():
            if name == "Article":
                continue
            current = existing.get(name)
            expected_type = next(iter(definition))
            if current is None:
                updates[name] = definition
            elif current.get("type") != expected_type:
                raise RuntimeError(
                    f"Notion property '{name}' has type {current.get('type')!r}; expected {expected_type!r}"
                )
        if updates:
            self._request("PATCH", f"/data_sources/{data_source_id}", json={"properties": updates})

    def existing_article_pages(self, data_source_id: str) -> dict[str, str]:
        """Return page ids indexed by DOI and normalized title for deduplication."""
        index: dict[str, str] = {}
        cursor: str | None = None
        while True:
            payload: dict[str, Any] = {"page_size": 100}
            if cursor:
                payload["start_cursor"] = cursor
            response = self._request("POST", f"/data_sources/{data_source_id}/query", json=payload)
            for page in response.get("results", []):
                if not isinstance(page, dict) or not page.get("id"):
                    continue
                properties = page.get("properties")
                if not isinstance(properties, dict):
                    continue
                page_id = str(page["id"])
                doi = normalize_doi(notion_property_text(properties.get("DOI")))
                title = normalize_text(notion_property_text(properties.get("Article")))
                if doi:
                    index.setdefault(f"doi:{doi}", page_id)
                if title:
                    index.setdefault(f"title:{title}", page_id)
            cursor = response.get("next_cursor")
            if not cursor or not response.get("has_more"):
                return index

    def create_article_page(
        self, data_source_id: str, properties: dict[str, Any], children: list[dict[str, Any]]
    ) -> str:
        response = self._request(
            "POST",
            "/pages",
            json={
                "parent": {"type": "data_source_id", "data_source_id": data_source_id},
                "properties": properties,
                "children": children,
            },
        )
        return str(response["id"])

    def update_article_page(self, page_id: str, properties: dict[str, Any]) -> None:
        self._request("PATCH", f"/pages/{page_id}", json={"properties": properties})

    def _request(self, method: str, path: str, **kwargs: Any) -> Any:
        try:
            import requests
        except ModuleNotFoundError as exc:  # pragma: no cover
            raise RuntimeError("requests is required for Notion synchronization") from exc
        if not self.api_key:
            raise RuntimeError("NOTION_API_KEY is not configured")
        headers = {
            "Authorization": f"Bearer {self.api_key}",
            "Notion-Version": NOTION_API_VERSION,
            "Content-Type": "application/json",
        }
        try:
            response = requests.request(
                method, f"{self.base_url}{path}", headers=headers, timeout=30, **kwargs
            )
        except requests.RequestException as exc:
            raise RuntimeError("Notion API request could not be completed") from exc
        try:
            response.raise_for_status()
        except Exception as exc:
            detail = response.text[:500] if response.content else ""
            raise RuntimeError(f"Notion API request failed: {detail}") from exc
        return response.json() if response.content else {}

ensure_article_schema(data_source_id)

Add Alberto's fields without removing existing user-created fields.

Source code in src/alberto_research/notion.py
def ensure_article_schema(self, data_source_id: str) -> None:
    """Add Alberto's fields without removing existing user-created fields."""
    source = self._request("GET", f"/data_sources/{data_source_id}")
    existing = source.get("properties") if isinstance(source, dict) else None
    if not isinstance(existing, dict):
        raise RuntimeError("Notion data source did not return its property schema")
    updates: dict[str, Any] = {}
    article = existing.get("Article")
    if article is not None and article.get("type") != "title":
        raise RuntimeError("Notion property 'Article' exists but is not a title property")
    if article is None:
        title_property = next(
            (value for value in existing.values() if value.get("type") == "title"), None
        )
        if title_property and title_property.get("id"):
            updates[str(title_property["id"])] = {"name": "Article"}
        else:
            updates["Article"] = {"title": {}}
    for name, definition in article_database_properties().items():
        if name == "Article":
            continue
        current = existing.get(name)
        expected_type = next(iter(definition))
        if current is None:
            updates[name] = definition
        elif current.get("type") != expected_type:
            raise RuntimeError(
                f"Notion property '{name}' has type {current.get('type')!r}; expected {expected_type!r}"
            )
    if updates:
        self._request("PATCH", f"/data_sources/{data_source_id}", json={"properties": updates})

existing_article_pages(data_source_id)

Return page ids indexed by DOI and normalized title for deduplication.

Source code in src/alberto_research/notion.py
def existing_article_pages(self, data_source_id: str) -> dict[str, str]:
    """Return page ids indexed by DOI and normalized title for deduplication."""
    index: dict[str, str] = {}
    cursor: str | None = None
    while True:
        payload: dict[str, Any] = {"page_size": 100}
        if cursor:
            payload["start_cursor"] = cursor
        response = self._request("POST", f"/data_sources/{data_source_id}/query", json=payload)
        for page in response.get("results", []):
            if not isinstance(page, dict) or not page.get("id"):
                continue
            properties = page.get("properties")
            if not isinstance(properties, dict):
                continue
            page_id = str(page["id"])
            doi = normalize_doi(notion_property_text(properties.get("DOI")))
            title = normalize_text(notion_property_text(properties.get("Article")))
            if doi:
                index.setdefault(f"doi:{doi}", page_id)
            if title:
                index.setdefault(f"title:{title}", page_id)
        cursor = response.get("next_cursor")
        if not cursor or not response.get("has_more"):
            return index

backfill_digest_readings_to_notion(repo, *, adapter=None, project_id=None)

Synchronize validated readings and historical digest candidates.

Source code in src/alberto_research/notion.py
def backfill_digest_readings_to_notion(
    repo: AlbertoRepository,
    *,
    adapter: NotionAdapter | None = None,
    project_id: str | None = None,
) -> tuple[int, NotionSyncReport]:
    """Synchronize validated readings and historical digest candidates."""
    adapter = adapter or NotionAdapter()
    if not adapter.configured:
        return 0, NotionSyncReport(status="not_configured")
    linked = repo.link_historical_digest_readings(project_id)
    rows = repo.all_readings_for_notion(project_id) + repo.digest_candidates_for_notion(project_id)
    return linked, sync_notion_rows(repo, rows, adapter=adapter)

OpenClaw

Resolução do binário, montagem da linha de comando e leitura da resposta JSON.

openclaw

resolve_openclaw_binary(config=None)

Resolve the openclaw executable without a hard-coded host path.

Resolution order: openclaw.binary in the project config, then the ALBERTO_OPENCLAW_BIN environment variable, then PATH lookup.

Source code in src/alberto_research/openclaw.py
def resolve_openclaw_binary(config: dict[str, Any] | None = None) -> str:
    """Resolve the ``openclaw`` executable without a hard-coded host path.

    Resolution order: ``openclaw.binary`` in the project config, then the
    ``ALBERTO_OPENCLAW_BIN`` environment variable, then ``PATH`` lookup.
    """
    settings = config.get("openclaw") if isinstance(config, dict) else None
    if isinstance(settings, dict):
        value = settings.get("binary")
        if isinstance(value, str) and value.strip():
            return value.strip()
    from_env = os.environ.get(OPENCLAW_BINARY_ENV)
    if from_env and from_env.strip():
        return from_env.strip()
    return shutil.which(DEFAULT_OPENCLAW_BINARY) or DEFAULT_OPENCLAW_BINARY

openclaw_agent_command(agent, *, config=None, timeout=None)

Build an openclaw agent command line for the given agent.

Source code in src/alberto_research/openclaw.py
def openclaw_agent_command(
    agent: str,
    *,
    config: dict[str, Any] | None = None,
    timeout: int | None = None,
) -> list[str]:
    """Build an ``openclaw agent`` command line for the given agent."""
    command = [resolve_openclaw_binary(config), "agent", "--agent", agent]
    if timeout is not None:
        command += ["--timeout", str(timeout)]
    return command

Feedback

Registro do retorno do leitor sobre os itens do digest.

feedback

Deduplicação

Normalização de identificadores e deduplicação de registros.

dedupe