Files
content-ingestion-agent/app/kab_ingestion/connectors/github.py

33 lines
1.6 KiB
Python

from __future__ import annotations
import json
from dataclasses import dataclass
from urllib.request import Request, urlopen
from ..domain.models import IngestionRequest, SourceDocument
@dataclass
class GitHubSCMConnector:
token: str | None = None
api_base: str = "https://api.github.com"
def fetch(self, request: IngestionRequest) -> list[SourceDocument]:
"""Fetch a file or repository contents using the GitHub contents API."""
url = request.locator if request.locator.startswith("http") else f"{self.api_base}/{request.locator.lstrip('/')}"
headers = {"Accept": "application/vnd.github+json", "User-Agent": "content-ingestion-agent"}
if self.token:
headers["Authorization"] = f"Bearer {self.token}"
with urlopen(Request(url, headers=headers), timeout=20) as response:
payload = json.loads(response.read().decode("utf-8"))
items = payload if isinstance(payload, list) else [payload]
return [self._document(item, request) for item in items if item.get("type", "file") == "file"]
@staticmethod
def _document(item: dict, request: IngestionRequest) -> SourceDocument:
body = item.get("content", "")
if item.get("encoding") == "base64":
import base64
body = base64.b64decode(body).decode("utf-8", errors="replace")
return SourceDocument(identifier=str(item.get("sha", item.get("path", "unknown"))), title=item.get("name", item.get("path", "Untitled")), body=body, source=request.source, url=item.get("html_url"), metadata={"path": item.get("path")})