Skip to content

Archive client

ArchiveClient queries the public judgment datasets and manages local metadata and PDF caches. Install the archive extra before using it at runtime.

Read the archive guide for metadata matching, streaming order, publication lag and yearly SCI tar downloads. High Court citation filters are ignored; party matching uses the title.

ArchiveClient

ArchiveClient(
    *,
    cache_dir: str | None = None,
    cache_max_bytes: int | None = None,
    metadata_cache: bool = True
)

Query the public AWS Open Data judgment archives.

Example::

async with ArchiveClient() as client:
    results = await client.search(
        court="sci", judge="chandrachud", year=(2018, 2024), limit=20
    )
    for j in results:
        print(j.decision_date, j.title)
Source code in src/bharat_judgements/archive/client.py
def __init__(
    self,
    *,
    cache_dir: str | None = None,
    cache_max_bytes: int | None = None,
    metadata_cache: bool = True,
) -> None:
    self._query = _ArchiveQuery()
    self._storage = _PdfStorage(cache_dir=cache_dir, max_bytes=cache_max_bytes)
    # Metadata cache mirrors parquet shards locally when the query's
    # partition is fully resolved. Disable for one-off scans where
    # caching is wasted.
    self._meta_cache: _MetadataCache | None = _MetadataCache() if metadata_cache else None

search async

search(
    *,
    court: Court | str | None = None,
    year: int | tuple[int, int] | None = None,
    judge: str | None = None,
    party: str | None = None,
    citation: str | None = None,
    cnr: str | None = None,
    limit: int = 50
) -> list[Judgment]

Search judgments in the archive.

:param court: Court instance, a court code ("sci", "delhi"), or None to query both Supreme Court and all High Courts. :param year: Single year (2020) or inclusive range ((2018, 2024)). Year filters benefit from partition pruning and are strongly recommended for non-CNR queries. :param judge: Case-insensitive substring match on the judge field. :param party: Case-insensitive match. For SCI, searches petitioner, respondent, and title; for HC, searches title (HC parquet has no petitioner/respondent columns). :param citation: Case-insensitive substring match (SCI only — HC parquet has no citation column; select court="sci" when filtering by citation.) :param cnr: Exact match on the CNR (Court Number Record). When court isn't given, the CNR's 4-letter prefix is used to infer the right court and avoid scanning all 25 HC partitions. :param limit: Maximum results to return (split across sources if both queried).

Source code in src/bharat_judgements/archive/client.py
async def search(
    self,
    *,
    court: Court | str | None = None,
    year: int | tuple[int, int] | None = None,
    judge: str | None = None,
    party: str | None = None,
    citation: str | None = None,
    cnr: str | None = None,
    limit: int = 50,
) -> list[Judgment]:
    """Search judgments in the archive.

    :param court:
        ``Court`` instance, a court code (``"sci"``, ``"delhi"``), or
        ``None`` to query both Supreme Court and all High Courts.
    :param year:
        Single year (``2020``) or inclusive range (``(2018, 2024)``).
        Year filters benefit from partition pruning and are strongly
        recommended for non-CNR queries.
    :param judge: Case-insensitive substring match on the judge field.
    :param party:
        Case-insensitive match. For SCI, searches petitioner, respondent,
        and title; for HC, searches title (HC parquet has no
        petitioner/respondent columns).
    :param citation: Case-insensitive substring match (SCI only — HC
        parquet has no citation column; select court="sci" when filtering by citation.)
    :param cnr: Exact match on the CNR (Court Number Record). When ``court``
        isn't given, the CNR's 4-letter prefix is used to infer the right
        court and avoid scanning all 25 HC partitions.
    :param limit: Maximum results to return (split across sources if both
        queried).
    """
    if isinstance(year, tuple) and (len(year) != 2 or year[0] > year[1]):
        raise ValueError("year range must be ordered")
    if citation and (
        court is None or self._resolve_court(court).court_type != CourtType.SUPREME_COURT
    ):
        raise ValueError("citation requires court='sci'; HC archive has no citation field")
    resolved = self._resolve_court(court)
    # CNR-only queries: infer source from the prefix so we don't scan every
    # HC partition for an SCI CNR (or vice-versa).
    if resolved is None and cnr:
        resolved = infer_court_from_cnr(cnr)

    if isinstance(limit, bool) or limit < 1:
        raise ValueError("limit must be positive")
    per_source_limit = limit
    results: list[Judgment] = []

    if resolved is None or resolved.court_type == CourtType.SUPREME_COURT:
        sci_paths = await self._sci_paths_for_cache(year)
        sci_rows = await asyncio.to_thread(
            self._query.search_sci,
            year=year,
            judge=judge,
            party=party,
            citation=citation,
            cnr=cnr,
            limit=per_source_limit,
            paths_override=sci_paths,
        )
        results.extend(row_to_judgment(r) for r in sci_rows)

    if resolved is None or resolved.court_type == CourtType.HIGH_COURT:
        hc_court = resolved if resolved else None
        hc_paths = await self._hc_paths_for_cache(hc_court, year)
        hc_rows = await asyncio.to_thread(
            self._query.search_hc,
            court=hc_court,
            year=year,
            judge=judge,
            party=party,
            cnr=cnr,
            limit=per_source_limit,
            paths_override=hc_paths,
        )
        results.extend(row_to_judgment(r) for r in hc_rows)

    # Sort merged results so the cross-source list is still date-ordered.
    results.sort(
        key=lambda j: j.decision_date or __import__("datetime").date.min,
        reverse=True,
    )
    return results[:limit]

iter_judgments async

iter_judgments(
    *,
    court: Court | str | None = None,
    year: int | tuple[int, int] | None = None,
    judge: str | None = None,
    party: str | None = None,
    citation: str | None = None,
    cnr: str | None = None,
    batch_size: int = 500,
    max_results: int | None = None
) -> AsyncIterator[Judgment]

Stream judgments matching the filters, paging via LIMIT/OFFSET.

Use this instead of :meth:search for bulk pulls — e.g. "all Delhi 2020 judgments" (~18k rows) — to avoid materialising everything in memory and to start consuming results immediately.

Sources are streamed sequentially: SCI first (if applicable), then HC. There is no cross-source date merge — that would require holding both fully sorted lists. Each source is internally sorted by decision_date DESC, cnr so individual pages are deterministic.

batch_size controls the SQL page size. max_results caps the total yielded count across sources; None means no cap.

Source code in src/bharat_judgements/archive/client.py
async def iter_judgments(
    self,
    *,
    court: Court | str | None = None,
    year: int | tuple[int, int] | None = None,
    judge: str | None = None,
    party: str | None = None,
    citation: str | None = None,
    cnr: str | None = None,
    batch_size: int = 500,
    max_results: int | None = None,
) -> AsyncIterator[Judgment]:
    """Stream judgments matching the filters, paging via ``LIMIT/OFFSET``.

    Use this instead of :meth:`search` for bulk pulls — e.g. "all Delhi
    2020 judgments" (~18k rows) — to avoid materialising everything in
    memory and to start consuming results immediately.

    Sources are streamed sequentially: SCI first (if applicable), then HC.
    There is **no** cross-source date merge — that would require holding
    both fully sorted lists. Each source is internally sorted by
    ``decision_date DESC, cnr`` so individual pages are deterministic.

    ``batch_size`` controls the SQL page size. ``max_results`` caps the
    total yielded count across sources; ``None`` means no cap.
    """
    if isinstance(year, tuple) and (len(year) != 2 or year[0] > year[1]):
        raise ValueError("year range must be ordered")
    if citation and (
        court is None or self._resolve_court(court).court_type != CourtType.SUPREME_COURT
    ):
        raise ValueError("citation requires court='sci'; HC archive has no citation field")
    resolved = self._resolve_court(court)
    if resolved is None and cnr:
        resolved = infer_court_from_cnr(cnr)

    if batch_size < 1 or (max_results is not None and max_results < 1):
        raise ValueError("batch_size and max_results must be positive")
    yielded = 0
    sci_paths = await self._sci_paths_for_cache(year)
    hc_paths = await self._hc_paths_for_cache(resolved, year)

    async def _drain_source(source: str) -> AsyncIterator[Judgment]:
        nonlocal yielded
        offset = 0
        while True:
            if max_results is not None and yielded >= max_results:
                return
            page_limit = batch_size
            if max_results is not None:
                page_limit = min(batch_size, max_results - yielded)
            if source == "sci":
                rows = await asyncio.to_thread(
                    self._query.search_sci,
                    year=year,
                    judge=judge,
                    party=party,
                    citation=citation,
                    cnr=cnr,
                    limit=page_limit,
                    offset=offset,
                    paths_override=sci_paths,
                )
            else:  # hc
                rows = await asyncio.to_thread(
                    self._query.search_hc,
                    court=resolved,
                    year=year,
                    judge=judge,
                    party=party,
                    cnr=cnr,
                    limit=page_limit,
                    offset=offset,
                    paths_override=hc_paths,
                )
            if not rows:
                return
            for r in rows:
                yield row_to_judgment(r)
                yielded += 1
                if max_results is not None and yielded >= max_results:
                    return
            if len(rows) < page_limit:
                return  # last page
            offset += len(rows)

    if resolved is None or resolved.court_type == CourtType.SUPREME_COURT:
        async for j in _drain_source("sci"):
            yield j

    if resolved is None or resolved.court_type == CourtType.HIGH_COURT:
        async for j in _drain_source("hc"):
            yield j

fetch_pdf async

fetch_pdf(
    judgment_or_cnr: Judgment | str,
    *,
    language: str = "english"
) -> bytes

Fetch the PDF bytes for a judgment.

Pass a :class:Judgment (preferred — no extra lookup) or a CNR string (triggers one metadata query first). The language argument is only meaningful for SCI judgments; HC has English-only PDFs in the archive.

Raises :class:ArchivePdfError for missing files, missing metadata fields needed to construct the S3 path, or HTTP failures.

Source code in src/bharat_judgements/archive/client.py
async def fetch_pdf(
    self,
    judgment_or_cnr: Judgment | str,
    *,
    language: str = "english",
) -> bytes:
    """Fetch the PDF bytes for a judgment.

    Pass a :class:`Judgment` (preferred — no extra lookup) or a CNR string
    (triggers one metadata query first). The ``language`` argument is only
    meaningful for SCI judgments; HC has English-only PDFs in the archive.

    Raises :class:`ArchivePdfError` for missing files, missing metadata
    fields needed to construct the S3 path, or HTTP failures.
    """
    if isinstance(judgment_or_cnr, str):
        j = await self._lookup_by_cnr(judgment_or_cnr)
    else:
        j = judgment_or_cnr

    if j.court is None:
        raise ArchivePdfError(
            f"Judgment has no resolved court — cannot route PDF fetch (cnr={j.cnr})"
        )

    if j.court.court_type == CourtType.SUPREME_COURT:
        return await self._storage.fetch_sci_pdf(j, language=language)
    return await self._storage.fetch_hc_pdf(j)

prefetch_sci_year async

prefetch_sci_year(
    year: int, language: str = "english"
) -> str

Pre-warm the SCI tar cache for a year. Returns the local path.

Source code in src/bharat_judgements/archive/client.py
async def prefetch_sci_year(self, year: int, language: str = "english") -> str:
    """Pre-warm the SCI tar cache for a year. Returns the local path."""
    path = await self._storage.prefetch_sci_tar(year, language=language)
    return str(path)

cache_info

cache_info() -> dict

Snapshot of cache directory + total bytes + cap.

Source code in src/bharat_judgements/archive/client.py
def cache_info(self) -> dict:
    """Snapshot of cache directory + total bytes + cap."""
    return self._storage.cache_info()

count async

count(
    *,
    court: Court | str | None = None,
    year: int | None = None
) -> dict[str, int]

Return per-source row counts. Useful for sizing and sanity checks.

Source code in src/bharat_judgements/archive/client.py
async def count(
    self,
    *,
    court: Court | str | None = None,
    year: int | None = None,
) -> dict[str, int]:
    """Return per-source row counts. Useful for sizing and sanity checks."""
    resolved = self._resolve_court(court)
    out: dict[str, int] = {}

    if resolved is None or resolved.court_type == CourtType.SUPREME_COURT:
        out["sci"] = await asyncio.to_thread(self._query.count_sci, year=year)

    if resolved is None or resolved.court_type == CourtType.HIGH_COURT:
        out["hc"] = await asyncio.to_thread(self._query.count_hc, court=resolved, year=year)

    return out