Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions CHANGES.md
Original file line number Diff line number Diff line change
@@ -1,3 +1,11 @@
Version 0.4.0 (unreleased)
---------------------------
* Changed: `FOLIO.generate_iri()` now mints `R` + base62 of 127 random bits (e.g. `R1cNH7TLMiSlSbIbdFsynUk`), matching the scheme used by 11,427 of the 18,325 published FOLIO concepts, instead of a base64url-derived token with no `R` prefix. Newly minted IRIs are now indistinguishable from their published siblings. No existing IRI changes; `generate_iri()` only mints new ones
* Fixed: the uniqueness check in `generate_iri()` never fired. It tested a bare local name for membership in `FOLIO.iri_to_index`, whose keys are full IRIs taken from `rdf:about`, so no candidate was ever rejected and the `MAX_IRI_ATTEMPTS` retry loop could not run. The check now accepts either spelling
* Fixed: `generate_iri()` could emit a degenerately short local name. It stripped non-alphanumeric characters *after* base64 encoding, so an unlucky uuid4 lost characters — about 1 mint in 29,000 fell to 16 characters or fewer, violating the library's own `assert len(b64_token) > 16`. Base62 has no stripping step, so the failure mode is gone
* Added: `folio.iri`, a standard-library-only module holding the generator (`generate_iri`, `generate_local_name`, `encode_base62`) plus the pinned constants. Minting no longer requires constructing a `FOLIO` graph, which downloads and parses the full ontology
* Added: `tests/test_iri.py` — 15 tests covering base62 round-tripping, output shape, entropy width, both collision-index spellings, and retry bounds

Version 0.3.7 (2026-07-24)
---------------------------
* Security: Bumped `lxml` floor to `>=6.1.0` (locked 6.1.1) — CVE-2026-41066 / GHSA-vfmq-68hx-4jfw, XXE to local files via the default configuration of `iterparse()` and `ETCompatXMLParser()`
Expand Down
35 changes: 6 additions & 29 deletions folio/graph.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,13 +15,11 @@

# imports
import asyncio
import base64
import hashlib
import importlib.util
import json
import time
import traceback
import uuid
from enum import Enum
from functools import cache
from pathlib import Path
Expand All @@ -46,6 +44,7 @@
DEFAULT_HTTP_URL,
DEFAULT_SOURCE_TYPE,
)
from folio.iri import MAX_IRI_ATTEMPTS, generate_iri
from folio.logger import get_logger
from folio.models import OWLClass, OWLObjectProperty, NSMAP

Expand Down Expand Up @@ -116,9 +115,6 @@ class FOLIOTypes(Enum):
# Default maximum depth for subgraph traversal safety
DEFAULT_MAX_DEPTH: int = 16

# IRI max generation attempt for safety.
MAX_IRI_ATTEMPTS: int = 16

# default max tokens to return from LLM
DEFAULT_MAX_TOKENS: int = 1024

Expand Down Expand Up @@ -2473,31 +2469,12 @@ def generate_iri(self) -> str:
"""
Generate a new IRI for the FOLIO ontology.

NOTE: This is designed to approximate the WebProtege IRI generation algorithm.
The scheme is "R" followed by base62 of 127 random bits, which is what
the great majority of published FOLIO concepts already use. See
folio.iri for the derivation and for the note on the previous
base64url-derived scheme.

Returns:
str: The new IRI.
"""

for _ in range(MAX_IRI_ATTEMPTS):
# generate a new base uuid4 value
base_value = uuid.uuid4()

# only use alphanumeric characters from restricted b64 encdoding to
base64_value = "".join(
[
c
for c in base64.urlsafe_b64encode(base_value.bytes)
.decode("utf-8")
.rstrip("=")
if c.isalnum()
]
)

# ensure it's unique
if base64_value in self.iri_to_index:
continue

return f"https://folio.openlegalstandard.org/{base64_value}"

raise RuntimeError("Failed to generate a unique IRI.")
return generate_iri(self.iri_to_index, max_attempts=MAX_IRI_ATTEMPTS)
150 changes: 150 additions & 0 deletions folio/iri.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,150 @@
"""
Concept IRI generation for the FOLIO namespace.

FOLIO concept IRIs are permanent: once published, one is never deleted and never
reused, so minting is a one-way door. This module is the single place where a
new one is generated.

Convention
----------
A new local name is ``R`` followed by the base62 encoding of 127 random bits::

https://folio.openlegalstandard.org/R1cNH7TLMiSlSbIbdFsynUk

That is the scheme the great majority of published FOLIO concepts already use.
Measured over the 18,325 ``owl:Class`` declarations in ``FOLIO.owl`` at
``alea-institute/FOLIO@8ebf17b``, the local names fall into these families:

========================================= ====== ==========
family count share
========================================= ====== ==========
``R`` + base62 (this module) 11,427 62.4%
``R`` + 23 hex characters (SALI legacy) 4,751 25.9%
raw base64url of a uuid4, 22 characters 1,992 10.9%
``R`` + base64url 151 0.8%
other 4 0.02%
========================================= ====== ==========

The 127-bit width is not arbitrary; it is recoverable from the published data.
Local-name body lengths of 20/21/22 characters occur at 0.39%/25.75%/73.85%.
Base62 of a uniform 127-bit integer predicts 0.39%/25.31%/74.29%. A 128-bit
generator predicts 0.21%/12.9%/86.9% and is excluded. Changing the width would
silently change the length profile of future IRIs, which is the only evidence
of how FOLIO IRIs are generated -- so it is pinned and tested.

NOTE (for review): this replaces a base64url-derived scheme that emitted no
``R`` prefix. The intent is to continue the scheme the vast majority of FOLIO
items already use, so that a concept minted today is indistinguishable from its
siblings. If dropping the ``R`` prefix was deliberate rather than incidental,
say so and this should be reverted to match -- the two schemes should not both
be in use.

This module itself imports only the standard library, so minting no longer
requires constructing a ``FOLIO`` graph -- which downloads and parses the full
18 MB ontology purely to produce one 23-character string. (Importing it through
the package still initialises ``folio``, and therefore ``folio.graph``; making
``folio/__init__.py`` lazy would remove that too, but is out of scope here.)
"""

# SPDX-License-Identifier: MIT
# (c) 2024 ALEA Institute.

from __future__ import annotations

import secrets
from collections.abc import Callable, Container

# The FOLIO namespace that concept IRIs are minted under.
FOLIO_NAMESPACE: str = "https://folio.openlegalstandard.org/"

# Prefix carried by every minted local name.
IRI_PREFIX: str = "R"

# Base62 digits, least significant value first. Ordering matters: it is the
# ordering already present in published IRIs.
BASE62_ALPHABET: str = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz"

# Random bits per local name. Recovered from the published length distribution;
# see the module docstring before changing this.
IRI_ENTROPY_BITS: int = 127

# Safety bound on collision retries.
MAX_IRI_ATTEMPTS: int = 16


def encode_base62(value: int) -> str:
"""
Encode a non-negative integer in base62, most significant digit first.

Args:
value (int): The integer to encode.

Returns:
str: The base62 encoding.

Raises:
ValueError: If value is negative.
"""
if value < 0:
raise ValueError("base62 encoding requires a non-negative integer")
if value == 0:
return BASE62_ALPHABET[0]

digits: list[str] = []
while value:
value, remainder = divmod(value, 62)
digits.append(BASE62_ALPHABET[remainder])
return "".join(reversed(digits))


def generate_local_name(*, randbits: Callable[[int], int] = secrets.randbits) -> str:
"""
Generate one local name, without checking it against anything.

Args:
randbits (Callable[[int], int]): Source of randomness. Injectable for
tests; production callers should leave this as secrets.randbits.

Returns:
str: A local name such as "R1cNH7TLMiSlSbIbdFsynUk".
"""
return IRI_PREFIX + encode_base62(randbits(IRI_ENTROPY_BITS))


def generate_iri(
existing: Container[str] | None = None,
*,
max_attempts: int = MAX_IRI_ATTEMPTS,
randbits: Callable[[int], int] = secrets.randbits,
) -> str:
"""
Generate a new, unused FOLIO concept IRI.

Args:
existing (Optional[Container[str]]): Names already in use. Both full
IRIs and bare local names are recognised, so an index keyed either
way can be passed directly.
max_attempts (int): Collision retries before giving up.
randbits (Callable[[int], int]): Source of randomness.

Returns:
str: The new IRI.

Raises:
RuntimeError: If no unused IRI was found within max_attempts.
"""
for _ in range(max_attempts):
local_name = generate_local_name(randbits=randbits)
iri = f"{FOLIO_NAMESPACE}{local_name}"

# Check both spellings. The previous implementation tested a bare local
# name against FOLIO.iri_to_index, whose keys are full IRIs taken from
# rdf:about -- so the guard could never fire and the retry loop could
# never run. Accepting either form makes the check work regardless of
# how the caller's index is keyed.
if existing is not None and (iri in existing or local_name in existing):
continue

return iri

raise RuntimeError("Failed to generate a unique IRI.")
9 changes: 6 additions & 3 deletions tests/test_folio.py
Original file line number Diff line number Diff line change
Expand Up @@ -786,6 +786,9 @@ def test_iri_generation(folio_graph):
iri = folio_graph.generate_iri()
assert iri is not None
assert iri.startswith("https://folio.openlegalstandard.org/")
b64_token = iri.split("/")[-1]
assert b64_token.isalnum()
assert len(b64_token) > 16
token = iri.split("/")[-1]
assert token.isalnum()
assert len(token) > 16
# "R" + base62, matching the majority of published FOLIO concepts.
assert token.startswith("R")
assert iri not in folio_graph.iri_to_index
128 changes: 128 additions & 0 deletions tests/test_iri.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,128 @@
"""Tests for folio.iri, the concept IRI generator.

folio.iri imports only the standard library, so these run without network
access, without a FOLIO graph, and without the optional search extras.
"""

# SPDX-License-Identifier: MIT
# (c) 2024 ALEA Institute.

import re

import pytest

from folio.iri import (
BASE62_ALPHABET,
FOLIO_NAMESPACE,
IRI_ENTROPY_BITS,
IRI_PREFIX,
encode_base62,
generate_iri,
generate_local_name,
)

# "R" plus 19-22 base62 characters. 127 bits encodes to 22 digits 74% of the
# time and 21 digits 25% of the time; shorter outputs are rare but legitimate
# and one (R92xsHpdHu6BJjtepuRu) is already published.
LOCAL_NAME = re.compile(r"R[0-9A-Za-z]{19,22}\Z")


def decode_base62(text: str) -> int:
"""Independent inverse; Python's int() only parses up to base 36."""
value = 0
for character in text:
value = value * 62 + BASE62_ALPHABET.index(character)
return value


def test_base62_round_trips() -> None:
for value in (0, 1, 61, 62, 3843, 62**21 + 5, 2**127 - 1):
assert decode_base62(encode_base62(value)) == value


def test_base62_digit_order() -> None:
assert encode_base62(0) == "0"
assert encode_base62(10) == "A"
assert encode_base62(61) == "z"
assert encode_base62(62) == "10"


def test_base62_rejects_negative() -> None:
with pytest.raises(ValueError):
encode_base62(-1)


def test_generate_local_name_shape() -> None:
for _ in range(200):
local_name = generate_local_name()
assert LOCAL_NAME.fullmatch(local_name)
assert local_name.isalnum()
assert local_name.startswith(IRI_PREFIX)


def test_generate_iri_shape() -> None:
for _ in range(50):
iri = generate_iri()
assert iri.startswith(FOLIO_NAMESPACE)
local_name = iri[len(FOLIO_NAMESPACE) :]
assert LOCAL_NAME.fullmatch(local_name)
# The previous implementation could emit a local name of 16 characters
# or fewer roughly once in 29,000 mints, because it stripped
# non-alphanumeric characters after base64 encoding. Base62 has no
# such step.
assert len(local_name) > 16


def test_entropy_width_is_pinned() -> None:
"""Recovered from the published length distribution; see folio/iri.py."""
assert IRI_ENTROPY_BITS == 127


def test_generate_iri_is_deterministic_given_randbits() -> None:
value = 62**21 + 5
iri = generate_iri(randbits=lambda _bits: value)
assert iri == f"{FOLIO_NAMESPACE}R{encode_base62(value)}"


def test_collision_check_accepts_full_iris() -> None:
"""An index keyed by full IRI -- what FOLIO.iri_to_index actually holds."""
taken, free = 62**21 + 5, 62**21 + 7
taken_iri = f"{FOLIO_NAMESPACE}R{encode_base62(taken)}"
values = iter([taken, taken, free])
iri = generate_iri({taken_iri}, randbits=lambda _bits: next(values))
assert iri == f"{FOLIO_NAMESPACE}R{encode_base62(free)}"


def test_collision_check_accepts_bare_local_names() -> None:
"""An index keyed by bare local name is also honoured."""
taken, free = 62**21 + 5, 62**21 + 7
values = iter([taken, taken, free])
iri = generate_iri(
{f"R{encode_base62(taken)}"}, randbits=lambda _bits: next(values)
)
assert iri == f"{FOLIO_NAMESPACE}R{encode_base62(free)}"


def test_collision_retries_are_bounded() -> None:
taken = 62**21 + 5
taken_iri = f"{FOLIO_NAMESPACE}R{encode_base62(taken)}"
with pytest.raises(RuntimeError):
generate_iri({taken_iri}, max_attempts=4, randbits=lambda _bits: taken)


def test_generated_iris_are_distinct() -> None:
assert len({generate_iri() for _ in range(500)}) == 500


@pytest.mark.parametrize(
"published",
[
"RCzxVprwB3RwZ8EMewt18Jt", # Investment Funds
"R7e7pNl5IOMFbxKN2GV2C41", # Hedge Fund
"RSYBzf149Mi5KE0YtmpUmr", # Area of Law
"R92xsHpdHu6BJjtepuRu", # shortest published local name
],
)
def test_published_iris_match_the_generated_shape(published: str) -> None:
"""Minted names must be indistinguishable from ones already published."""
assert LOCAL_NAME.fullmatch(published)
Loading