Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
22 commits
Select commit Hold shift + click to select a range
669f98d
fix: use UTF-16 offsets for Text operations (fixes #308)
xrl Apr 10, 2026
f3a2e7e
test: add 12 Unicode/emoji tests for Text operations
xrl Apr 10, 2026
8264d95
test: add granular diff tests from jupyter_ydoc#370
xrl Apr 10, 2026
7cf5af4
fix: move SequenceMatcher import to top of file (ruff E402)
xrl Apr 10, 2026
b1ed6ae
test: add tests for _utf16_to_char helper
xrl Apr 14, 2026
f48610b
fix: convert UTF-16 offset in Text.__iadd__ and drop unused _utf16_to…
xrl Apr 17, 2026
01a1756
[pre-commit.ci] auto fixes from pre-commit.com hooks
pre-commit-ci[bot] Apr 17, 2026
41c8cc5
feat: parameterize Doc offset_kind (default UTF-8)
xrl Apr 28, 2026
855af5c
refactor: make utf16/utf8 index helpers public
xrl Apr 28, 2026
e41d345
test: parametrize Unicode tests over offset_kind
xrl Apr 28, 2026
b6b938c
[pre-commit.ci] auto fixes from pre-commit.com hooks
pre-commit-ci[bot] Apr 28, 2026
3e9933a
fix: update _pycrdt.pyi stub for offset_kind
xrl Apr 28, 2026
67b2dea
test: cover the offset_kind/doc mismatch ValueError
xrl Apr 28, 2026
430dab8
Merge remote-tracking branch 'upstream/main' into 308-fix-utf16-offse…
xrl Jun 12, 2026
87bb6d1
fix: give index-conversion helpers Python slice-bound semantics
xrl Jun 12, 2026
5e1df54
fix: restrict offset_kind to 'utf8'/'utf16' and type it as a Literal
xrl Jun 12, 2026
63af1e2
perf: avoid full-text materialization in __iadd__ and stop=None slices
xrl Jun 12, 2026
9a73db6
test: tighten Unicode test cases
xrl Jun 12, 2026
5566062
docs: document Text character-index semantics and offset_kind accurately
xrl Jun 12, 2026
3207533
refactor: rename 'ok' locals to 'offset_kind'
xrl Jun 12, 2026
7158e04
Merge upstream main into 308-fix-utf16-offset-encoding
xrl Sep 18, 2026
1f37640
fix: account for embeds in Text character offsets
xrl Sep 19, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions docs/api_reference.md
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,8 @@
- handle_sync_message
- get_state
- get_update
- get_utf8_index
- get_utf16_index
- merge_updates
- read_message
- write_message
Expand Down
26 changes: 26 additions & 0 deletions docs/usage.md
Original file line number Diff line number Diff line change
Expand Up @@ -327,6 +327,32 @@ print(str(text))

Undoing a change doesn't remove the change from the document's history, but applies a change that is the opposite of the previous change.

## Text indices and Unicode

`Text` indices count each Python character (code point) and each embedded object as one position:

```py
from pycrdt import Doc, Text

doc = Doc()
doc["text"] = text = Text("A📊B")
text.insert(2, "X") # character index, even though 📊 is 4 UTF-8 bytes
print(str(text))
# prints: "A📊XB"
```

Internally, yrs counts text positions in the units selected by the document's `offset_kind`, and pycrdt converts character indices to those units. The default is `"utf8"` (byte offsets, the yrs default). Passing `offset_kind="utf16"` makes yrs count in UTF-16 code units, which matches the index semantics of JS [yjs](https://github.com/yjs/yjs) (JavaScript strings are UTF-16). The setting doesn't affect the update wire format — documents with different offset kinds stay in sync — but it matters when raw yrs offsets are shared with yjs peers, for instance through sticky indices or event deltas. The conversion helpers [get_utf8_index][pycrdt.get_utf8_index] and [get_utf16_index][pycrdt.get_utf16_index] are available for code that needs to do the same conversion.

Embeds count toward `len(text)`, indexing, iteration, membership, and mutation
positions. Indexed reads and iteration use U+FFFC for embeds; `diff()` returns their
values, while `str()` and `to_py()` omit them. Use `len(text)`, not `len(str(text))`,
to insert at the end. The string conversion helpers do not account for embeds.

Yrs' delta API cannot distinguish nonempty string-valued embeds from text runs.
Character-indexed operations on documents containing them raise `ValueError` before
editing. Use a structured embed such as `{"value": "label"}` instead. `str()`,
`diff()`, `+=`, and `clear()` still work on those documents.

## Type annotations

`Array`, `Map` and `Doc` can be type-annotated for static type analysis. For instance, here is how to declare a `Doc` where all root types are `Array`s of `int`s:
Expand Down
2 changes: 2 additions & 0 deletions python/pycrdt/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,8 @@
from ._sync import write_var_uint as write_var_uint
from ._text import Text as Text
from ._text import TextEvent as TextEvent
from ._text import get_utf8_index as get_utf8_index
from ._text import get_utf16_index as get_utf16_index
from ._transaction import NewTransaction as NewTransaction
from ._transaction import ReadTransaction as ReadTransaction
from ._transaction import Transaction as Transaction
Expand Down
7 changes: 6 additions & 1 deletion python/pycrdt/_base.py
Original file line number Diff line number Diff line change
Expand Up @@ -62,14 +62,19 @@ def __init__(
client_id: int | None = None,
skip_gc: bool | None = None,
guid: str | None = None,
offset_kind: Literal["utf8", "utf16"] | None = None,
doc: _Doc | None = None,
Model=None,
allow_multithreading: bool = False,
**data,
) -> None:
super().__init__(**data)
if doc is None:
doc = _Doc(client_id, skip_gc, guid)
doc = _Doc(client_id, skip_gc, guid, offset_kind)
elif offset_kind is not None and offset_kind != doc.offset_kind:
raise ValueError(
f"offset_kind={offset_kind!r} does not match doc.offset_kind={doc.offset_kind!r}"
)
self._doc = doc
self._txn = None
self._exceptions = []
Expand Down
22 changes: 22 additions & 0 deletions python/pycrdt/_doc.py
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,7 @@ def __init__(
client_id: int | None = None,
skip_gc: bool | None = None,
guid: str | None = None,
offset_kind: Literal["utf8", "utf16"] | None = None,
doc: _Doc | None = None,
Model=None,
allow_multithreading: bool = False,
Expand All @@ -67,12 +68,24 @@ def __init__(
skip_gc: Whether to skip garbage collection on deleted collections
on transaction commit.
guid: An optional globally unique identifier for the document.
offset_kind: How yrs counts text positions internally. ``"utf8"``
(the yrs default) uses byte offsets; ``"utf16"`` uses UTF-16
code unit offsets, matching the index semantics of JS yjs.
``None`` (default) selects the yrs default of ``"utf8"``.
The setting doesn't affect the update wire format, but it
matters when raw yrs offsets are shared with yjs peers (e.g.
sticky indices, event deltas). It applies to this document
only: a subdocument carries the offset kind chosen by the
peer that created it. Regardless of this setting, the public
``Text`` API counts each Python character and embedded object
as one position; ``str()`` omits embedded objects.
allow_multithreading: Whether to allow the document to be used in different threads.
"""
super().__init__(
client_id=client_id,
skip_gc=skip_gc,
guid=guid,
offset_kind=offset_kind,
doc=doc,
Model=Model,
allow_multithreading=allow_multithreading,
Expand All @@ -96,6 +109,15 @@ def client_id(self) -> int:
"""The document client ID."""
return self._doc.client_id()

@property
def offset_kind(self) -> Literal["utf8", "utf16"]:
"""The text offset kind used internally by yrs.

Returns ``"utf8"`` or ``"utf16"``. See [Doc.__init__][pycrdt.Doc.__init__]
for the meaning.
"""
return self._doc.offset_kind

def transaction(self, origin: Any = None) -> Transaction:
"""
Creates a new transaction or gets the current one, if any.
Expand Down
14 changes: 12 additions & 2 deletions python/pycrdt/_pycrdt.pyi
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
from typing import Any, Callable, Generic, Iterator, TypeVar
from typing import Any, Callable, Generic, Iterator, Literal, TypeVar

class Snapshot:
"""A snapshot of a document's state at a given point in time."""
Expand All @@ -17,7 +17,13 @@ class Snapshot:
class Doc:
"""Shared document."""

def __init__(self, client_id: int | None, skip_gc: bool | None, guid: str | None) -> None:
def __init__(
self,
client_id: int | None,
skip_gc: bool | None,
guid: str | None,
offset_kind: Literal["utf8", "utf16"] | None,
) -> None:
"""Create a new document with an optional global client ID.
If no client ID is passed, a random one will be generated."""

Expand All @@ -28,6 +34,10 @@ class Doc:
def client_id(self) -> int:
"""Returns the document unique client identifier."""

@property
def offset_kind(self) -> Literal["utf8", "utf16"]:
"""Returns the offset kind ('utf8' or 'utf16')."""

def guid(self) -> str:
"""Returns the document globally unique identifier."""

Expand Down
Loading
Loading