Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/test.yml
Original file line number Diff line number Diff line change
Expand Up @@ -41,7 +41,7 @@ jobs:
fail-fast: false
matrix:
os: [ubuntu-latest, windows-latest, macos-latest]
python-version: ["3.11", "3.12", "3.13", "3.14"]
python-version: ["3.11", "3.12", "3.13", "3.14", "3.14t"]

steps:
- uses: actions/checkout@v7
Expand Down
2 changes: 2 additions & 0 deletions CHANGES.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,8 @@

## Unreleased

- Support free-threaded CPython 3.14 without re-enabling the GIL, including
dedicated tests, wheels, and before/after performance measurements
- Accelerate uchardet language-model lookups with per-detector code-point
caches while preserving candidate results
- Cache multibyte candidate analysis and eliminate duplicate known-language
Expand Down
8 changes: 6 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -11,8 +11,8 @@ cChardet is a high-speed universal character encoding detector built on the

## Python support

cChardet supports CPython 3.11 through 3.14.
Each version is tested on Linux, macOS, and Windows.
cChardet supports CPython 3.11 through 3.14, including the free-threaded
CPython 3.14 build. Each version is tested on Linux, macOS, and Windows.

## Development

Expand Down Expand Up @@ -209,6 +209,10 @@ uv run --no-sync python benchmarks/pyperf_compare.py \
--pythonpath ../chardet/src --pythonpath ../charset_normalizer/src \
--corpus src/ext/uchardet/test --rigorous -o benchmark-pure.json

# Compare regular and free-threaded CPython installations independently.
uv run --no-sync python benchmarks/pyperf_free_threading.py \
--corpus src/ext/uchardet/test --rigorous -o free-threading.json

# Compare encoding and language accuracy on separately reported corpora.
uv run --no-sync python benchmarks/accuracy.py \
--uchardet-corpus src/ext/uchardet/test \
Expand Down
91 changes: 91 additions & 0 deletions benchmarks/pyperf_free_threading.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,91 @@
"""Compare cChardet throughput with regular and free-threaded CPython.

Install cChardet into each interpreter before running this script. Corpus
files are content-deduplicated and loaded before pyperf starts timing.
"""

from __future__ import annotations

import hashlib
import sys
import sysconfig
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
from typing import Any

import pyperf

import cchardet


def _load_corpus(roots: list[Path]) -> list[bytes]:
paths = sorted(
{
path
for root in roots
for path in ([root] if root.is_file() else root.rglob("*"))
if path.is_file() and not any(part.startswith(".") for part in path.parts)
}
)
if not paths:
raise ValueError(f"no corpus files found under {roots}")
unique: dict[bytes, bytes] = {}
for path in paths:
data = path.read_bytes()
unique.setdefault(hashlib.sha256(data).digest(), data)
return list(unique.values())


def _detect_batch(batch: list[bytes]) -> int:
return sum(cchardet.detect(data)["encoding"] is not None for data in batch)


def _serial(corpus: list[bytes]) -> int:
return _detect_batch(corpus)


def _parallel(executor: ThreadPoolExecutor, batches: list[list[bytes]]) -> int:
return sum(executor.map(_detect_batch, batches))


def _add_worker_args(cmd: list[str], args: Any) -> None:
for root in args.corpus or ():
cmd.extend(("--corpus", str(root)))


def main() -> None:
runner = pyperf.Runner(add_cmdline_args=_add_worker_args)
runner.argparser.add_argument(
"--corpus",
action="append",
type=Path,
default=None,
help="input file or directory (repeatable)",
)
args = runner.parse_args()
roots = args.corpus or [Path("src/ext/uchardet/test"), Path("tests/samples")]
corpus = _load_corpus(roots)
gil_enabled = getattr(sys, "_is_gil_enabled", lambda: True)()
runner.metadata.update(
{
"cchardet_version": cchardet.__version__,
"cchardet_module": str(Path(cchardet.__file__).resolve()),
"free_threaded_build": bool(sysconfig.get_config_var("Py_GIL_DISABLED")),
"gil_enabled": gil_enabled,
"corpus": ":".join(str(root.resolve()) for root in roots),
"corpus_files": len(corpus),
"corpus_bytes": sum(map(len, corpus)),
}
)

_serial(corpus)
runner.bench_func("cchardet: serial corpus", _serial, corpus)

batches = [corpus[offset::4] for offset in range(4)]
with ThreadPoolExecutor(max_workers=4) as executor:
_parallel(executor, batches)
runner.bench_func("cchardet: 4-thread corpus", _parallel, executor, batches)


if __name__ == "__main__":
main()
21 changes: 21 additions & 0 deletions docs/performance-analysis.md
Original file line number Diff line number Diff line change
Expand Up @@ -142,6 +142,27 @@ ordering, encodings, languages, and confidence values for whole-file and
64-byte input. A further differential check covered 100 deterministic-size
random byte strings at chunk sizes 1, 7, 64, and 1,024 bytes.

### Free-threaded CPython

`pyperf_free_threading.py` measures cChardet alone so that interpreter and
extension-module behavior are not mixed with the availability of native
accelerators in other packages. The same CPython 3.14.2 free-threaded build
and preloaded 160-file, 99,819-byte uchardet corpus were used before and after
declaring the Cython module free-threading compatible:

| CPython 3.14t | Before | After | Change |
|---|---:|---:|---:|
| Serial corpus | 30.6 ms | 30.6 ms | No significant change |
| 4-thread corpus | 9.55 ms | 9.25 ms | 1.03x faster |

Before the change, importing cChardet emitted a runtime warning and enabled
the GIL for the process. Afterwards it leaves the GIL disabled. On regular
CPython 3.14.2, serial throughput changed from 30.2 to 30.3 ms and four-thread
throughput from 9.39 to 9.38 ms; pyperf found no significant four-thread
difference. The existing Cython `nogil` regions remain intentional: they keep
native detection independent of Python thread state and preserve parallel
native work on regular CPython as well as free-threaded builds.

## Accuracy benchmark

Performance without accuracy is misleading. `accuracy.py` applies the same
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -66,7 +66,7 @@ format.quote-style = "double"
format.indent-style = "space"

[tool.cibuildwheel]
build = "cp311-* cp312-* cp313-* cp314-*"
build = "cp311-* cp312-* cp313-* cp314-* cp314t-*"
skip = "*-musllinux_*"
archs = "auto"
test-command = "python tools/wheel_smoke.py"
Expand Down
8 changes: 8 additions & 0 deletions setup.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@
import glob
import os
import sys
import sysconfig

from setuptools import Extension, setup

Expand Down Expand Up @@ -82,6 +83,12 @@
else:
extra_compile_args = ["-std=c++11"]

# The Windows headers are shared by regular and free-threaded CPython, so the
# build must provide the configuration macro explicitly.
define_macros = []
if sys.platform == "win32" and sysconfig.get_config_var("Py_GIL_DISABLED"):
define_macros.append(("Py_GIL_DISABLED", "1"))

setup(
package_dir={"": "src"},
packages=[
Expand All @@ -94,6 +101,7 @@
sources=sources,
include_dirs=[uchardet_dir],
language="c++",
define_macros=define_macros,
extra_compile_args=extra_compile_args,
)
],
Expand Down
1 change: 1 addition & 0 deletions src/cchardet/_cchardet.pyx
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
# coding: utf-8
#cython: embedsignature=True, c_string_encoding=ascii, language_level=3
#cython: freethreading_compatible=True

from libc.stddef cimport size_t

Expand Down
18 changes: 15 additions & 3 deletions tests/test_1.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@
import glob
import os
import sys
import sysconfig
from concurrent.futures import ThreadPoolExecutor

import cchardet
Expand Down Expand Up @@ -30,6 +32,10 @@


class TestCChardet:
def test_free_threaded_build_keeps_gil_disabled(self):
if sysconfig.get_config_var("Py_GIL_DISABLED"):
assert not getattr(sys, "_is_gil_enabled")()

def test_ascii(self):
detected_encoding = cchardet.detect(b"abcdefghijklmnopqrstuvwxyz")
got_enc = None
Expand Down Expand Up @@ -118,9 +124,15 @@ def test_detector_exposes_uchardet_early_completion(self):

def test_detect_is_thread_safe(self):
samples = [b"plain ASCII", "日本語".encode(), "français".encode()]
with ThreadPoolExecutor(max_workers=4) as executor:
results = list(executor.map(cchardet.detect, samples * 100))
assert all(result["encoding"] is not None for result in results)

def detect_results(sample):
return cchardet.detect(sample), cchardet.detect_all(sample)

expected = [detect_results(sample) for sample in samples]
repeated_samples = samples * 100
with ThreadPoolExecutor(max_workers=8) as executor:
results = list(executor.map(detect_results, repeated_samples))
assert results == expected * 100

def test_detect_max_bytes(self):
data = b"plain ASCII" + "日本語".encode()
Expand Down
4 changes: 4 additions & 0 deletions tools/wheel_smoke.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,13 +3,17 @@
import json
import subprocess
import sys
import sysconfig
import tempfile
from pathlib import Path

import cchardet


def main() -> None:
if sysconfig.get_config_var("Py_GIL_DISABLED"):
assert not getattr(sys, "_is_gil_enabled")()

package_dir = Path(cchardet.__file__).parent
assert (package_dir / "py.typed").is_file()
assert (package_dir / "_cchardet.pyi").is_file()
Expand Down
Loading