From d90871081487c127f8f9679a79b05f693486d416 Mon Sep 17 00:00:00 2001 From: ZhuchkaTriplesix Date: Mon, 28 Sep 2026 13:50:49 +0300 Subject: [PATCH 1/2] perf(api): skip GIL release for HMAC encode/decode encode, encode_json, decode and decode_complete always called py.detach, releasing and reacquiring the GIL around every native call. For HS256/384/512 the sign/verify itself takes roughly a microsecond, so under thread contention the release-and-reacquire cycle can cost as much as the operation itself (or more, from futex/scheduler overhead on the contended lock), for negligible concurrency benefit on a call that short. Add maybe_detach, which releases the GIL unless told to skip it, and skip it specifically for HMAC: on encode, the resolved algorithm is already known; on decode/decode_complete, checked against the caller's algorithms allow-list before the token header is parsed (an HMAC-only allow-list already guarantees the verified algorithm is HMAC too, since a mixed-family allow-list is rejected). RSA, EC and EdDSA keep releasing the GIL unconditionally. Measured (release build): - Single-threaded HS256 decode: ~1.18us -> ~1.15us. - 8 threads continuously decoding the same HS256 token: ~410k -> ~790k decodes/sec (~1.9x more). The GIL release/reacquire cycle was itself a source of contention overhead under load, not just single-call cost. - RS256 decode (unaffected path) still scales with threads: ~67k/sec on 1 thread, ~339k/sec on 8, confirming detach is still exercised for non-HMAC algorithms. No behavioural change. Safe on free-threaded Python 3.13t/3.14t: the HMAC path never calls back into Python, so holding the GIL/attachment throughout is never a correctness hazard, only a choice not to release it. Closes #124 --- CHANGELOG.md | 18 +++++++ rust/src/api.rs | 127 ++++++++++++++++++++++++++++++++---------------- 2 files changed, 102 insertions(+), 43 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5fd05b9..f274951 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -57,6 +57,24 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 uninterned `PyString`, unchanged from before. Measured on an 8-claim payload: native `decode` dropped from ~2.05 µs to ~1.92 µs (~6% less). No behavioural change. (#123) +- **HMAC `encode`/`decode` no longer release the GIL.** `encode`, + `encode_json`, `decode` and `decode_complete` used to call `py.detach` + unconditionally, releasing and reacquiring the GIL around every native + call. For HS256/384/512 the signing/verification itself takes roughly a + microsecond, so under thread contention the release-and-reacquire cycle + cost as much as the operation, or more: measured with 8 threads + continuously decoding the same HS256 token, throughput went from ~410k to + ~790k decodes/sec (~1.9× more) once the release was skipped, and + single-threaded decode dropped from ~1.18 µs to ~1.15 µs. RSA, EC and + EdDSA are unaffected — the check is on the resolved algorithm (`encode`) + or on the caller's allow-list (`decode`, checked before the algorithm in + the token header is known: an HMAC-only allow-list already guarantees the + verified algorithm is HMAC too), and those algorithms still release the + GIL, confirmed to keep scaling with threads (RS256 decode: ~67k/sec on 1 + thread, ~339k/sec on 8). No behavioural change; safe on free-threaded + Python 3.13t/3.14t since the HMAC path never calls back into Python + either way, so holding the GIL throughout is never a hazard, only a + choice not to release it. (#124) ## [0.7.0] — 2026-08-26 diff --git a/rust/src/api.rs b/rust/src/api.rs index 0615115..c1cfd5b 100644 --- a/rust/src/api.rs +++ b/rust/src/api.rs @@ -9,7 +9,8 @@ use pyo3::types::PyBytes; use serde_json::Value; use crate::algorithms::{ - algorithm_name, ensure_single_family, parse_algorithm, parse_algorithm_name, + algorithm_family, algorithm_name, ensure_single_family, parse_algorithm, parse_algorithm_name, + KeyFamily, }; use crate::claims::{json_to_py, py_to_json_for_encode}; use crate::claims_validate; @@ -133,6 +134,36 @@ fn ensure_single_family_for_decode( .map_err(|_| decode_fail(ErrorKind::InvalidAlgorithm)) } +/// True when every algorithm in the allow-list is HMAC. Checked against the +/// caller's allow-list rather than the algorithm actually used, because for +/// `decode` the latter is only known after parsing the token header, which +/// happens inside the (possibly GIL-attached) verification step itself; an +/// HMAC-only allow-list already guarantees the verified algorithm is HMAC +/// too (`ensure_single_family_for_decode` rejects a mixed-family list). +fn algorithms_are_all_hmac(algorithms: &[jsonwebtoken::Algorithm]) -> bool { + algorithms + .iter() + .all(|algorithm| algorithm_family(*algorithm) == KeyFamily::Hmac) +} + +/// Releases the GIL around `f` unless `skip_detach` is set, in which case `f` +/// runs while still attached. HMAC sign/verify is fast enough (roughly a +/// microsecond) that `py.detach`'s release-and-reacquire can cost as much as +/// the operation itself, for negligible concurrency benefit on that single +/// call; RSA/EC/EdDSA operations are one to several orders of magnitude +/// slower and keep releasing the GIL unconditionally. See #124. +fn maybe_detach(py: Python<'_>, skip_detach: bool, f: F) -> T +where + F: pyo3::marker::Ungil + FnOnce() -> T, + T: pyo3::marker::Ungil, +{ + if skip_detach { + f() + } else { + py.detach(f) + } +} + /// The `aws_lc_rs` padding/digest scheme for an RSA/RSA-PSS algorithm. #[cfg(feature = "aws_lc_rs")] fn rsa_padding_for(algorithm: jsonwebtoken::Algorithm) -> &'static dyn RsaEncoding { @@ -216,9 +247,12 @@ pub fn encode( } let encoding_key = encoding_key_from_py(key, algorithm)?; + let skip_detach = algorithm_family(algorithm) == KeyFamily::Hmac; - py.detach(|| jwt_encode(&header, &claims, &encoding_key)) - .map_err(errors::from_jwt_encode_error) + maybe_detach(py, skip_detach, || { + jwt_encode(&header, &claims, &encoding_key) + }) + .map_err(errors::from_jwt_encode_error) } #[pyfunction] @@ -251,10 +285,12 @@ pub fn decode( algorithms, audience, issuer, subject, leeway, options, require, )?; let decoding_key = decoding_key_from_py(key, decode_validation.algorithms())?; + let skip_detach = algorithms_are_all_hmac(decode_validation.algorithms()); - let verified = py - .detach(|| verify_and_parse(token, &decoding_key, &decode_validation)) - .map_err(map_decode_fail)?; + let verified = maybe_detach(py, skip_detach, || { + verify_and_parse(token, &decoding_key, &decode_validation) + }) + .map_err(map_decode_fail)?; json_to_py(py, &verified.claims) } @@ -306,13 +342,17 @@ pub fn decode_verified_complete( ); } - let (verified, signature) = py - .detach(|| -> Result<(VerifiedToken, Vec), DecodeFail> { + let skip_detach = algorithms_are_all_hmac(decode_validation.algorithms()); + let (verified, signature) = maybe_detach( + py, + skip_detach, + || -> Result<(VerifiedToken, Vec), DecodeFail> { let verified = verify_and_parse(token, &decoding_key, &decode_validation)?; let signature = jws::extract_signature_bytes(token).map_err(DecodeFail::Decode)?; Ok((verified, signature)) - }) - .map_err(map_decode_fail)?; + }, + ) + .map_err(map_decode_fail)?; let claims_py = json_to_py(py, &verified.claims)?; let header_py = json_to_py(py, &verified.header)?; @@ -334,41 +374,41 @@ fn decode_rfc7797_verified_complete( ))); } - let (parts, claims, signature) = py - .detach(|| -> Result<_, DecodeFail> { - let parts = jws::parse_rfc7797_compact(token).map_err(DecodeFail::Token)?; - let claims: Value = serde_json::from_slice(detached_payload) - .map_err(|err| DecodeFail::Decode(format!("Invalid payload string: {err}")))?; - if !claims.is_object() { - return Err(DecodeFail::Decode( - "Invalid payload string: must be a json object".to_owned(), - )); - } + let skip_detach = algorithms_are_all_hmac(decode_validation.algorithms()); + let (parts, claims, signature) = maybe_detach(py, skip_detach, || -> Result<_, DecodeFail> { + let parts = jws::parse_rfc7797_compact(token).map_err(DecodeFail::Token)?; + let claims: Value = serde_json::from_slice(detached_payload) + .map_err(|err| DecodeFail::Decode(format!("Invalid payload string: {err}")))?; + if !claims.is_object() { + return Err(DecodeFail::Decode( + "Invalid payload string: must be a json object".to_owned(), + )); + } - let algorithm = header_algorithm(&parts.header)?; - if !decode_validation.algorithms().contains(&algorithm) { - return Err(decode_fail(ErrorKind::InvalidAlgorithm)); - } - ensure_single_family_for_decode(decode_validation.algorithms())?; - - let signing_input = jws::signing_input_rfc7797(&parts.header_segment, detached_payload); - if !jwt_crypto_verify( - &parts.signature_segment, - &signing_input, - decoding_key, - algorithm, - )? { - return Err(decode_fail(ErrorKind::InvalidSignature)); - } + let algorithm = header_algorithm(&parts.header)?; + if !decode_validation.algorithms().contains(&algorithm) { + return Err(decode_fail(ErrorKind::InvalidAlgorithm)); + } + ensure_single_family_for_decode(decode_validation.algorithms())?; + + let signing_input = jws::signing_input_rfc7797(&parts.header_segment, detached_payload); + if !jwt_crypto_verify( + &parts.signature_segment, + &signing_input, + decoding_key, + algorithm, + )? { + return Err(decode_fail(ErrorKind::InvalidSignature)); + } - claims_validate::validate_claims_value(&claims, &decode_validation.validation)?; + claims_validate::validate_claims_value(&claims, &decode_validation.validation)?; - let signature = URL_SAFE_NO_PAD - .decode(&parts.signature_segment) - .map_err(|err| DecodeFail::Decode(err.to_string()))?; - Ok((parts, claims, signature)) - }) - .map_err(map_decode_fail)?; + let signature = URL_SAFE_NO_PAD + .decode(&parts.signature_segment) + .map_err(|err| DecodeFail::Decode(err.to_string()))?; + Ok((parts, claims, signature)) + }) + .map_err(map_decode_fail)?; let header_py = json_to_py(py, &parts.header)?; let claims_py = json_to_py(py, &claims)?; @@ -523,8 +563,9 @@ pub fn encode_json( } let encoding_key = encoding_key_from_py(key, algorithm)?; + let skip_detach = algorithm_family(algorithm) == KeyFamily::Hmac; - py.detach(move || { + maybe_detach(py, skip_detach, move || { use base64::engine::general_purpose::URL_SAFE_NO_PAD; use base64::Engine; From e23d6fb046712b798b7fd5fa936db95907b74bcd Mon Sep 17 00:00:00 2001 From: ZhuchkaTriplesix Date: Mon, 28 Sep 2026 13:50:49 +0300 Subject: [PATCH 2/2] test(encode): cover HMAC encode/decode concurrency Regression test for the previous commit: with the GIL no longer released around HS256 encode/decode, many threads round-tripping their own payload through the same secret must still get back exactly what they encoded, with no cross-talk between threads. --- tests/test_encode_decode.py | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/tests/test_encode_decode.py b/tests/test_encode_decode.py index 72896ca..4ca403d 100644 --- a/tests/test_encode_decode.py +++ b/tests/test_encode_decode.py @@ -1,6 +1,7 @@ from __future__ import annotations import time +from concurrent.futures import ThreadPoolExecutor from datetime import datetime, timezone import orjson @@ -215,3 +216,26 @@ def test_custom_claim_key_is_not_interned() -> None: assert key is not custom_key +def test_hmac_encode_decode_are_thread_safe_without_gil_release() -> None: + """HS256 `encode`/`decode` no longer release the GIL around the native + call (see #124). That is purely a scheduling change on the Rust side, but + it is worth a concurrency regression test in its own right: each thread + must still get back exactly the token/payload it asked for, with no + cross-talk between threads sharing the same secret. + """ + secret = "concurrent-hmac-secret-with-plenty-of-length" + + def roundtrip(i: int) -> bool: + payload = {"sub": f"user-{i}", "n": i} + for _ in range(200): + token = oxyjwt.encode(payload, secret, "HS256") + decoded = oxyjwt.decode(token, secret, algorithms=["HS256"]) + if decoded != payload: + return False + return True + + with ThreadPoolExecutor(max_workers=8) as pool: + results = list(pool.map(roundtrip, range(32))) + assert all(results) + +