Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions AUTHORS.rst
Original file line number Diff line number Diff line change
Expand Up @@ -13,3 +13,4 @@ Authors
* Nathan McDougall - https://github.com/nathanjmcdougall
* Oleksandr Zaiats - https://github.com/z4y4ts
* Nikhil Dabas - https://github.com/ndabas
* Pierre-Yves Le Borgne - https://github.com/pylaterreur
3 changes: 3 additions & 0 deletions CHANGELOG.rst
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,9 @@ latest

* Fix missing macOS wheels for regular (non-freethreaded) Python 3.14+
(https://github.com/python-grimp/grimp/issues/317).
* Fix panic when a module declares its encoding using a name that Python accepts but that isn't a
WHATWG label, such as ``latin-1`` or ``utf_8``. Raise a ``UnicodeError`` naming the file, rather
than panicking, if a module can't be decoded.

3.17 (2026-09-04)
-----------------
Expand Down
11 changes: 10 additions & 1 deletion rust/src/errors.rs
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
use crate::exceptions;
use pyo3::PyErr;
use pyo3::exceptions::PyValueError;
use pyo3::exceptions::{PyFileNotFoundError, PyUnicodeError, PyValueError};
use ruff_python_parser::ParseError as RuffParseError;
use thiserror::Error;

Expand Down Expand Up @@ -39,6 +39,12 @@ pub enum GrimpError {

#[error("Cache file {0} was written by a different version of Grimp.")]
CacheVersionMismatch(String),

#[error("{0}")]
FileNotFound(String),

#[error("{0}")]
UndecodableFile(String),
}

pub type GrimpResult<T> = Result<T, GrimpError>;
Expand All @@ -61,6 +67,9 @@ impl From<GrimpError> for PyErr {
GrimpError::CacheVersionMismatch(_) => {
exceptions::CacheVersionMismatch::new_err(value.to_string())
}
GrimpError::FileNotFound(_) => PyFileNotFoundError::new_err(value.to_string()),
// Not UnicodeDecodeError, as that can't be created from just a message.
GrimpError::UndecodableFile(_) => PyUnicodeError::new_err(value.to_string()),
}
}
}
59 changes: 43 additions & 16 deletions rust/src/filesystem.rs
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
use crate::errors::{GrimpError, GrimpResult};
use itertools::Itertools;
use pyo3::exceptions::{PyFileNotFoundError, PyTypeError, PyUnicodeDecodeError};
use pyo3::exceptions::PyTypeError;
use pyo3::prelude::*;
use regex::Regex;
use std::collections::HashMap;
Expand All @@ -14,6 +15,33 @@ use unindent::unindent;
static ENCODING_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"^[ \t\f]*#.*?coding[:=][ \t]*([-_.a-zA-Z0-9]+)").unwrap());

/// Look up the encoding named in a Python source file's encoding declaration.
///
/// encoding_rs only knows the WHATWG labels, but Python accepts other spellings too, such as
/// `latin-1`, `utf_8`, `utf-8-sig` or `euc_jp`.
fn lookup_encoding(name: &str) -> Option<&'static encoding_rs::Encoding> {
encoding_rs::Encoding::for_label(name.as_bytes()).or_else(|| {
// Normalize the name like Python does: see `_get_normal_name` in Lib/tokenize.py, which
// special cases UTF-8 and Latin-1. Python's codec lookup also ignores the difference
// between underscores and hyphens.
let normalized = name.to_ascii_lowercase().replace('_', "-");
let is_spelling_of = |canonical: &str| {
normalized == canonical || normalized.starts_with(&format!("{canonical}-"))
};
let label = if is_spelling_of("utf-8") {
"utf-8"
} else if ["latin-1", "iso-8859-1", "iso-latin-1"]
.into_iter()
.any(is_spelling_of)
{
"iso-8859-1"
} else {
&normalized
};
encoding_rs::Encoding::for_label(label.as_bytes())
})
}

pub trait FileSystem: Send + Sync {
fn sep(&self) -> &str;

Expand All @@ -23,7 +51,7 @@ pub trait FileSystem: Send + Sync {

fn exists(&self, file_name: &str) -> bool;

fn read(&self, file_name: &str) -> PyResult<String>;
fn read(&self, file_name: &str) -> GrimpResult<String>;

fn write(&mut self, file_name: &str, contents: &str) -> PyResult<()>;
}
Expand Down Expand Up @@ -83,7 +111,7 @@ impl FileSystem for RealBasicFileSystem {
Path::new(file_name).is_file()
}

fn read(&self, file_name: &str) -> PyResult<String> {
fn read(&self, file_name: &str) -> GrimpResult<String> {
// Python files are assumed UTF-8 by default (PEP 686), but they can specify an alternative
// encoding, which we need to take into account here.
// See https://peps.python.org/pep-0263/
Expand All @@ -92,7 +120,7 @@ impl FileSystem for RealBasicFileSystem {

let path = Path::new(file_name);
let bytes = fs::read(path).map_err(|e| {
PyFileNotFoundError::new_err(format!("Failed to read file {file_name}: {e}"))
GrimpError::FileNotFound(format!("Failed to read file {file_name}: {e}"))
})?;

let s = String::from_utf8_lossy(&bytes);
Expand All @@ -110,15 +138,14 @@ impl FileSystem for RealBasicFileSystem {
}

if let Some(enc_name) = detected_encoding {
let encoding =
encoding_rs::Encoding::for_label(enc_name.as_bytes()).ok_or_else(|| {
PyUnicodeDecodeError::new_err(format!(
"Failed to decode file {file_name} (unknown encoding '{enc_name}')"
))
})?;
let encoding = lookup_encoding(&enc_name).ok_or_else(|| {
GrimpError::UndecodableFile(format!(
"Failed to decode file {file_name} (unknown encoding '{enc_name}')"
))
})?;
let (decoded_s, _, had_errors) = encoding.decode(&bytes);
if had_errors {
Err(PyUnicodeDecodeError::new_err(format!(
Err(GrimpError::UndecodableFile(format!(
"Failed to decode file {file_name} with encoding '{enc_name}'"
)))
} else {
Expand All @@ -127,7 +154,7 @@ impl FileSystem for RealBasicFileSystem {
} else {
// Default to UTF-8 if no encoding is specified
String::from_utf8(bytes).map_err(|e| {
PyUnicodeDecodeError::new_err(format!(
GrimpError::UndecodableFile(format!(
"Failed to decode file {file_name} as UTF-8: {e}"
))
})
Expand Down Expand Up @@ -173,7 +200,7 @@ impl PyRealBasicFileSystem {
}

fn read(&self, file_name: &str) -> PyResult<String> {
self.inner.read(file_name)
Ok(self.inner.read(file_name)?)
}

fn write(&mut self, file_name: &str, contents: &str) -> PyResult<()> {
Expand Down Expand Up @@ -253,11 +280,11 @@ impl FileSystem for FakeBasicFileSystem {
self.contents.lock().unwrap().contains_key(file_name)
}

fn read(&self, file_name: &str) -> PyResult<String> {
fn read(&self, file_name: &str) -> GrimpResult<String> {
let contents = self.contents.lock().unwrap();
match contents.get(file_name) {
Some(file_contents) => Ok(file_contents.clone()),
None => Err(PyFileNotFoundError::new_err(format!(
None => Err(GrimpError::FileNotFound(format!(
"No such file: {file_name}"
))),
}
Expand Down Expand Up @@ -301,7 +328,7 @@ impl PyFakeBasicFileSystem {
}

fn read(&self, file_name: &str) -> PyResult<String> {
self.inner.read(file_name)
Ok(self.inner.read(file_name)?)
}

fn write(&mut self, file_name: &str, contents: &str) -> PyResult<()> {
Expand Down
2 changes: 1 addition & 1 deletion rust/src/import_scanning.rs
Original file line number Diff line number Diff line change
Expand Up @@ -130,7 +130,7 @@ fn scan_for_imports_no_py_single_module(
let found_package_for_module = found_packages_by_module[module];
let module_filename =
_determine_module_filename(module, found_package_for_module, file_system).unwrap();
let module_contents = file_system.read(&module_filename).unwrap();
let module_contents = file_system.read(&module_filename)?;
let imported_objects =
import_parsing::parse_imports_from_code(&module_contents, &module_filename)?;

Expand Down
51 changes: 51 additions & 0 deletions tests/functional/test_encoding_handling.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,5 @@
import pytest

import grimp


Expand Down Expand Up @@ -41,3 +43,52 @@ def test_build_graph_of_non_utf8_source():
"line_contents": "from .imported import 蟺",
},
] == result


@pytest.mark.parametrize(
"declared_encoding, codec, imported_name",
(
# Python special cases these spellings of UTF-8 and Latin-1.
("utf_8", "utf-8", "蟺"),
("UTF-8-sig", "utf-8", "蟺"),
("latin-1", "latin-1", "jalape帽o"),
("latin_1", "latin-1", "jalape帽o"),
("iso_8859_1", "latin-1", "jalape帽o"),
# Python also accepts underscores in place of hyphens in other encoding names.
("euc_jp", "euc_jp", "銉┿兗銉°兂"),
("iso8859_15", "iso8859_15", "jalape帽o"),
),
)
def test_build_graph_of_source_declaring_encoding_with_python_specific_name(
tmp_path, monkeypatch, declared_encoding, codec, imported_name
):
"""
Tests we can cope with source files that declare their encoding using a name that Python
accepts, but that isn't a WHATWG encoding label.
"""
package_directory = tmp_path / "declaredencodingpackage"
package_directory.mkdir()
(package_directory / "__init__.py").write_text("")
(package_directory / "imported.py").write_text("")
(package_directory / "importer.py").write_bytes(
f"# -*- coding: {declared_encoding} -*-\nfrom .imported import {imported_name}\n".encode(
codec
)
)
monkeypatch.syspath_prepend(str(tmp_path))

graph = grimp.build_graph("declaredencodingpackage", cache_dir=None)

result = graph.get_import_details(
importer="declaredencodingpackage.importer", imported="declaredencodingpackage.imported"
)

assert [
{
"importer": "declaredencodingpackage.importer",
"imported": "declaredencodingpackage.imported",
"is_lazy": False,
"line_number": 2,
"line_contents": f"from .imported import {imported_name}",
},
] == result
34 changes: 34 additions & 0 deletions tests/functional/test_error_handling.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,4 @@
import re
from pathlib import Path

import pytest
Expand All @@ -19,3 +20,36 @@ def test_syntax_error_includes_module():
filename=filename, lineno=5, text="fromb . import two"
)
assert expected_exception == excinfo.value


@pytest.mark.parametrize(
"contents, expected_problem",
(
pytest.param(b"x = '\xff'\n", "as UTF-8", id="invalid-utf-8"),
pytest.param(
b"# -*- coding: euc-jp -*-\nx = '\xff\xff'\n",
"with encoding 'euc-jp'",
id="invalid-for-declared-encoding",
),
pytest.param(
b"# -*- coding: nonexistent -*-\n",
"(unknown encoding 'nonexistent')",
id="unknown-encoding",
),
),
)
def test_undecodable_source_raises_unicode_error_including_filename(
tmp_path, monkeypatch, contents, expected_problem
):
package_directory = tmp_path / "undecodablepackage"
package_directory.mkdir()
(package_directory / "__init__.py").write_text("")
module_filename = package_directory / "undecodable.py"
module_filename.write_bytes(contents)
monkeypatch.syspath_prepend(str(tmp_path))

with pytest.raises(
UnicodeError,
match=re.escape(f"Failed to decode file {module_filename} {expected_problem}"),
):
build_graph("undecodablepackage", cache_dir=None)
Loading