From 987fb1ad4dc50e334226dbe9f0a7d1ba519b73ba Mon Sep 17 00:00:00 2001 From: James Parrott <80779630+JamesParrott@users.noreply.github.com> Date: Sat, 25 Jul 2026 10:46:34 +0100 Subject: [PATCH 1/4] Describe 3.1.6 --- README.md | 8 ++++++-- changelog.txt | 4 ++++ src/shapefile.py | 2 +- 3 files changed, 11 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 173aef2..e11b321 100644 --- a/README.md +++ b/README.md @@ -8,8 +8,8 @@ The Python Shapefile Library (PyShp) reads and writes ESRI Shapefiles in pure Py - **Author**: [Joel Lawhead](https://github.com/GeospatialPython) - **Maintainers**: [James Parrott](https://github.com/JamesParrott) & [Karim Bahgat](https://github.com/karimbahgat) -- **Version**: 3.1.6.dev -- **Date**: 22nd July 2026 +- **Version**: 3.1.6 +- **Date**: 25th July 2026 - **License**: [MIT](https://github.com/GeospatialPython/pyshp/blob/master/LICENSE.TXT) ## Contents @@ -93,6 +93,10 @@ part of your geospatial project. # Version Changes +## 3.1.6 +### Feature + - Encodings can now be read from .cpg files (and optionally written to them). + ## 3.1.5 ### Bug fix - Fixed another bug causing dates before the year 1000 to be encoded as less than 8 chars under "%Y%m%d (found by [Thomas Beierlein](https://github.com/GeospatialPython/pyshp/issues/435)) diff --git a/changelog.txt b/changelog.txt index 3e0b2b3..fa92b2f 100644 --- a/changelog.txt +++ b/changelog.txt @@ -1,3 +1,7 @@ +VERSION 3.1.6 +2026-07-25 + * Encodings can now be read from .cpg files (and optionally written to them). + VERSION 3.1.5 2026-07-22 * Fixed another bug causing dates before the year 1000 to be encoded as less than 8 chars under "%Y%m%d (found by [Thomas Beierlein](https://github.com/GeospatialPython/pyshp/issues/435)) diff --git a/src/shapefile.py b/src/shapefile.py index fb5a989..55b864c 100644 --- a/src/shapefile.py +++ b/src/shapefile.py @@ -8,7 +8,7 @@ from __future__ import annotations -__version__ = "3.1.6.dev" +__version__ = "3.1.6" import abc import array From 0a4ef10f69ef210278cd77ff7511f6324a66050b Mon Sep 17 00:00:00 2001 From: James Parrott <80779630+JamesParrott@users.noreply.github.com> Date: Sat, 25 Jul 2026 13:14:38 +0100 Subject: [PATCH 2/4] Add test_read_latin1_shapefile & xfailed cpg file test Add test_read_latin1_shapefile and fixture: tmp_latin1_shapefile_shp Update test_shapefile.py Xfail cpg file test and exclude utf8 raises (dbfFileError) case from it --- tests/test_shapefile.py | 49 ++++++++++++++++++++++++++++++++++++++++- 1 file changed, 48 insertions(+), 1 deletion(-) diff --git a/tests/test_shapefile.py b/tests/test_shapefile.py index b1d6ad2..705ce03 100644 --- a/tests/test_shapefile.py +++ b/tests/test_shapefile.py @@ -8,6 +8,7 @@ import json import os.path from pathlib import Path +import shutil # third party imports import pytest @@ -2164,4 +2165,50 @@ def test_encode_dbf_field_values(value,encoded_len,codec,errors): with context: r = shapefile.DbfReader(stream, encoding=codec, encodingErrors=errors, strict=False) assert r.record(0)[0] == value - r.close() \ No newline at end of file + r.close() + +@pytest.fixture +def tmp_latin1_shapefile_shp(tmp_path): + name = "latin1" + test_shapefile_dir = tmp_path / name + test_shapefile_dir.mkdir() + for file in shapefiles_dir.glob("latin1.*"): + shutil.copy(file, test_shapefile_dir) + test_shapefile = test_shapefile_dir / f"{name}.shp" + return test_shapefile + +ENCODINGS_AND_CONTEXTS = [ + ("latin1", contextlib.nullcontext()), + ("utf8", pytest.raises(shapefile.dbfFileException)), +] +@pytest.mark.parametrize("encoding, context", ENCODINGS_AND_CONTEXTS) +def test_read_latin1_shapefile(encoding, context, tmp_latin1_shapefile_shp): + """ Extend the smoke test in README.md doctests """ + + assert tmp_latin1_shapefile_shp.is_file() + + r = shapefile.Reader(tmp_latin1_shapefile_shp, encoding=encoding) + with context: + rec = r.record(0) + r.close() + if encoding == "latin1": + assert rec == [2, u'Ñandú'] + +@pytest.mark.xfail(reason="Support for reading encodings from .cpg files not implemented yet") +@pytest.mark.parametrize("encoding, context", ENCODINGS_AND_CONTEXTS[:1]) +def test_read_latin1_shapefile_cpg_file(encoding, context, tmp_latin1_shapefile_shp): + """ Extend the smoke test in README.md doctests """ + + assert tmp_latin1_shapefile_shp.is_file() + + cpg_file = tmp_latin1_shapefile_shp.with_suffix(".cpg") + cpg_file.write_text(encoding.upper().replace("_","-")) + + r = shapefile.Reader(tmp_latin1_shapefile_shp) + with context: + rec = r.record(0) + r.close() + if encoding == "latin1": + assert rec == [2, u'Ñandú'] + + From 1a8a939dc8817f6342162581dabfebd4f7bd6ac3 Mon Sep 17 00:00:00 2001 From: James Parrott <80779630+JamesParrott@users.noreply.github.com> Date: Sat, 25 Jul 2026 15:16:20 +0100 Subject: [PATCH 3/4] Move exit stack joining into _try_get_open_constituent_file and make it a method of Reader Fix formatting etc. --- src/shapefile.py | 73 ++++++++++++++++++++++++------------------------ 1 file changed, 37 insertions(+), 36 deletions(-) diff --git a/src/shapefile.py b/src/shapefile.py index 55b864c..53d84f1 100644 --- a/src/shapefile.py +++ b/src/shapefile.py @@ -2819,25 +2819,6 @@ def _try_to_download_binary_file( return initial_bytes, cast(ReadableBinStream, resp) -def _try_get_open_constituent_file( - file: Path, - ext: Literal[".shp", ".shx", ".dbf"], -) -> IO[bytes] | None: - """ - Attempts to open a .shp, .dbf or .shx file, - with both lower case and upper case file extensions, - and return it. If it was not possible to open the file, None is returned. - """ - exts = {ext, ext.upper(), ext.lower()} - - for candidate_ext in exts: - try: - return file.with_suffix(candidate_ext).open("rb") - except OSError: - pass - return None - - def ensure_within_bounds(i: int, num_records: int) -> int: """Provides list-like handling of a record index with a clearer error message if the index is out of bounds.""" @@ -3359,7 +3340,7 @@ def shape_lengths_B(self) -> list[int]: class ShpReader(_HasCheckedReadableFile): - """Reads an shp file.""" + """Reads a .shp file.""" FileProto = ReadSeekableBinStream new_file_obj_mode = "rb" @@ -3618,21 +3599,24 @@ def __init__( shapefile_path: str | PathLike[Any] = "", /, *, - encoding: str = "utf-8", + encoding: str = "utf-8", # | None = None, encodingErrors: str = "strict", shp: _NoShpSentinel | BinaryFileT | None = _NoShpSentinel(), shx: BinaryFileT | None = None, dbf: BinaryFileT | None = None, + # cpg: BinaryFileT | None = None, # Keep kwargs even though unused, to preserve PyShp 2.4 API **kwargs: Any, ): super().__init__() # Store encoding info to use if lazy loading DbfReader later. - self.encoding = encoding + # encoding_from_cpg = + self.encoding = encoding # encoding_from_cpg if encoding is None else encoding self.encodingErrors = encodingErrors self._shp = None self._shx = None self._dbf = None + # self._cpg = None self.shapeName = "Not specified" self.numShapes: int = 0 self.path: str | os.PathLike[Any] | None = None @@ -3651,13 +3635,15 @@ def __init__( discarded_kwargs["shx"] = shx if dbf is not None: discarded_kwargs["dbf"] = dbf + # if cpg is not None: + # discarded_kwargs["cpg"] = cpg if discarded_kwargs: raise TypeError( "Please be specific about the shapefile you want to load. " f"Got: {shapefile_path}, plus the following unusable " " kwargs: {discarded_kwargs} which previous versions of PyShp ignored. \n" "Only either: i) exactly one positional arg \n" - " or: ii) one or both of shp and dbf, optionally plus shx kwargs\n" + " or: ii) one or both of shp and dbf, optionally plus shx \n" # and cpg kwargs\n" "is currently supported. All other kwargs may be set (or not). " ) self.path = shapefile_path @@ -3854,6 +3840,27 @@ def iterRecords( ) -> Iterator[_Record | None]: return self.dbf_reader.iterRecords(fields, start, stop, deleted_as_None) + def _try_get_open_constituent_file( + self, + file: Path, + ext: Literal[".shp", ".shx", ".dbf"], + ) -> IO[bytes] | None: + """ + Attempts to open a .shp, .dbf or .shx file, + with both lower case and upper case file extensions, + and return it. If it was not possible to open the file, None is returned. + """ + exts = {ext, ext.upper(), ext.lower()} + + for candidate_ext in exts: + try: + file_obj = file.with_suffix(candidate_ext).open("rb") + except OSError: + continue + self.exit_stack.enter_context(file_obj) + return file_obj + return None + def _seek_0_on_file_obj_wrap_or_open_from_name( self, ext: Literal[".shp", ".shx", ".dbf"], @@ -3863,10 +3870,7 @@ def _seek_0_on_file_obj_wrap_or_open_from_name( return None if isinstance(file, (str, PathLike)): - file_obj = _try_get_open_constituent_file(Path(file), ext) - if file_obj is not None: - self.exit_stack.enter_context(file_obj) - return file_obj + return self._try_get_open_constituent_file(Path(file), ext) if hasattr(file, "read"): # Copy if required @@ -4024,27 +4028,24 @@ def load_shp(self, file: Path) -> None: """ Attempts to load file with .shp extension as both lower and upper case """ - self._shp = _try_get_open_constituent_file(file, ".shp") - if self._shp: - self.exit_stack.enter_context(self._shp) + self._shp = self._try_get_open_constituent_file(file, ".shp") + if self._shp is not None: self._get_shp_reader() def load_shx(self, file: Path) -> None: """ Attempts to load file with .shx extension as both lower and upper case """ - self._shx = _try_get_open_constituent_file(file, ".shx") - if self._shx: - self.exit_stack.enter_context(self._shx) + self._shx = self._try_get_open_constituent_file(file, ".shx") + if self._shx is not None: self._get_shx_reader() def load_dbf(self, file: Path) -> None: """ Attempts to load file with .dbf extension as both lower and upper case """ - self._dbf = _try_get_open_constituent_file(file, ".dbf") - if self._dbf: - self.exit_stack.enter_context(self._dbf) + self._dbf = self._try_get_open_constituent_file(file, ".dbf") + if self._dbf is not None: self._get_dbf_reader() def __len__(self) -> int: From 21b15aec758780d8d5c2ccf92190e11f2e9e166b Mon Sep 17 00:00:00 2001 From: James Parrott <80779630+JamesParrott@users.noreply.github.com> Date: Sat, 25 Jul 2026 17:37:02 +0100 Subject: [PATCH 4/4] Support reading encodings from .cpg files (all the various ways) --- src/shapefile.py | 74 ++++++++++++++++++++++++++++------------- tests/test_shapefile.py | 3 +- 2 files changed, 51 insertions(+), 26 deletions(-) diff --git a/src/shapefile.py b/src/shapefile.py index 53d84f1..68aceb1 100644 --- a/src/shapefile.py +++ b/src/shapefile.py @@ -3599,24 +3599,23 @@ def __init__( shapefile_path: str | PathLike[Any] = "", /, *, - encoding: str = "utf-8", # | None = None, + encoding: str | None = None, encodingErrors: str = "strict", shp: _NoShpSentinel | BinaryFileT | None = _NoShpSentinel(), shx: BinaryFileT | None = None, dbf: BinaryFileT | None = None, - # cpg: BinaryFileT | None = None, + cpg: BinaryFileT | None = None, # Keep kwargs even though unused, to preserve PyShp 2.4 API **kwargs: Any, ): super().__init__() # Store encoding info to use if lazy loading DbfReader later. - # encoding_from_cpg = - self.encoding = encoding # encoding_from_cpg if encoding is None else encoding + self._user_specified_encoding = encoding self.encodingErrors = encodingErrors self._shp = None self._shx = None self._dbf = None - # self._cpg = None + self._cpg = None self.shapeName = "Not specified" self.numShapes: int = 0 self.path: str | os.PathLike[Any] | None = None @@ -3635,15 +3634,15 @@ def __init__( discarded_kwargs["shx"] = shx if dbf is not None: discarded_kwargs["dbf"] = dbf - # if cpg is not None: - # discarded_kwargs["cpg"] = cpg + if cpg is not None: + discarded_kwargs["cpg"] = cpg if discarded_kwargs: raise TypeError( "Please be specific about the shapefile you want to load. " f"Got: {shapefile_path}, plus the following unusable " " kwargs: {discarded_kwargs} which previous versions of PyShp ignored. \n" "Only either: i) exactly one positional arg \n" - " or: ii) one or both of shp and dbf, optionally plus shx \n" # and cpg kwargs\n" + " or: ii) one or both of shp and dbf, optionally plus shx and cpg kwargs\n" "is currently supported. All other kwargs may be set (or not). " ) self.path = shapefile_path @@ -3670,6 +3669,7 @@ def __init__( zipfileobj = self._download_binary_file_from_url( url_info, ".zip", + add_to_exit_stack=False, suppress_http_errors=False, ) if zipfileobj is None: @@ -3708,12 +3708,14 @@ def __init__( self._shp = self._seek_0_on_file_obj_wrap_or_open_from_name(".shp", shp) self._shx = self._seek_0_on_file_obj_wrap_or_open_from_name(".shx", shx) + self._cpg = self._seek_0_on_file_obj_wrap_or_open_from_name(".cpg", cpg) self._dbf = self._seek_0_on_file_obj_wrap_or_open_from_name(".dbf", dbf) # Load the files if self._shp: self._get_shp_reader() if self._dbf: + # Sets self.encoding self._get_dbf_reader() if self._shx: self._get_shx_reader() @@ -3747,12 +3749,37 @@ def _get_dbf_reader(self) -> DbfReader: raise ShapefileException( "DbfReader requires a .dbf file or file-like object." ) + self._set_encoding() return DbfReader( dbf=self._dbf, encoding=self.encoding, encodingErrors=self.encodingErrors, ) + @functools.cache + def _set_encoding(self) -> None: + if self._cpg is None: + encoding_from_cpg = "" + else: + encoding_from_cpg = ( + self._cpg.read().decode().lower().replace("-", "_").strip() + ) + if not encoding_from_cpg: + warnings.warn("Empty .cpg file (no encoding found). ") + + if self._user_specified_encoding is None: + encoding = encoding_from_cpg + else: + encoding = self._user_specified_encoding.lower().replace("-", "_").strip() + + if encoding_from_cpg and encoding != encoding_from_cpg: + warnings.warn( + f"Specified encoding: {encoding} " + "different to encoding read from " + f".cpg file: {encoding_from_cpg}", + ) + self.encoding = encoding or "utf-8" + @property def shp_reader(self) -> ShpReader: return self._get_shp_reader() @@ -3843,7 +3870,7 @@ def iterRecords( def _try_get_open_constituent_file( self, file: Path, - ext: Literal[".shp", ".shx", ".dbf"], + ext: Literal[".shp", ".shx", ".dbf", ".cpg"], ) -> IO[bytes] | None: """ Attempts to open a .shp, .dbf or .shx file, @@ -3863,15 +3890,18 @@ def _try_get_open_constituent_file( def _seek_0_on_file_obj_wrap_or_open_from_name( self, - ext: Literal[".shp", ".shx", ".dbf"], + ext: Literal[".shp", ".shx", ".dbf", ".cpg"], file: BinaryFileT | None, ) -> None | IO[bytes]: if file is None: return None if isinstance(file, (str, PathLike)): + # Added to exit stack if opened. return self._try_get_open_constituent_file(Path(file), ext) + # Other user-opened file objects not added to exit stack. + # The user must close them. if hasattr(file, "read"): # Copy if required try: @@ -3888,7 +3918,8 @@ def _seek_0_on_file_obj_wrap_or_open_from_name( def _download_binary_file_from_url( self, urlinfo: SplitResult, - ext: Literal[".shp", ".shx", ".dbf", ".zip"], + ext: Literal[".shp", ".shx", ".dbf", ".cpg", ".zip"], + add_to_exit_stack: bool = True, suppress_http_errors: bool = True, ) -> tempfile._TemporaryFileWrapper[bytes] | None: sniffed_bytes, resp = _try_to_download_binary_file( @@ -3900,6 +3931,8 @@ def _download_binary_file_from_url( return None # Use tempfile as source for url data. fileobj = _save_to_named_tmp_file(resp, initial_bytes=sniffed_bytes) + if add_to_exit_stack: + self.exit_stack.enter_context(fileobj) return fileobj def _load_from_url(self, urlinfo: SplitResult) -> None: @@ -3907,19 +3940,10 @@ def _load_from_url(self, urlinfo: SplitResult) -> None: # Download each file to temporary path and treat as normal shapefile path self._shp = self._download_binary_file_from_url(urlinfo, ".shp") self._shx = self._download_binary_file_from_url(urlinfo, ".shx") + self._cpg = self._download_binary_file_from_url(urlinfo, ".cpg") self._dbf = self._download_binary_file_from_url(urlinfo, ".dbf") - shp_or_dbf_loaded = False - if self._shx is not None: - self.exit_stack.enter_context(self._shx) - if self._shp is not None: - self.exit_stack.enter_context(self._shp) - shp_or_dbf_loaded = True - if self._dbf is not None: - self.exit_stack.enter_context(self._dbf) - shp_or_dbf_loaded = True - - if not shp_or_dbf_loaded: + if self._shp is None and self._dbf is None: raise ShapefileException( f"Failed to download .shp or .dbf from: {urlunsplit(urlinfo)}" ) @@ -3928,7 +3952,7 @@ def _load_file_from_zip_to_tmp_file( self, archive: zipfile.ZipFile, file: Path, - ext: Literal[".shp", ".shx", ".dbf"], + ext: Literal[".shp", ".shx", ".dbf", ".cpg"], ) -> tempfile._TemporaryFileWrapper[bytes] | None: for cased_ext in {ext.lower(), ext.upper(), ext}: try: @@ -3982,7 +4006,7 @@ def _load_from_zipfileobj( constituent_files = ( Path(name) for name in archive.namelist() - if name.lower().endswith((".shp", ".dbf", ".shx")) + if name.lower().endswith((".shp", ".dbf", ".shx", ".cpg")) ) def without_ext(path: Path) -> Path: @@ -4005,6 +4029,7 @@ def without_ext(path: Path) -> Path: # Try to extract file-like objects from zipfile self._shp = self._load_file_from_zip_to_tmp_file(archive, shapefile, ".shp") self._shx = self._load_file_from_zip_to_tmp_file(archive, shapefile, ".shx") + self._cpg = self._load_file_from_zip_to_tmp_file(archive, shapefile, ".cpg") self._dbf = self._load_file_from_zip_to_tmp_file(archive, shapefile, ".dbf") def load(self, file: str | os.PathLike[Any]) -> None: @@ -4046,6 +4071,7 @@ def load_dbf(self, file: Path) -> None: """ self._dbf = self._try_get_open_constituent_file(file, ".dbf") if self._dbf is not None: + self._cpg = self._try_get_open_constituent_file(file, ".cpg") self._get_dbf_reader() def __len__(self) -> int: diff --git a/tests/test_shapefile.py b/tests/test_shapefile.py index 705ce03..168bd63 100644 --- a/tests/test_shapefile.py +++ b/tests/test_shapefile.py @@ -2194,8 +2194,7 @@ def test_read_latin1_shapefile(encoding, context, tmp_latin1_shapefile_shp): if encoding == "latin1": assert rec == [2, u'Ñandú'] -@pytest.mark.xfail(reason="Support for reading encodings from .cpg files not implemented yet") -@pytest.mark.parametrize("encoding, context", ENCODINGS_AND_CONTEXTS[:1]) +@pytest.mark.parametrize("encoding, context", ENCODINGS_AND_CONTEXTS) def test_read_latin1_shapefile_cpg_file(encoding, context, tmp_latin1_shapefile_shp): """ Extend the smoke test in README.md doctests """