Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 6 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,8 +8,8 @@ The Python Shapefile Library (PyShp) reads and writes ESRI Shapefiles in pure Py

- **Author**: [Joel Lawhead](https://github.com/GeospatialPython)
- **Maintainers**: [James Parrott](https://github.com/JamesParrott) & [Karim Bahgat](https://github.com/karimbahgat)
- **Version**: 3.1.6.dev
- **Date**: 22nd July 2026
- **Version**: 3.1.6
- **Date**: 25th July 2026
- **License**: [MIT](https://github.com/GeospatialPython/pyshp/blob/master/LICENSE.TXT)

## Contents
Expand Down Expand Up @@ -93,6 +93,10 @@ part of your geospatial project.

# Version Changes

## 3.1.6
### Feature
- Encodings can now be read from .cpg files (and optionally written to them).

## 3.1.5
### Bug fix
- Fixed another bug causing dates before the year 1000 to be encoded as less than 8 chars under "%Y%m%d (found by [Thomas Beierlein](https://github.com/GeospatialPython/pyshp/issues/435))
Expand Down
4 changes: 4 additions & 0 deletions changelog.txt
Original file line number Diff line number Diff line change
@@ -1,3 +1,7 @@
VERSION 3.1.6
2026-07-25
* Encodings can now be read from .cpg files (and optionally written to them).

VERSION 3.1.5
2026-07-22
* Fixed another bug causing dates before the year 1000 to be encoded as less than 8 chars under "%Y%m%d (found by [Thomas Beierlein](https://github.com/GeospatialPython/pyshp/issues/435))
Expand Down
131 changes: 79 additions & 52 deletions src/shapefile.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@

from __future__ import annotations

__version__ = "3.1.6.dev"
__version__ = "3.1.6"

import abc
import array
Expand Down Expand Up @@ -2819,25 +2819,6 @@ def _try_to_download_binary_file(
return initial_bytes, cast(ReadableBinStream, resp)


def _try_get_open_constituent_file(
file: Path,
ext: Literal[".shp", ".shx", ".dbf"],
) -> IO[bytes] | None:
"""
Attempts to open a .shp, .dbf or .shx file,
with both lower case and upper case file extensions,
and return it. If it was not possible to open the file, None is returned.
"""
exts = {ext, ext.upper(), ext.lower()}

for candidate_ext in exts:
try:
return file.with_suffix(candidate_ext).open("rb")
except OSError:
pass
return None


def ensure_within_bounds(i: int, num_records: int) -> int:
"""Provides list-like handling of a record index with a clearer
error message if the index is out of bounds."""
Expand Down Expand Up @@ -3359,7 +3340,7 @@ def shape_lengths_B(self) -> list[int]:


class ShpReader(_HasCheckedReadableFile):
"""Reads an shp file."""
"""Reads a .shp file."""

FileProto = ReadSeekableBinStream
new_file_obj_mode = "rb"
Expand Down Expand Up @@ -3618,21 +3599,23 @@ def __init__(
shapefile_path: str | PathLike[Any] = "",
/,
*,
encoding: str = "utf-8",
encoding: str | None = None,
encodingErrors: str = "strict",
shp: _NoShpSentinel | BinaryFileT | None = _NoShpSentinel(),
shx: BinaryFileT | None = None,
dbf: BinaryFileT | None = None,
cpg: BinaryFileT | None = None,
# Keep kwargs even though unused, to preserve PyShp 2.4 API
**kwargs: Any,
):
super().__init__()
# Store encoding info to use if lazy loading DbfReader later.
self.encoding = encoding
self._user_specified_encoding = encoding
self.encodingErrors = encodingErrors
self._shp = None
self._shx = None
self._dbf = None
self._cpg = None
self.shapeName = "Not specified"
self.numShapes: int = 0
self.path: str | os.PathLike[Any] | None = None
Expand All @@ -3651,13 +3634,15 @@ def __init__(
discarded_kwargs["shx"] = shx
if dbf is not None:
discarded_kwargs["dbf"] = dbf
if cpg is not None:
discarded_kwargs["cpg"] = cpg
if discarded_kwargs:
raise TypeError(
"Please be specific about the shapefile you want to load. "
f"Got: {shapefile_path}, plus the following unusable "
" kwargs: {discarded_kwargs} which previous versions of PyShp ignored. \n"
"Only either: i) exactly one positional arg \n"
" or: ii) one or both of shp and dbf, optionally plus shx kwargs\n"
" or: ii) one or both of shp and dbf, optionally plus shx and cpg kwargs\n"
"is currently supported. All other kwargs may be set (or not). "
)
self.path = shapefile_path
Expand All @@ -3684,6 +3669,7 @@ def __init__(
zipfileobj = self._download_binary_file_from_url(
url_info,
".zip",
add_to_exit_stack=False,
suppress_http_errors=False,
)
if zipfileobj is None:
Expand Down Expand Up @@ -3722,12 +3708,14 @@ def __init__(
self._shp = self._seek_0_on_file_obj_wrap_or_open_from_name(".shp", shp)
self._shx = self._seek_0_on_file_obj_wrap_or_open_from_name(".shx", shx)

self._cpg = self._seek_0_on_file_obj_wrap_or_open_from_name(".cpg", cpg)
self._dbf = self._seek_0_on_file_obj_wrap_or_open_from_name(".dbf", dbf)

# Load the files
if self._shp:
self._get_shp_reader()
if self._dbf:
# Sets self.encoding
self._get_dbf_reader()
if self._shx:
self._get_shx_reader()
Expand Down Expand Up @@ -3761,12 +3749,37 @@ def _get_dbf_reader(self) -> DbfReader:
raise ShapefileException(
"DbfReader requires a .dbf file or file-like object."
)
self._set_encoding()
return DbfReader(
dbf=self._dbf,
encoding=self.encoding,
encodingErrors=self.encodingErrors,
)

@functools.cache
def _set_encoding(self) -> None:
if self._cpg is None:
encoding_from_cpg = ""
else:
encoding_from_cpg = (
self._cpg.read().decode().lower().replace("-", "_").strip()
)
if not encoding_from_cpg:
warnings.warn("Empty .cpg file (no encoding found). ")

if self._user_specified_encoding is None:
encoding = encoding_from_cpg
else:
encoding = self._user_specified_encoding.lower().replace("-", "_").strip()

if encoding_from_cpg and encoding != encoding_from_cpg:
warnings.warn(
f"Specified encoding: {encoding} "
"different to encoding read from "
f".cpg file: {encoding_from_cpg}",
)
self.encoding = encoding or "utf-8"

@property
def shp_reader(self) -> ShpReader:
return self._get_shp_reader()
Expand Down Expand Up @@ -3854,20 +3867,41 @@ def iterRecords(
) -> Iterator[_Record | None]:
return self.dbf_reader.iterRecords(fields, start, stop, deleted_as_None)

def _try_get_open_constituent_file(
self,
file: Path,
ext: Literal[".shp", ".shx", ".dbf", ".cpg"],
) -> IO[bytes] | None:
"""
Attempts to open a .shp, .dbf or .shx file,
with both lower case and upper case file extensions,
and return it. If it was not possible to open the file, None is returned.
"""
exts = {ext, ext.upper(), ext.lower()}

for candidate_ext in exts:
try:
file_obj = file.with_suffix(candidate_ext).open("rb")
except OSError:
continue
self.exit_stack.enter_context(file_obj)
return file_obj
return None

def _seek_0_on_file_obj_wrap_or_open_from_name(
self,
ext: Literal[".shp", ".shx", ".dbf"],
ext: Literal[".shp", ".shx", ".dbf", ".cpg"],
file: BinaryFileT | None,
) -> None | IO[bytes]:
if file is None:
return None

if isinstance(file, (str, PathLike)):
file_obj = _try_get_open_constituent_file(Path(file), ext)
if file_obj is not None:
self.exit_stack.enter_context(file_obj)
return file_obj
# Added to exit stack if opened.
return self._try_get_open_constituent_file(Path(file), ext)

# Other user-opened file objects not added to exit stack.
# The user must close them.
if hasattr(file, "read"):
# Copy if required
try:
Expand All @@ -3884,7 +3918,8 @@ def _seek_0_on_file_obj_wrap_or_open_from_name(
def _download_binary_file_from_url(
self,
urlinfo: SplitResult,
ext: Literal[".shp", ".shx", ".dbf", ".zip"],
ext: Literal[".shp", ".shx", ".dbf", ".cpg", ".zip"],
add_to_exit_stack: bool = True,
suppress_http_errors: bool = True,
) -> tempfile._TemporaryFileWrapper[bytes] | None:
sniffed_bytes, resp = _try_to_download_binary_file(
Expand All @@ -3896,26 +3931,19 @@ def _download_binary_file_from_url(
return None
# Use tempfile as source for url data.
fileobj = _save_to_named_tmp_file(resp, initial_bytes=sniffed_bytes)
if add_to_exit_stack:
self.exit_stack.enter_context(fileobj)
return fileobj

def _load_from_url(self, urlinfo: SplitResult) -> None:
# Shapefile is from a url
# Download each file to temporary path and treat as normal shapefile path
self._shp = self._download_binary_file_from_url(urlinfo, ".shp")
self._shx = self._download_binary_file_from_url(urlinfo, ".shx")
self._cpg = self._download_binary_file_from_url(urlinfo, ".cpg")
self._dbf = self._download_binary_file_from_url(urlinfo, ".dbf")

shp_or_dbf_loaded = False
if self._shx is not None:
self.exit_stack.enter_context(self._shx)
if self._shp is not None:
self.exit_stack.enter_context(self._shp)
shp_or_dbf_loaded = True
if self._dbf is not None:
self.exit_stack.enter_context(self._dbf)
shp_or_dbf_loaded = True

if not shp_or_dbf_loaded:
if self._shp is None and self._dbf is None:
raise ShapefileException(
f"Failed to download .shp or .dbf from: {urlunsplit(urlinfo)}"
)
Expand All @@ -3924,7 +3952,7 @@ def _load_file_from_zip_to_tmp_file(
self,
archive: zipfile.ZipFile,
file: Path,
ext: Literal[".shp", ".shx", ".dbf"],
ext: Literal[".shp", ".shx", ".dbf", ".cpg"],
) -> tempfile._TemporaryFileWrapper[bytes] | None:
for cased_ext in {ext.lower(), ext.upper(), ext}:
try:
Expand Down Expand Up @@ -3978,7 +4006,7 @@ def _load_from_zipfileobj(
constituent_files = (
Path(name)
for name in archive.namelist()
if name.lower().endswith((".shp", ".dbf", ".shx"))
if name.lower().endswith((".shp", ".dbf", ".shx", ".cpg"))
)

def without_ext(path: Path) -> Path:
Expand All @@ -4001,6 +4029,7 @@ def without_ext(path: Path) -> Path:
# Try to extract file-like objects from zipfile
self._shp = self._load_file_from_zip_to_tmp_file(archive, shapefile, ".shp")
self._shx = self._load_file_from_zip_to_tmp_file(archive, shapefile, ".shx")
self._cpg = self._load_file_from_zip_to_tmp_file(archive, shapefile, ".cpg")
self._dbf = self._load_file_from_zip_to_tmp_file(archive, shapefile, ".dbf")

def load(self, file: str | os.PathLike[Any]) -> None:
Expand All @@ -4024,27 +4053,25 @@ def load_shp(self, file: Path) -> None:
"""
Attempts to load file with .shp extension as both lower and upper case
"""
self._shp = _try_get_open_constituent_file(file, ".shp")
if self._shp:
self.exit_stack.enter_context(self._shp)
self._shp = self._try_get_open_constituent_file(file, ".shp")
if self._shp is not None:
self._get_shp_reader()

def load_shx(self, file: Path) -> None:
"""
Attempts to load file with .shx extension as both lower and upper case
"""
self._shx = _try_get_open_constituent_file(file, ".shx")
if self._shx:
self.exit_stack.enter_context(self._shx)
self._shx = self._try_get_open_constituent_file(file, ".shx")
if self._shx is not None:
self._get_shx_reader()

def load_dbf(self, file: Path) -> None:
"""
Attempts to load file with .dbf extension as both lower and upper case
"""
self._dbf = _try_get_open_constituent_file(file, ".dbf")
if self._dbf:
self.exit_stack.enter_context(self._dbf)
self._dbf = self._try_get_open_constituent_file(file, ".dbf")
if self._dbf is not None:
self._cpg = self._try_get_open_constituent_file(file, ".cpg")
self._get_dbf_reader()

def __len__(self) -> int:
Expand Down
48 changes: 47 additions & 1 deletion tests/test_shapefile.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
import json
import os.path
from pathlib import Path
import shutil

# third party imports
import pytest
Expand Down Expand Up @@ -2164,4 +2165,49 @@ def test_encode_dbf_field_values(value,encoded_len,codec,errors):
with context:
r = shapefile.DbfReader(stream, encoding=codec, encodingErrors=errors, strict=False)
assert r.record(0)[0] == value
r.close()
r.close()

@pytest.fixture
def tmp_latin1_shapefile_shp(tmp_path):
name = "latin1"
test_shapefile_dir = tmp_path / name
test_shapefile_dir.mkdir()
for file in shapefiles_dir.glob("latin1.*"):
shutil.copy(file, test_shapefile_dir)
test_shapefile = test_shapefile_dir / f"{name}.shp"
return test_shapefile

ENCODINGS_AND_CONTEXTS = [
("latin1", contextlib.nullcontext()),
("utf8", pytest.raises(shapefile.dbfFileException)),
]
@pytest.mark.parametrize("encoding, context", ENCODINGS_AND_CONTEXTS)
def test_read_latin1_shapefile(encoding, context, tmp_latin1_shapefile_shp):
""" Extend the smoke test in README.md doctests """

assert tmp_latin1_shapefile_shp.is_file()

r = shapefile.Reader(tmp_latin1_shapefile_shp, encoding=encoding)
with context:
rec = r.record(0)
r.close()
if encoding == "latin1":
assert rec == [2, u'Ñandú']

@pytest.mark.parametrize("encoding, context", ENCODINGS_AND_CONTEXTS)
def test_read_latin1_shapefile_cpg_file(encoding, context, tmp_latin1_shapefile_shp):
""" Extend the smoke test in README.md doctests """

assert tmp_latin1_shapefile_shp.is_file()

cpg_file = tmp_latin1_shapefile_shp.with_suffix(".cpg")
cpg_file.write_text(encoding.upper().replace("_","-"))

r = shapefile.Reader(tmp_latin1_shapefile_shp)
with context:
rec = r.record(0)
r.close()
if encoding == "latin1":
assert rec == [2, u'Ñandú']


Loading