Coverage for pygeoapi/provider/geojson.py: 58%
107 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 08:15 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 08:15 +0000
1# =================================================================
2#
3# Authors: Matthew Perry <perrygeo@gmail.com>
4#
5# Copyright (c) 2018 Matthew Perry
6# Copyright (c) 2022 Tom Kralidis
7#
8# Permission is hereby granted, free of charge, to any person
9# obtaining a copy of this software and associated documentation
10# files (the "Software"), to deal in the Software without
11# restriction, including without limitation the rights to use,
12# copy, modify, merge, publish, distribute, sublicense, and/or sell
13# copies of the Software, and to permit persons to whom the
14# Software is furnished to do so, subject to the following
15# conditions:
16#
17# The above copyright notice and this permission notice shall be
18# included in all copies or substantial portions of the Software.
19#
20# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
21# EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES
22# OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
23# NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
24# HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
25# WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
26# FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
27# OTHER DEALINGS IN THE SOFTWARE.
28#
29# =================================================================
31import json
32import logging
33import os
34import uuid
36from shapely.geometry import box, shape
38from pygeoapi.crs import crs_transform
39from pygeoapi.provider.base import BaseProvider, ProviderItemNotFoundError
41LOGGER = logging.getLogger(__name__)
44class GeoJSONProvider(BaseProvider):
45 """Provider class backed by local GeoJSON files
47 This is meant to be simple
48 (no external services, no dependencies, no schema)
50 at the expense of performance
51 (no indexing, full serialization roundtrip on each request)
53 Not thread safe, a single server process is assumed
55 This implementation uses the feature 'id' heavily
56 and will override any 'id' provided in the original data.
57 The feature 'properties' will be preserved.
59 TODO:
60 * query method should take bbox
61 * instead of methods returning FeatureCollections,
62 we should be yielding Features and aggregating in the view
63 * there are strict id semantics; all features in the input GeoJSON file
64 must be present and be unique strings. Otherwise it will break.
65 * How to raise errors in the provider implementation such that
66 * appropriate HTTP responses will be raised
67 """
69 def __init__(self, provider_def):
70 """initializer"""
72 super().__init__(provider_def)
73 self.get_fields()
75 def get_fields(self):
76 """
77 Get provider field information (names, types)
79 :returns: dict of fields
80 """
82 if not self._fields: 82 ↛ 99line 82 didn't jump to line 99 because the condition on line 82 was always true
83 LOGGER.debug('Treating all columns as string types')
84 if os.path.exists(self.data): 84 ↛ 97line 84 didn't jump to line 97 because the condition on line 84 was always true
85 with open(self.data) as src:
86 data = json.loads(src.read())
87 for key, value in data['features'][0]['properties'].items():
88 if isinstance(value, float): 88 ↛ 89line 88 didn't jump to line 89 because the condition on line 88 was never true
89 type_ = 'number'
90 elif isinstance(value, int):
91 type_ = 'integer'
92 else:
93 type_ = 'string'
95 self._fields[key] = {'type': type_}
96 else:
97 LOGGER.warning(f'File {self.data} does not exist.')
99 return self._fields
101 def _load(self, bbox=[], skip_geometry=None, properties=[],
102 select_properties=[]):
103 """Load and validate the source GeoJSON file
104 at self.data
106 Yes loading from disk, deserializing and validation
107 happens on every request. This is not efficient.
108 """
110 if os.path.exists(self.data): 110 ↛ 114line 110 didn't jump to line 114 because the condition on line 110 was always true
111 with open(self.data) as src:
112 data = json.loads(src.read())
113 else:
114 LOGGER.warning(f'File {self.data} does not exist.')
115 data = {
116 'type': 'FeatureCollection',
117 'features': []}
119 # Must be a FeatureCollection
120 assert data['type'] == 'FeatureCollection'
122 # filter by properties if set
123 if properties:
124 data['features'] = [f for f in data['features'] if \
125 all([str(f['properties'][p[0]]) == str(p[1]) for p in properties])] # noqa
127 # filter by bbox if set
128 if bbox:
129 LOGGER.debug('processing bbox parameter')
130 if len(bbox) > 4:
131 LOGGER.debug("bbox reduced to 4 elements")
132 bbox = bbox[:4]
133 data['features'] = [f for f in data['features'] if \
134 self._intersects(f['geometry'], bbox)] # noqa
136 # All features must have ids, TODO must be unique strings
137 for i in data['features']:
138 if 'id' not in i and self.id_field in i['properties']: 138 ↛ 140line 138 didn't jump to line 140 because the condition on line 138 was always true
139 i['id'] = i['properties'][self.id_field]
140 if skip_geometry: 140 ↛ 141line 140 didn't jump to line 141 because the condition on line 140 was never true
141 i['geometry'] = None
142 if self.properties or select_properties: 142 ↛ 143line 142 didn't jump to line 143 because the condition on line 142 was never true
143 i['properties'] = {k: v for k, v in i['properties'].items()
144 if k in set(self.properties) | set(select_properties)} # noqa
145 return data
147 def _intersects(self, geometry, bbox):
148 """
149 Helper function to evaluate feature geometry intersection with a bbox
151 :param geometry: `dict` of GeoJSON geometry
152 :param bbox: `list` of bbox
154 :returns: `bool` of whether geometry intersects with bbox
155 """
157 if geometry is None: 157 ↛ 158line 157 didn't jump to line 158 because the condition on line 157 was never true
158 return True
160 bbox2 = box(*bbox)
161 geometry2 = shape(geometry)
163 return geometry2.intersects(bbox2)
165 @crs_transform
166 def query(self, offset=0, limit=10, resulttype='results',
167 bbox=[], datetime_=None, properties=[], sortby=[],
168 select_properties=[], skip_geometry=False, q=None, **kwargs):
169 """
170 query the provider
172 :param offset: starting record to return (default 0)
173 :param limit: number of records to return (default 10)
174 :param resulttype: return results or hit limit (default results)
175 :param bbox: bounding box [minx,miny,maxx,maxy]
176 :param datetime_: temporal (datestamp or extent)
177 :param properties: list of tuples (name, value)
178 :param sortby: list of dicts (property, order)
179 :param select_properties: list of property names
180 :param skip_geometry: bool of whether to skip geometry (default False)
181 :param q: full-text search term(s)
183 :returns: FeatureCollection dict of 0..n GeoJSON features
184 """
186 # TODO filter by bbox without resorting to third-party libs
187 data = self._load(bbox=bbox, skip_geometry=skip_geometry,
188 properties=properties,
189 select_properties=select_properties)
191 data['numberMatched'] = len(data['features'])
193 if resulttype == 'hits': 193 ↛ 194line 193 didn't jump to line 194 because the condition on line 193 was never true
194 data['features'] = []
195 else:
196 data['features'] = data['features'][offset:offset+limit]
197 data['numberReturned'] = len(data['features'])
199 return data
201 @crs_transform
202 def get(self, identifier, **kwargs):
203 """
204 query the provider by id
206 :param identifier: feature id
207 :returns: dict of single GeoJSON feature
208 """
210 all_data = self._load()
211 # if matches
212 for feature in all_data['features']:
213 if str(feature.get('id')) == identifier:
214 return feature
215 # default, no match
216 err = f'item {identifier} not found'
217 LOGGER.error(err)
218 raise ProviderItemNotFoundError(err)
220 def create(self, new_feature):
221 """Create a new feature
223 :param new_feature: new GeoJSON feature dictionary
224 """
226 all_data = self._load()
228 if self.id_field not in new_feature and\
229 self.id_field not in new_feature['properties']:
230 new_feature['properties'][self.id_field] = str(uuid.uuid4())
232 all_data['features'].append(new_feature)
234 with open(self.data, 'w') as dst:
235 dst.write(json.dumps(all_data))
237 def update(self, identifier, new_feature):
238 """Updates an existing feature id with new_feature
240 :param identifier: feature id
241 :param new_feature: new GeoJSON feature dictionary
242 """
244 all_data = self._load()
245 for i, feature in enumerate(all_data['features']):
246 if self.id_field in feature:
247 if feature[self.id_field] == identifier:
248 new_feature['properties'][self.id_field] = identifier
249 all_data['features'][i] = new_feature
250 elif self.id_field in feature['properties']:
251 if feature['properties'][self.id_field] == identifier:
252 new_feature['properties'][self.id_field] = identifier
253 all_data['features'][i] = new_feature
254 with open(self.data, 'w') as dst:
255 dst.write(json.dumps(all_data))
257 def delete(self, identifier):
258 """Deletes an existing feature
260 :param identifier: feature id
261 """
263 all_data = self._load()
264 for i, feature in enumerate(all_data['features']):
265 if self.id_field in feature:
266 if feature[self.id_field] == identifier:
267 all_data['features'].pop(i)
268 elif self.id_field in feature['properties']:
269 if feature['properties'][self.id_field] == identifier:
270 all_data['features'].pop(i)
271 with open(self.data, 'w') as dst:
272 dst.write(json.dumps(all_data))
274 def __repr__(self):
275 return f'<GeoJSONProvider> {self.data}'