Coverage for pygeoapi/provider/csv_.py: 86%

118 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 08:15 +0000

1# ================================================================= 

2# 

3# Authors: Tom Kralidis <tomkralidis@gmail.com> 

4# Francesco Bartoli <xbartolone@gmail.com> 

5# 

6# Copyright (c) 2025 Tom Kralidis 

7# Copyright (c) 2025 Francesco Bartoli 

8# 

9# Permission is hereby granted, free of charge, to any person 

10# obtaining a copy of this software and associated documentation 

11# files (the "Software"), to deal in the Software without 

12# restriction, including without limitation the rights to use, 

13# copy, modify, merge, publish, distribute, sublicense, and/or sell 

14# copies of the Software, and to permit persons to whom the 

15# Software is furnished to do so, subject to the following 

16# conditions: 

17# 

18# The above copyright notice and this permission notice shall be 

19# included in all copies or substantial portions of the Software. 

20# 

21# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, 

22# EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES 

23# OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND 

24# NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT 

25# HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, 

26# WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING 

27# FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR 

28# OTHER DEALINGS IN THE SOFTWARE. 

29# 

30# ================================================================= 

31 

32from collections import OrderedDict 

33import csv 

34import itertools 

35import logging 

36 

37from shapely.geometry import box, Point 

38 

39from pygeoapi.crs import crs_transform 

40from pygeoapi.provider.base import (BaseProvider, ProviderInvalidQueryError, 

41 ProviderItemNotFoundError, 

42 ProviderQueryError) 

43from pygeoapi.util import get_typed_value 

44 

45LOGGER = logging.getLogger(__name__) 

46 

47 

48class CSVProvider(BaseProvider): 

49 """CSV provider""" 

50 

51 def __init__(self, provider_def): 

52 """ 

53 Initialize object 

54 

55 :param provider_def: provider definition 

56 

57 :returns: pygeoapi.provider.csv_.CSVProvider 

58 """ 

59 

60 super().__init__(provider_def) 

61 self.geometry_x = provider_def['geometry']['x_field'] 

62 self.geometry_y = provider_def['geometry']['y_field'] 

63 self.get_fields() 

64 

65 def get_fields(self): 

66 """ 

67 Get provider field information (names, types) 

68 

69 :returns: dict of fields 

70 """ 

71 if not self._fields: 71 ↛ 95line 71 didn't jump to line 95 because the condition on line 71 was always true

72 LOGGER.debug('Treating all columns as string types') 

73 with open(self.data) as ff: 

74 LOGGER.debug('Serializing DictReader') 

75 data_ = csv.DictReader(ff) 

76 

77 row = next(data_) 

78 

79 for key, value in row.items(): 

80 LOGGER.debug(f'key: {key}, value: {value}') 

81 value2 = get_typed_value(value) 

82 if key in [self.geometry_x, self.geometry_y]: 

83 continue 

84 if key == self.id_field: 

85 type_ = 'string' 

86 elif isinstance(value2, float): 

87 type_ = 'number' 

88 elif isinstance(value2, int): 

89 type_ = 'integer' 

90 else: 

91 type_ = 'string' 

92 

93 self._fields[key] = {'type': type_} 

94 

95 return self._fields 

96 

97 def _load(self, offset=0, limit=10, resulttype='results', 

98 identifier=None, bbox=[], datetime_=None, properties=[], 

99 select_properties=[], skip_geometry=False, q=None): 

100 """ 

101 Load CSV data 

102 

103 :param offset: starting record to return (default 0) 

104 :param limit: number of records to return (default 10) 

105 :param datetime_: temporal (datestamp or extent) 

106 :param resulttype: return results or hit limit (default results) 

107 :param bbox: bounding box [minx,miny,maxx,maxy] 

108 :param properties: list of tuples (name, value) 

109 :param select_properties: list of property names 

110 :param skip_geometry: bool of whether to skip geometry (default False) 

111 :param q: full-text search term(s) 

112 

113 :returns: dict of GeoJSON FeatureCollection 

114 """ 

115 

116 found = False 

117 result = None 

118 feature_collection = { 

119 'type': 'FeatureCollection', 

120 'features': [] 

121 } 

122 if identifier is not None: 

123 # Loop through all rows when searching for a single feature 

124 limit = self._load(resulttype='hits').get('numberMatched') 

125 

126 with open(self.data) as ff: 

127 LOGGER.debug('Serializing DictReader') 

128 data_ = csv.DictReader(ff) 

129 

130 if properties: 

131 for prop in properties: 

132 if prop[0] not in data_.fieldnames: 132 ↛ 133line 132 didn't jump to line 133 because the condition on line 132 was never true

133 msg = 'Invalid fieldname' 

134 LOGGER.error(msg) 

135 raise ProviderInvalidQueryError(msg) 

136 

137 data_ = filter( 

138 lambda p: all( 

139 [p[prop[0]] == prop[1] for prop in properties]), data_) 

140 

141 if bbox: 

142 LOGGER.debug('processing bbox parameter') 

143 if len(bbox) > 4: 

144 LOGGER.debug("bbox reduced to 4 elements") 

145 bbox = bbox[:4] 

146 data_ = filter( 

147 lambda f: all( 

148 [self._intersects(f, bbox)]), data_) 

149 

150 if resulttype == 'hits': 

151 LOGGER.debug('Returning hits only') 

152 feature_collection['numberMatched'] = len(list(data_)) 

153 return feature_collection 

154 

155 LOGGER.debug('Slicing CSV rows') 

156 for row in itertools.islice(data_, 0, None): 

157 try: 

158 coordinates = [ 

159 float(row.pop(self.geometry_x)), 

160 float(row.pop(self.geometry_y)), 

161 ] 

162 except ValueError: 

163 msg = f'Row with invalid geometry: {row.get(self.id_field)}, setting coordinates to None' # noqa 

164 LOGGER.warning(msg) 

165 coordinates = None 

166 

167 feature = {'type': 'Feature'} 

168 feature['id'] = row[self.id_field] 

169 if not skip_geometry: 169 ↛ 175line 169 didn't jump to line 175 because the condition on line 169 was always true

170 feature['geometry'] = { 

171 'type': 'Point', 

172 'coordinates': coordinates 

173 } 

174 else: 

175 feature['geometry'] = None 

176 

177 feature['properties'] = OrderedDict() 

178 

179 if self.properties or select_properties: 179 ↛ 180line 179 didn't jump to line 180 because the condition on line 179 was never true

180 for p in set(self.properties) | set(select_properties): 

181 try: 

182 feature['properties'][p] = get_typed_value(row[p]) 

183 except KeyError as err: 

184 LOGGER.error(err) 

185 raise ProviderQueryError() 

186 else: 

187 for key, value in row.items(): 

188 LOGGER.debug(f'key: {key}, value: {value}') 

189 feature['properties'][key] = get_typed_value(value) 

190 

191 if identifier is not None and feature['id'] == identifier: 

192 found = True 

193 result = feature 

194 

195 feature_collection['features'].append(feature) 

196 

197 feature_collection['numberMatched'] = \ 

198 len(feature_collection['features']) 

199 

200 if identifier is not None and not found: 

201 return None 

202 elif identifier is not None and found: 

203 return result 

204 

205 features_returned = feature_collection['features'][offset:offset+limit] 

206 feature_collection['features'] = features_returned 

207 

208 feature_collection['numberReturned'] = len( 

209 feature_collection['features']) 

210 

211 return feature_collection 

212 

213 def _intersects(self, data, bbox): 

214 """ 

215 Helper function to evaluate point geometry intersection with a bbox 

216 

217 :param geometry: `dict` of CSV row 

218 :param bbox: `list` of bbox 

219 

220 :returns: `bool` of whether point geometry intersects with bbox 

221 """ 

222 

223 if None in [data.get(self.geometry_x), data.get(self.geometry_y)]: 223 ↛ 224line 223 didn't jump to line 224 because the condition on line 223 was never true

224 return True 

225 

226 point = Point(data[self.geometry_x], data[self.geometry_y]) 

227 bbox2 = box(*bbox) 

228 

229 return bbox2.intersects(point) 

230 

231 @crs_transform 

232 def query(self, offset=0, limit=10, resulttype='results', 

233 bbox=[], datetime_=None, properties=[], sortby=[], 

234 select_properties=[], skip_geometry=False, q=None, **kwargs): 

235 """ 

236 CSV query 

237 

238 :param offset: starting record to return (default 0) 

239 :param limit: number of records to return (default 10) 

240 :param resulttype: return results or hit limit (default results) 

241 :param bbox: bounding box [minx,miny,maxx,maxy] 

242 :param datetime_: temporal (datestamp or extent) 

243 :param properties: list of tuples (name, value) 

244 :param sortby: list of dicts (property, order) 

245 :param select_properties: list of property names 

246 :param skip_geometry: bool of whether to skip geometry (default False) 

247 :param q: full-text search term(s) 

248 

249 :returns: dict of GeoJSON FeatureCollection 

250 """ 

251 

252 return self._load(offset, limit, resulttype, 

253 bbox=bbox, properties=properties, 

254 select_properties=select_properties, 

255 skip_geometry=skip_geometry) 

256 

257 @crs_transform 

258 def get(self, identifier, **kwargs): 

259 """ 

260 query CSV id 

261 

262 :param identifier: feature id 

263 

264 :returns: dict of single GeoJSON feature 

265 """ 

266 item = self._load(identifier=identifier) 

267 if item: 

268 return item 

269 else: 

270 err = f'item {identifier} not found' 

271 LOGGER.error(err) 

272 raise ProviderItemNotFoundError(err) 

273 

274 def __repr__(self): 

275 return f'<CSVProvider> {self.data}'