@@ -276,6 +276,7 @@ def _read_csv(self) -> Table:
276276 ):
277277 return pa .Table .from_pydict ({})
278278 length = self ._get_content_length ()
279+ binary_columns = {d [0 ] for d in self .description or [] if d [1 ] == "varbinary" }
279280 if length and self .output_location .endswith (".txt" ):
280281 description = self .description if self .description else []
281282 column_names = [d [0 ] for d in description ]
@@ -296,6 +297,7 @@ def _read_csv(self) -> Table:
296297 parse_opts = csv .ParseOptions (
297298 delimiter = "," ,
298299 quote_char = '"' ,
300+ ignore_empty_lines = not binary_columns ,
299301 double_quote = True ,
300302 escape_char = False ,
301303 )
@@ -304,16 +306,25 @@ def _read_csv(self) -> Table:
304306
305307 bucket , key = parse_output_location (self .output_location )
306308 try :
307- return csv .read_csv (
309+ table = csv .read_csv (
308310 self ._fs .open_input_stream (f"{ bucket } /{ key } " ),
309311 read_options = read_opts ,
310312 parse_options = parse_opts ,
311313 convert_options = csv .ConvertOptions (
314+ strings_can_be_null = bool (binary_columns ),
312315 quoted_strings_can_be_null = False ,
313316 timestamp_parsers = self .timestamp_parsers ,
314317 column_types = self .column_types ,
315318 ),
316319 )
320+ if binary_columns :
321+ for index , field in enumerate (table .schema ):
322+ if field .name not in binary_columns and (
323+ pa .types .is_string (field .type ) or pa .types .is_binary (field .type )
324+ ):
325+ # Preserve the existing CSV behavior for non-binary Athena columns.
326+ table = table .set_column (index , field , table .column (index ).fill_null ("" ))
327+ return table
317328 except Exception as e :
318329 _logger .exception (f"Failed to read { bucket } /{ key } ." )
319330 raise OperationalError (* e .args ) from e
0 commit comments