Skip to content

Commit

Permalink
fix webdataset pickling
Browse files Browse the repository at this point in the history
  • Loading branch information
lhoestq committed Jun 14, 2024
1 parent 087671d commit 6e380f7
Show file tree
Hide file tree
Showing 2 changed files with 6 additions and 4 deletions.
8 changes: 5 additions & 3 deletions src/datasets/utils/file_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -1567,18 +1567,20 @@ def xxml_dom_minidom_parse(filename_or_file, download_config: Optional[DownloadC
class _IterableFromGenerator(TrackedIterable):
"""Utility class to create an iterable from a generator function, in order to reset the generator when needed."""

def __init__(self, generator: Callable, *args, **kwargs):
def __init__(self, generator: Callable, *args):
super().__init__()
self.generator = generator
self.args = args
self.kwargs = kwargs

def __iter__(self):
for x in self.generator(*self.args, **self.kwargs):
for x in self.generator(*self.args):
self.last_item = x
yield x
self.last_item = None

def __reduce__(self):
return (self.__class__, (self.generator, *self.args))


class ArchiveIterable(_IterableFromGenerator):
"""An iterable of (path, fileobj) from a TAR archive, used by `iter_archive`"""
Expand Down
2 changes: 1 addition & 1 deletion src/datasets/utils/track.py
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,6 @@ def __init__(self) -> None:

def __repr__(self) -> str:
if self.last_item is None:
super().__repr__()
return super().__repr__()
else:
return f"{self.__class__.__name__}(current={self.last_item})"

0 comments on commit 6e380f7

Please sign in to comment.