apache · mengxr · Nov 14, 2014 · Nov 14, 2014 · davies · Nov 14, 2014
diff --git a/python/pyspark/context.py b/python/pyspark/context.py
@@ -289,12 +289,31 @@ def stop(self):
 
     def parallelize(self, c, numSlices=None):
         """
-        Distribute a local Python collection to form an RDD.
-
-        >>> sc.parallelize(range(5), 5).glom().collect()
-        [[0], [1], [2], [3], [4]]
-        """
-        numSlices = numSlices or self.defaultParallelism
+        Distribute a local Python collection to form an RDD. Use xrange if
+        the input represents a range for performance.
+
+        >>> sc.parallelize([0, 2, 3, 4, 6], 5).glom().collect()
+        [[0], [2], [3], [4], [6]]
+        >>> sc.parallelize(xrange(0, 6, 2), 5).glom().collect()
+        [[], [0], [], [2], [4]]
+        """
+        numSlices = int(numSlices) if numSlices is not None else self.defaultParallelism
+        if isinstance(c, xrange):
+            size = len(c)
+            if size == 0:
+                return self.parallelize([], numSlices)
+            step = c[1] - c[0] if size > 1 else 1
+            c1 = xrange(c[0], c[0] + (size + 1) * step, step)
+
+            def getStartIndex(split):
+                return split * size / numSlices
+
+            def f(split, iterator):
+                startIndex = getStartIndex(split)
+                endIndex = getStartIndex(split + 1)
+                return xrange(c1[startIndex], c1[endIndex], step)
+
+            return self.parallelize([], numSlices).mapPartitionsWithIndex(f)
         # Calling the Java parallelize() method with an ArrayList is too slow,
         # because it sends O(n) Py4J commands.  As an alternative, serialized
         # objects are written to a file and loaded through textFile().