| Stage Id ▾ | Description | Submitted | Duration | Tasks: Succeeded/Total | Input | Output | Shuffle Read | Shuffle Write |
|---|---|---|---|---|---|---|---|---|
| 26 | parquet at NativeMethodAccessorImpl.java:0
RDD: AdaptiveSparkPlan isFinalPlan=false
+- Filter (isnotnull(is_ride_id_valid#474) AND ((is_ride_id_valid#474 AND is_duration_valid#475) AND is_station_valid#476))
+- Window [row_number() windowspecdefinition(start_station_id#31, started_at#28 ASC NULLS FIRST, specifiedwindowframe(RowFrame, unboundedpreceding$(), currentrow$())) AS _start_station_ride_num#536], [start_station_id#31], [started_at#28 ASC NULLS FIRST]
+- Sort [start_station_id#31 ASC NULLS FIRST, started_at#28 ASC NULLS FIRST], false, 0
+- Exchange hashpartitioning(start_station_id#31, 200), ENSURE_REQUIREMENTS, [plan_id=68]
+- Project [ride_id#26, rideable_type#27, started_at#28, ended_at#29, start_station_name#30, start_station_id#31, end_station_name#32, end_station_id#33, start_lat#34, start_lng#35, end_lat#36, end_lng#37, member_casual#38, is_ride_id_valid#474, is_duration_valid#475, is_station_valid#476, _source_file#493, 2026-08-18 07:26:21.132535 AS _processed_dttm#494, year(cast(started_at#28 as date)) AS ye...
org.apache.spark.sql.DataFrameWriter.parquet(DataFrameWriter.scala:792) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:27:34 | 31 s |
6/6
| 306.3 MiB | 198.8 MiB | ||
| 24 | csv at NativeMethodAccessorImpl.java:0
RDD: AdaptiveSparkPlan isFinalPlan=false
+- Filter ((NOT is_ride_id_valid#474 OR NOT is_duration_valid#475) OR NOT is_station_valid#476)
+- Window [row_number() windowspecdefinition(start_station_id#31, started_at#28 ASC NULLS FIRST, specifiedwindowframe(RowFrame, unboundedpreceding$(), currentrow$())) AS _start_station_ride_num#536], [start_station_id#31], [started_at#28 ASC NULLS FIRST]
+- Sort [start_station_id#31 ASC NULLS FIRST, started_at#28 ASC NULLS FIRST], false, 0
+- Exchange hashpartitioning(start_station_id#31, 200), ENSURE_REQUIREMENTS, [plan_id=91]
+- Project [ride_id#26, rideable_type#27, started_at#28, ended_at#29, start_station_name#30, start_station_id#31, end_station_name#32, end_station_id#33, start_lat#34, start_lng#35, end_lat#36, end_lng#37, member_casual#38, is_ride_id_valid#474, is_duration_valid#475, is_station_valid#476, _source_file#493, 2026-08-18 07:26:21.350521 AS _processed_dttm#494, year(cast(started_at#28 as date)) AS year#513, month(cast(started_a...
org.apache.spark.sql.DataFrameWriter.csv(DataFrameWriter.scala:850) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:27:25 | 5 s |
6/6
| 8.1 MiB | 31.6 MiB | ||
| 22 | count at NativeMethodAccessorImpl.java:0 org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:27:24 | 25 ms |
1/1
| 354.0 B | |||
| 19 | count at NativeMethodAccessorImpl.java:0
RDD: AdaptiveSparkPlan isFinalPlan=false
+- Filter ((NOT is_ride_id_valid#474 OR NOT is_duration_valid#475) OR NOT is_station_valid#476)
+- Window [row_number() windowspecdefinition(start_station_id#31, started_at#28 ASC NULLS FIRST, specifiedwindowframe(RowFrame, unboundedpreceding$(), currentrow$())) AS _start_station_ride_num#536], [start_station_id#31], [started_at#28 ASC NULLS FIRST]
+- Sort [start_station_id#31 ASC NULLS FIRST, started_at#28 ASC NULLS FIRST], false, 0
+- Exchange hashpartitioning(start_station_id#31, 200), ENSURE_REQUIREMENTS, [plan_id=91]
+- Project [ride_id#26, rideable_type#27, started_at#28, ended_at#29, start_station_name#30, start_station_id#31, end_station_name#32, end_station_id#33, start_lat#34, start_lng#35, end_lat#36, end_lng#37, member_casual#38, is_ride_id_valid#474, is_duration_valid#475, is_station_valid#476, _source_file#493, 2026-08-18 07:26:21.350521 AS _processed_dttm#494, year(cast(started_at#28 as date)) AS year#513, month(cast(started_a...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:27:24 | 67 ms |
6/6
| 8.1 MiB | 354.0 B | ||
| 17 | count at NativeMethodAccessorImpl.java:0
RDD: AdaptiveSparkPlan isFinalPlan=false
+- Filter ((NOT is_ride_id_valid#474 OR NOT is_duration_valid#475) OR NOT is_station_valid#476)
+- Window [row_number() windowspecdefinition(start_station_id#31, started_at#28 ASC NULLS FIRST, specifiedwindowframe(RowFrame, unboundedpreceding$(), currentrow$())) AS _start_station_ride_num#536], [start_station_id#31], [started_at#28 ASC NULLS FIRST]
+- Sort [start_station_id#31 ASC NULLS FIRST, started_at#28 ASC NULLS FIRST], false, 0
+- Exchange hashpartitioning(start_station_id#31, 200), ENSURE_REQUIREMENTS, [plan_id=91]
+- Project [ride_id#26, rideable_type#27, started_at#28, ended_at#29, start_station_name#30, start_station_id#31, end_station_name#32, end_station_id#33, start_lat#34, start_lng#35, end_lat#36, end_lng#37, member_casual#38, is_ride_id_valid#474, is_duration_valid#475, is_station_valid#476, _source_file#493, 2026-08-18 07:26:21.350521 AS _processed_dttm#494, year(cast(started_at#28 as date)) AS year#513, month(cast(started_a...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:27:13 | 11 s |
6/6
| 343.7 MiB | |||
| 15 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:27:05 | 8 s |
8/8
| 335.8 MiB | 343.7 MiB | ||
| 14 | count at NativeMethodAccessorImpl.java:0 org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:27:05 | 26 ms |
1/1
| 354.0 B | |||
| 11 | count at NativeMethodAccessorImpl.java:0
RDD: AdaptiveSparkPlan isFinalPlan=false
+- Filter (isnotnull(is_ride_id_valid#474) AND ((is_ride_id_valid#474 AND is_duration_valid#475) AND is_station_valid#476))
+- Window [row_number() windowspecdefinition(start_station_id#31, started_at#28 ASC NULLS FIRST, specifiedwindowframe(RowFrame, unboundedpreceding$(), currentrow$())) AS _start_station_ride_num#536], [start_station_id#31], [started_at#28 ASC NULLS FIRST]
+- Sort [start_station_id#31 ASC NULLS FIRST, started_at#28 ASC NULLS FIRST], false, 0
+- Exchange hashpartitioning(start_station_id#31, 200), ENSURE_REQUIREMENTS, [plan_id=68]
+- Project [ride_id#26, rideable_type#27, started_at#28, ended_at#29, start_station_name#30, start_station_id#31, end_station_name#32, end_station_id#33, start_lat#34, start_lng#35, end_lat#36, end_lng#37, member_casual#38, is_ride_id_valid#474, is_duration_valid#475, is_station_valid#476, _source_file#493, 2026-08-18 07:26:21.132535 AS _processed_dttm#494, year(cast(started_at#28 as date)) AS ye...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:27:05 | 0.1 s |
6/6
| 306.3 MiB | 354.0 B | ||
| 9 | count at NativeMethodAccessorImpl.java:0
RDD: AdaptiveSparkPlan isFinalPlan=false
+- Filter (isnotnull(is_ride_id_valid#474) AND ((is_ride_id_valid#474 AND is_duration_valid#475) AND is_station_valid#476))
+- Window [row_number() windowspecdefinition(start_station_id#31, started_at#28 ASC NULLS FIRST, specifiedwindowframe(RowFrame, unboundedpreceding$(), currentrow$())) AS _start_station_ride_num#536], [start_station_id#31], [started_at#28 ASC NULLS FIRST]
+- Sort [start_station_id#31 ASC NULLS FIRST, started_at#28 ASC NULLS FIRST], false, 0
+- Exchange hashpartitioning(start_station_id#31, 200), ENSURE_REQUIREMENTS, [plan_id=68]
+- Project [ride_id#26, rideable_type#27, started_at#28, ended_at#29, start_station_name#30, start_station_id#31, end_station_name#32, end_station_id#33, start_lat#34, start_lng#35, end_lat#36, end_lng#37, member_casual#38, is_ride_id_valid#474, is_duration_valid#475, is_station_valid#476, _source_file#493, 2026-08-18 07:26:21.132535 AS _processed_dttm#494, year(cast(started_at#28 as date)) AS ye...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:26:30 | 34 s |
6/6
| 343.8 MiB | |||
| 7 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:26:22 | 8 s |
8/8
| 335.8 MiB | 343.8 MiB | ||
| 6 | count at NativeMethodAccessorImpl.java:0 org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:26:21 | 87 ms |
1/1
| 472.0 B | |||
| 4 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:26:21 | 0.2 s |
8/8
| 335.8 MiB | 472.0 B | ||
| 3 | count at NativeMethodAccessorImpl.java:0 org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:26:20 | 0.3 s |
1/1
| 472.0 B | |||
| 1 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:26:19 | 0.5 s |
8/8
| 335.8 MiB | 472.0 B | ||
| 0 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | 2026/08/18 07:25:35 | 44 s |
8/8
| 954.0 MiB |
| Stage Id ▾ | Description | Submitted | Duration | Tasks: Succeeded/Total | Input | Output | Shuffle Read | Shuffle Write |
|---|---|---|---|---|---|---|---|---|
| 25 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/8
| ||||
| 23 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/8
| ||||
| 21 | count at NativeMethodAccessorImpl.java:0
RDD: AdaptiveSparkPlan isFinalPlan=false
+- Filter ((NOT is_ride_id_valid#474 OR NOT is_duration_valid#475) OR NOT is_station_valid#476)
+- Window [row_number() windowspecdefinition(start_station_id#31, started_at#28 ASC NULLS FIRST, specifiedwindowframe(RowFrame, unboundedpreceding$(), currentrow$())) AS _start_station_ride_num#536], [start_station_id#31], [started_at#28 ASC NULLS FIRST]
+- Sort [start_station_id#31 ASC NULLS FIRST, started_at#28 ASC NULLS FIRST], false, 0
+- Exchange hashpartitioning(start_station_id#31, 200), ENSURE_REQUIREMENTS, [plan_id=91]
+- Project [ride_id#26, rideable_type#27, started_at#28, ended_at#29, start_station_name#30, start_station_id#31, end_station_name#32, end_station_id#33, start_lat#34, start_lng#35, end_lat#36, end_lng#37, member_casual#38, is_ride_id_valid#474, is_duration_valid#475, is_station_valid#476, _source_file#493, 2026-08-18 07:26:21.350521 AS _processed_dttm#494, year(cast(started_at#28 as date)) AS year#513, month(cast(started_a...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/6
| ||||
| 20 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/8
| ||||
| 18 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/8
| ||||
| 16 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/8
| ||||
| 13 | count at NativeMethodAccessorImpl.java:0
RDD: AdaptiveSparkPlan isFinalPlan=false
+- Filter (isnotnull(is_ride_id_valid#474) AND ((is_ride_id_valid#474 AND is_duration_valid#475) AND is_station_valid#476))
+- Window [row_number() windowspecdefinition(start_station_id#31, started_at#28 ASC NULLS FIRST, specifiedwindowframe(RowFrame, unboundedpreceding$(), currentrow$())) AS _start_station_ride_num#536], [start_station_id#31], [started_at#28 ASC NULLS FIRST]
+- Sort [start_station_id#31 ASC NULLS FIRST, started_at#28 ASC NULLS FIRST], false, 0
+- Exchange hashpartitioning(start_station_id#31, 200), ENSURE_REQUIREMENTS, [plan_id=68]
+- Project [ride_id#26, rideable_type#27, started_at#28, ended_at#29, start_station_name#30, start_station_id#31, end_station_name#32, end_station_id#33, start_lat#34, start_lng#35, end_lat#36, end_lng#37, member_casual#38, is_ride_id_valid#474, is_duration_valid#475, is_station_valid#476, _source_file#493, 2026-08-18 07:26:21.132535 AS _processed_dttm#494, year(cast(started_at#28 as date)) AS ye...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/6
| ||||
| 12 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/8
| ||||
| 10 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/8
| ||||
| 8 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/8
| ||||
| 5 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/8
| ||||
| 2 | count at NativeMethodAccessorImpl.java:0
RDD: FileScan csv [ride_id#0,rideable_type#1,started_at#2,ended_at#3,start_station_name#4,start_station_id#5,end_station_name#6,end_station_id#7,start_lat#8,start_lng#9,end_lat#10,end_lng#11,member_casual#12] Batched: false, DataFilters: [], Format: CSV, Location: InMemoryFileIndex(6 paths)[s3a://rzvde-g10-chepusov-vladislav/raw/citibike_data/202508/202508-cit..., PartitionFilters: [], PushedFilters: [], ReadSchema: struct<ride_id:string,rideable_type:string,started_at:timestamp,ended_at:timestamp,start_station_...
org.apache.spark.sql.Dataset.count(Dataset.scala:3625) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke0(Native Method) java.base/jdk.internal.reflect.NativeMethodAccessorImpl.invoke(NativeMethodAccessorImpl.java:77) java.base/jdk.internal.reflect.DelegatingMethodAccessorImpl.invoke(DelegatingMethodAccessorImpl.java:43) java.base/java.lang.reflect.Method.invoke(Method.java:568) py4j.reflection.MethodInvoker.invoke(MethodInvoker.java:244) py4j.reflection.ReflectionEngine.invoke(ReflectionEngine.java:374) py4j.Gateway.invoke(Gateway.java:282) py4j.commands.AbstractCommand.invokeMethod(AbstractCommand.java:132) py4j.commands.CallCommand.execute(CallCommand.java:79) py4j.ClientServerConnection.waitForCommands(ClientServerConnection.java:182) py4j.ClientServerConnection.run(ClientServerConnection.java:106) java.base/java.lang.Thread.run(Thread.java:840) | Unknown | Unknown |
0/8
|