Repository navigation
Expand file tree
/
Copy pathinspect_data.py
More file actions
118 lines (90 loc) · 3.42 KB
/
Copy pathinspect_data.py
File metadata and controls
118 lines (90 loc) · 3.42 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
import os
import sys
import pandas as pd
def load_data(data_path: str) -> pd.DataFrame:
"""Load the dataset from the given path."""
try:
df = pd.read_csv(
data_path, compression="gzip" if data_path.endswith(".gz") else None
)
print(f"Loaded data with {len(df)} records from {data_path}.")
return df
except Exception as e: # noqa: BLE001
print(f"Error loading data from {data_path}: {e}")
sys.exit(1)
def get_timestamp_range(df: pd.DataFrame) -> None:
"""Get the timestamp range in epoch and human-readable format."""
min_timestamp = df["timestamp"].min()
max_timestamp = df["timestamp"].max()
min_datetime = pd.to_datetime(min_timestamp, unit="s", utc=True)
max_datetime = pd.to_datetime(max_timestamp, unit="s", utc=True)
print("Timestamp Range:")
print(f" From: {min_timestamp} ({min_datetime})")
print(f" To: {max_timestamp} ({max_datetime})")
def print_data_schema(df: pd.DataFrame) -> None:
"""Print the data schema."""
print("\nData Schema:")
print(df.dtypes)
def check_missing_values(df: pd.DataFrame) -> None:
"""Check and print missing values per column."""
missing = df.isnull().sum()
print("\nMissing Values per Column:")
print(missing)
def check_duplicates(df: pd.DataFrame) -> None:
"""Check and print the number of duplicate timestamps."""
duplicates = df.duplicated(subset="timestamp").sum()
if duplicates > 0:
print(f"\nNumber of Duplicate Timestamps: {duplicates}")
else:
print("\nNo duplicate timestamps found.")
def print_descriptive_statistics(df: pd.DataFrame) -> None:
"""Print descriptive statistics of the dataset."""
print("\nDescriptive Statistics:")
print(df.describe())
def print_sample_rows(df: pd.DataFrame) -> None:
"""Print first and last few rows of the dataset."""
print("\nFirst 5 rows:")
print(df.head())
print("\nMost recent 5 rows:")
print(df.tail())
def main() -> None:
# Determine which dataset to inspect based on command-line argument
data_type = sys.argv[1] if len(sys.argv) > 1 else "merged"
# Define paths to the datasets
data_paths = {
"bulk": os.path.join(
"data", "historical", "btcusd_bitstamp_1min_2012-2025.csv.gz"
),
"updated": os.path.join("data", "updates", "btcusd_bitstamp_1min_latest.csv"),
}
# Load and merge datasets if 'merged' is selected
if data_type == "merged":
bulk_df = load_data(data_paths["bulk"])
updated_df = load_data(data_paths["updated"])
df = pd.concat([bulk_df, updated_df])
print(f"Merged data with {len(df)} records.")
else:
# Get the path for the selected dataset
data_path = data_paths.get(data_type)
if not data_path:
print(
f"Invalid data type specified: {data_type}."
"Choose from 'bulk', 'updated', or 'merged'."
)
sys.exit(1)
# Load the selected data
df = load_data(data_path)
# Get and print the timestamp range
get_timestamp_range(df)
# Print the data schema
print_data_schema(df)
# Check for missing values
check_missing_values(df)
# Check for duplicate timestamps
check_duplicates(df)
# Print descriptive statistics
print_descriptive_statistics(df)
# Print sample rows
print_sample_rows(df)
if __name__ == "__main__":
main()