-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathJSON.py
More file actions
74 lines (54 loc) · 2.75 KB
/
Copy pathJSON.py
File metadata and controls
74 lines (54 loc) · 2.75 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
import requests # For fetching data from the web
from bs4 import BeautifulSoup # For parsing web content (not needed in this case)
import pandas as pd # For handling and analyzing data
import numpy as np # For numerical operations
import re # For regular expressions (cleaning data)
import matplotlib.pyplot as plt # For visualizations
import seaborn as sns # For statistical data visualization
# ------------------- Step 1: Fetch JSON Data ------------------- #
# URL containing the JSON dataset
url = "https://raw.githubusercontent.com/ozlerhakan/mongodb-json-files/master/datasets/grades.json"
# Fetching data from the URL
req = requests.get(url)
# Reading JSON data into a Pandas DataFrame
df = pd.read_json(url, lines=True)
# ------------------- Step 2: Data Cleaning ------------------- #
# Function to clean unwanted characters from the '_id' column
def clean_id(value):
return re.sub(r"^{.*: '|'}", " ", value)
# Converting '_id' to string type and cleaning it
df['_id'] = df['_id'].astype(str).apply(clean_id)
# Converting relevant columns to string type
df['student_id'] = df['student_id'].astype(str)
df['class_id'] = df['class_id'].astype(str)
# ------------------- Step 3: Extracting Scores ------------------- #
# The 'scores' column contains a list of dictionaries (exam, quiz, homework)
# We need to extract the 'score' values from this list
df['exam_score'] = df['scores'].apply(lambda x: x[0]['score'] if isinstance(x, list) else None) # Extract exam score
df['quiz_score'] = df['scores'].apply(lambda x: x[1]['score'] if isinstance(x, list) else None) # Extract quiz score
df['homework_score'] = df['scores'].apply(lambda x: x[2]['score'] if isinstance(x, list) else None) # Extract homework score
# Dropping the 'scores' column as we have extracted the required data
df.drop(columns=['scores'], inplace=True)
# ------------------- Step 4: Display Cleaned Data ------------------- #
print(df.head()) # Display first few rows of cleaned data
# ------------------- Step 5: Data Visualization ------------------- #
# Setting up the style for plots
sns.set_style("whitegrid")
# Histogram of Exam Scores
plt.figure(figsize=(8, 5))
sns.histplot(df['exam_score'], bins=20, kde=True, color='blue')
plt.title("Distribution of Exam Scores")
plt.xlabel("Exam Score")
plt.ylabel("Frequency")
plt.show()
# Scatter Plot: Exam Score vs Quiz Score
plt.figure(figsize=(8, 5))
sns.scatterplot(x=df['exam_score'], y=df['quiz_score'], color='red')
plt.title("Exam Score vs Quiz Score")
plt.xlabel("Exam Score")
plt.ylabel("Quiz Score")
plt.show()
# ------------------- Step 6: Statistical Insights ------------------- #
# Calculate the mean, median, and standard deviation for scores
stats_summary = df[['exam_score', 'quiz_score', 'homework_score']].describe()
print(stats_summary)