Skip to content

Commit 023b1ba

Browse files
committed
Converted date_aired to datetime object and BSON Date for MongoDB
1 parent 4ae5e2e commit 023b1ba

3 files changed

Lines changed: 8 additions & 2 deletions

File tree

‎backend/mongo/AnimeDocument.py‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
11
from pydantic import BaseModel
2+
from datetime import datetime
23
from typing import List, Dict
34

45
# Class to Model a typical Anime Document
@@ -12,7 +13,7 @@ class AnimeDocument(BaseModel):
1213
demographic: str # Primary demographic
1314
age_rating: str # g | pg | pg-13 | r | r+ | rx
1415
cover_image_url: str # link to MAL image (not CDN)
15-
date_aired: str # date aired in YYYY-MM-DD
16+
date_aired: datetime # date aired as datetime (YYYY-MM-DD stored, not hr/mins/secs/ms)
1617
status: str # finished_airing | currently_airing | not_aired
1718
episode_count: int # number of episodes in the anime
1819
avg_episode_len_mins: int # average duration per episode in mins

‎data/anime_scrape.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
11
import json
22
import time
3+
from datetime import datetime
34
from typing import List, Dict
45

56
import pandas as pd
@@ -48,6 +49,7 @@ def _normalize_from_mal(item: Dict) -> Dict:
4849
studios = node.get("studios", [])
4950
publishing_company = studios[0].get("name", "Unknown") if studios else "Unknown"
5051
synopsis = node.get("synopsis").replace("[Written by MAL Rewrite]", "").strip()
52+
date_aired = (datetime.strptime(node.get("start_date"), "%Y-%m-%d"))
5153

5254
# Alt titles (in case main is not en)
5355
alt_titles = node.get("alternative_titles")
@@ -67,7 +69,7 @@ def _normalize_from_mal(item: Dict) -> Dict:
6769
"genres": genres,
6870
"demographic": demographic,
6971
"cover_image_url": node.get("main_picture").get("medium", ""),
70-
"date_aired": node.get("start_date"),
72+
"date_aired": date_aired,
7173
"status": node.get("status", "not_aired"),
7274
"episode_count": node.get("num_episodes", 0),
7375
"publishing_company": publishing_company,

‎data/populate_mongo_db_scraped.py‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -30,6 +30,9 @@ def clean_data():
3030
# Project current name as its own node_name column (since they may not be english, but useful data)
3131
anime_df["node_name"] = anime_df["name"]
3232

33+
# Convert date_aired to datetime
34+
anime_df["date_aired"] = pd.to_datetime(anime_df["date_aired"], format="mixed")
35+
3336
# Filter out animes without English Names (check alt_titles dict for "en" key, and ensure it is not empty value)
3437
anime_df = anime_df[anime_df["alt_titles"].apply(lambda x: "en" in x and x["en"] != "")]
3538

0 commit comments

Comments
 (0)