diff --git a/.gitignore b/.gitignore index 7b004e5..4866162 100644 --- a/.gitignore +++ b/.gitignore @@ -191,4 +191,18 @@ cython_debug/ # exclude from AI features like autocomplete and code analysis. Recommended for sensitive data # refer to https://docs.cursor.com/context/ignore-files .cursorignore -.cursorindexingignore \ No newline at end of file +.cursorindexingignore + +# LMS + +*.json +test.Movies-1-final-no-keywords.ipynb +test.Movies-1-fixed.ipynb +test.final_movies.csv +test.keywords.csv +test.links.csv +test.links_small.csv +test.ratings.csv +test.ratings_small.csv +test.tcc_ceds_music.csv +test.music_data.json diff --git a/.vscode/settings.json b/.vscode/settings.json new file mode 100644 index 0000000..a456028 --- /dev/null +++ b/.vscode/settings.json @@ -0,0 +1,3 @@ +{ + "python-envs.defaultEnvManager": "ms-python.python:pyenv" +} \ No newline at end of file diff --git a/data/booksout.json b/data/booksout.json new file mode 100644 index 0000000..7c4df3c --- /dev/null +++ b/data/booksout.json @@ -0,0 +1,941 @@ +[ + { + "title": "The Declaration of Independence of the United States of America", + "authors": [ + "Jefferson, Thomas" + ], + "author_lifespan": [ + "1743-1826" + ], + "isbn": "2237017484115", + "numberOfPages": null, + "genre": null, + "subjects": [ + "United States -- History -- Revolution, 1775-1783 -- Sources", + "United States. Declaration of Independence" + ], + "locc": "E201; JK" + }, + { + "title": "The United States Bill of Rights\nThe Ten Original Amendments to the Constitution of the United States", + "authors": [ + "United States" + ], + "author_lifespan": [], + "isbn": "9811626450377", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Civil rights -- United States -- Sources", + "United States. Constitution. 1st-10th Amendments" + ], + "locc": "JK; KF" + }, + { + "title": "John F. Kennedy'sMilton", + "authors": [ + "en" + ], + "author_lifespan": [], + "isbn": "9751360059472", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Kennedy, John F. (John Fitzgerald), 1917-1963" + ], + "locc": "United States -- Foreign relations -- 1961-1963; Presidents -- United States -- Inaugural addresses" + }, + { + "title": "Lincoln's Gettysburg Address\nGiven November 19, 1863 on the battlefield near Gettysburg, Pennsylvania, USA", + "authors": [ + "Lincoln, Abraham" + ], + "author_lifespan": [ + "1809-1865" + ], + "isbn": "7537213558237", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Consecration of cemeteries -- Pennsylvania -- Gettysburg", + "Soldiers' National Cemetery (Gettysburg, Pa.)", + "Lincoln, Abraham, 1809-1865. Gettysburg address" + ], + "locc": "E456" + }, + { + "title": "The United States Constitution", + "authors": [ + "United States" + ], + "author_lifespan": [], + "isbn": "4271978397752", + "numberOfPages": null, + "genre": null, + "subjects": [ + "United States -- Politics and government -- 1783-1789 -- Sources", + "United States. Constitution" + ], + "locc": "JK; KF" + }, + { + "title": "Give Me Liberty or Give Me Death", + "authors": [ + "Henry, Patrick" + ], + "author_lifespan": [ + "1736-1799" + ], + "isbn": "7799634185306", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Speeches, addresses, etc., American", + "United States -- Politics and government -- 1775-1783 -- Sources", + "Virginia -- Politics and government -- 1775-1783 -- Sources" + ], + "locc": "E201" + }, + { + "title": "Abraham Lincoln's Second Inaugural Address", + "authors": [ + "Lincoln, Abraham" + ], + "author_lifespan": [ + "1809-1865" + ], + "isbn": "1123563265749", + "numberOfPages": null, + "genre": null, + "subjects": [ + "United States -- Politics and government -- 1861-1865", + "Presidents -- United States -- Inaugural addresses" + ], + "locc": "E456" + }, + { + "title": "Abraham Lincoln's First Inaugural Address", + "authors": [ + "Lincoln, Abraham" + ], + "author_lifespan": [ + "1809-1865" + ], + "isbn": "3467959189888", + "numberOfPages": null, + "genre": null, + "subjects": [ + "United States -- Politics and government -- 1861-1865", + "Presidents -- United States -- Inaugural addresses" + ], + "locc": "E456" + }, + { + "title": "Alice's Adventures in Wonderland", + "authors": [ + "Carroll, Lewis" + ], + "author_lifespan": [ + "1832-1898" + ], + "isbn": "9777771910407", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Fantasy fiction", + "Children's stories", + "Imaginary places -- Juvenile fiction", + "Alice (Fictitious character from Carroll) -- Juvenile fiction" + ], + "locc": "PR; PZ" + }, + { + "title": "Through the Looking-Glass", + "authors": [ + "Carroll, Lewis" + ], + "author_lifespan": [ + "1832-1898" + ], + "isbn": "6292519052466", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Fantasy fiction", + "Children's stories", + "Imaginary places -- Juvenile fiction", + "Alice (Fictitious character from Carroll) -- Juvenile fiction" + ], + "locc": "PR; PZ" + }, + { + "title": "The Hunting of the Snark: An Agony in Eight Fits", + "authors": [ + "Carroll, Lewis" + ], + "author_lifespan": [ + "1832-1898" + ], + "isbn": "1002432292942", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Nonsense verses, English" + ], + "locc": "PR" + }, + { + "title": "The 1990 CIA World Factbook", + "authors": [ + "United States. Central Intelligence Agency" + ], + "author_lifespan": [], + "isbn": "5078132711984", + "numberOfPages": null, + "genre": null, + "subjects": [ + "World politics -- Handbooks, manuals, etc.", + "Geography -- Handbooks, manuals, etc.", + "Political science -- Handbooks, manuals, etc.", + "Political statistics -- Handbooks, manuals, etc." + ], + "locc": "G" + }, + { + "title": "Moby-Dick; or, The Whale", + "authors": [ + "Melville, Herman" + ], + "author_lifespan": [ + "1819-1891" + ], + "isbn": "3536847369637", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Whaling -- Fiction", + "Sea stories", + "Psychological fiction", + "Ship captains -- Fiction", + "Adventure stories", + "Mentally ill -- Fiction", + "Ahab, Captain (Fictitious character) -- Fiction", + "Whales -- Fiction", + "Whaling ships -- Fiction" + ], + "locc": "PS" + }, + { + "title": "Peter Pan", + "authors": [ + "Barrie, J. M. (James Matthew)" + ], + "author_lifespan": [ + "1860-1937" + ], + "isbn": "7916834897771", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Fantasy literature", + "Peter Pan (Fictitious character) -- Fiction", + "Never-Never Land (Imaginary place) -- Fiction", + "Pirates -- Fiction", + "Fairies -- Fiction" + ], + "locc": "PR; PZ" + }, + { + "title": "The Book of Mormon", + "authors": [ + "Smith, Joseph, Jr.", + "Church of Jesus Christ of Latter-day Saints" + ], + "author_lifespan": [ + "1805-1844" + ], + "isbn": "4710435027331", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Church of Jesus Christ of Latter-day Saints -- Sacred books", + "Latter Day Saint churches -- Sacred books" + ], + "locc": "BX" + }, + { + "title": "The Federalist Papers", + "authors": [ + "Hamilton, Alexander", + "Jay, John", + "Madison, James" + ], + "author_lifespan": [ + "1757-1804", + "1745-1829", + "1751-1836" + ], + "isbn": "8957053841387", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Constitutional history -- United States -- Sources", + "Constitutional law -- United States" + ], + "locc": "JK; KF" + }, + { + "title": "The Song of Hiawatha", + "authors": [ + "Longfellow, Henry Wadsworth", + "Morris, Woodrow W. [Editor]" + ], + "author_lifespan": [ + "1807-1882" + ], + "isbn": "6532892231626", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Indians of North America -- Poetry", + "Hiawatha, active 15th century -- Poetry", + "Iroquois Indians -- Kings and rulers -- Poetry" + ], + "locc": "PS" + }, + { + "title": "Paradise Lost", + "authors": [ + "" + ], + "author_lifespan": [ + "1608-1674" + ], + "isbn": "3589744407298", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Fall of man -- Poetry", + "Adam (Biblical figure) -- Poetry", + "Eve (Biblical figure) -- Poetry", + "Bible. Genesis -- History of Biblical events -- Poetry" + ], + "locc": "PR" + }, + { + "title": "Aesop's Fables\nTranslated by George Fyler Townsend", + "authors": [ + "Aesop", + "Townsend, George Fyler, [Translator]" + ], + "author_lifespan": [ + "621? BCE-565? BCE", + "1814-1900" + ], + "isbn": "9165562697694", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Fables, Greek -- Translations into English", + "Aesop's fables -- Translations into English" + ], + "locc": "PA; PZ" + }, + { + "title": "Roget's Thesaurus", + "authors": [ + "Roget, Peter Mark" + ], + "author_lifespan": [ + "1779-1869" + ], + "isbn": "8831346331568", + "numberOfPages": null, + "genre": null, + "subjects": [ + "English language -- Synonyms and antonyms" + ], + "locc": "PE" + }, + { + "title": "Narrative of the Life of Frederick Douglass, an American Slave", + "authors": [ + "Douglass, Frederick" + ], + "author_lifespan": [ + "1818-1895" + ], + "isbn": "4668666238824", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Douglass, Frederick, 1818-1895", + "African American abolitionists -- Biography", + "Abolitionists -- United States -- Biography", + "Enslaved persons -- United States -- Biography" + ], + "locc": "E300" + }, + { + "title": "O Pioneers!", + "authors": [ + "Cather, Willa" + ], + "author_lifespan": [ + "1873-1947" + ], + "isbn": "9030297296467", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Nebraska -- Fiction", + "Historical fiction", + "Frontier and pioneer life -- Nebraska -- Fiction", + "Domestic fiction", + "Siblings -- Fiction", + "Farm life -- Fiction", + "Women pioneers -- Fiction", + "Women farmers -- Fiction", + "Women immigrants -- Fiction", + "Swedish Americans -- Fiction" + ], + "locc": "PS" + }, + { + "title": "The 1991 CIA World Factbook", + "authors": [ + "United States. Central Intelligence Agency" + ], + "author_lifespan": [], + "isbn": "4042864181615", + "numberOfPages": null, + "genre": null, + "subjects": [ + "World politics -- Handbooks, manuals, etc.", + "Geography -- Handbooks, manuals, etc.", + "Political science -- Handbooks, manuals, etc.", + "Political statistics -- Handbooks, manuals, etc." + ], + "locc": "G" + }, + { + "title": "Paradise Lost", + "authors": [ + "Milton, John" + ], + "author_lifespan": [ + "1608-1674" + ], + "isbn": "8201233846207", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Fall of man -- Poetry", + "Adam (Biblical figure) -- Poetry", + "Eve (Biblical figure) -- Poetry", + "Bible. Genesis -- History of Biblical events -- Poetry" + ], + "locc": "PR" + }, + { + "title": "Far from the Madding Crowd", + "authors": [ + "Hardy, Thomas" + ], + "author_lifespan": [ + "1840-1928" + ], + "isbn": "7153487373502", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Didactic fiction", + "Love stories", + "Triangles (Interpersonal relations) -- Fiction", + "Pastoral fiction", + "Farm life -- Fiction", + "Women farmers -- Fiction", + "Wessex (England) -- Fiction" + ], + "locc": "PR" + }, + { + "title": "The Fables of Aesop\nSelected, Told Anew, and Their History Traced", + "authors": [ + "Aesop", + "Jacobs, Joseph, [Editor]" + ], + "author_lifespan": [ + "621? BCE-565? BCE", + "1854-1916" + ], + "isbn": "1922531942229", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Aesop's fables -- Adaptations", + "Fables, Greek -- Adaptations" + ], + "locc": "PA; PZ" + }, + { + "title": "The 1990 United States Census", + "authors": [ + "United States. Bureau of the Census" + ], + "author_lifespan": [], + "isbn": "9954105217190", + "numberOfPages": null, + "genre": null, + "subjects": [ + "United States -- Population -- Statistics", + "United States -- Census", + "Housing -- United States -- Statistics" + ], + "locc": "HA" + }, + { + "title": "Plays of Sophocles: Oedipus the King; Oedipus at Colonus; Antigone", + "authors": [ + "Sophocles", + "Storr, Francis, [Translator]" + ], + "author_lifespan": [ + "496? BCE-407 BCE", + "1839-1919" + ], + "isbn": "2676174089663", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Tragedies (Drama)", + "Antigone (Mythological character) -- Drama", + "Oedipus (Greek mythological figure) -- Drama", + "Greek drama (Tragedy) -- Translations into English" + ], + "locc": "PA" + }, + { + "title": "Herland", + "authors": [ + "Gilman, Charlotte Perkins" + ], + "author_lifespan": [ + "1860-1935" + ], + "isbn": "5102797678482", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Utopias -- Fiction", + "Women -- Fiction", + "Utopian fiction", + "Black humor" + ], + "locc": "PS" + }, + { + "title": "The Scarlet Letter", + "authors": [ + "Hawthorne, Nathaniel" + ], + "author_lifespan": [ + "1804-1864" + ], + "isbn": "5929426941741", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Adultery -- Fiction", + "Historical fiction", + "Revenge -- Fiction", + "Psychological fiction", + "Married women -- Fiction", + "Clergy -- Fiction", + "Triangles (Interpersonal relations) -- Fiction", + "Illegitimate children -- Fiction", + "Women immigrants -- Fiction", + "Puritans -- Fiction", + "Boston (Mass.) -- History -- Colonial period, ca. 1600-1775 -- Fiction" + ], + "locc": "PS" + }, + { + "title": "Zen and the Art of the Internet", + "authors": [ + "Kehoe, Brendan P." + ], + "author_lifespan": [ + "1970-2011" + ], + "isbn": "6140866460321", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Internet", + "Computer networks", + "Information networks", + "Information retrieval" + ], + "locc": "TK" + }, + { + "title": "The Time Machine", + "authors": [ + "Wells, H. G. (Herbert George)" + ], + "author_lifespan": [ + "1866-1946" + ], + "isbn": "1188721753985", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Science fiction", + "Time travel -- Fiction", + "Dystopias -- Fiction" + ], + "locc": "PR" + }, + { + "title": "The War of the Worlds", + "authors": [ + "Wells, H. G. (Herbert George)" + ], + "author_lifespan": [ + "1866-1946" + ], + "isbn": "8616541508430", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Science fiction", + "War stories", + "Martians -- Fiction", + "Mars (Planet) -- Fiction", + "Space warfare -- Fiction", + "Imaginary wars and battles -- Fiction", + "Life on other planets -- Fiction" + ], + "locc": "PR" + }, + { + "title": "The 1990 United States Census [2nd]", + "authors": [ + "United States. Bureau of the Census" + ], + "author_lifespan": [], + "isbn": "2486898126289", + "numberOfPages": null, + "genre": null, + "subjects": [ + "United States -- Population -- Statistics", + "United States -- Census", + "Housing -- United States -- Statistics" + ], + "locc": "HA" + }, + { + "title": "The Jargon File, Version 2.9.10, 01 Jul 1992", + "authors": [ + "Raymond, Eric S., 1957- [Editor]", + "Steele, Guy L., 1954- [Editor]" + ], + "author_lifespan": [], + "isbn": "8651556212049", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Electronic data processing -- Terminology -- Humor", + "Computers -- Humor", + "Computers -- Slang -- Dictionaries" + ], + "locc": "TK" + }, + { + "title": "Hitchhiker's Guide to the Internet", + "authors": [ + "Krol, Ed, 1951-" + ], + "author_lifespan": [], + "isbn": "9609959971023", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Internet", + "Computer networks" + ], + "locc": "TK" + }, + { + "title": "NorthWestNet User Services Internet Resource Guide (NUSIRG)", + "authors": [ + "Kochmer, Jonathan" + ], + "author_lifespan": [], + "isbn": "7843542105530", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Internet" + ], + "locc": "TK" + }, + { + "title": "The Legend of Sleepy Hollow", + "authors": [ + "Irving, Washington" + ], + "author_lifespan": [ + "1783-1859" + ], + "isbn": "8650543195446", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Ghosts -- Fiction", + "New York (State) -- History -- 1775-1865 -- Fiction" + ], + "locc": "PS" + }, + { + "title": "The Strange Case of Dr. Jekyll and Mr. Hyde", + "authors": [ + "Stevenson, Robert Louis" + ], + "author_lifespan": [ + "1850-1894" + ], + "isbn": "4107873333680", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Science fiction", + "Horror tales", + "London (England) -- Fiction", + "Physicians -- Fiction", + "Psychological fiction", + "Self-experimentation in medicine -- Fiction", + "Multiple personality -- Fiction" + ], + "locc": "PR" + }, + { + "title": "The Strange Case of Dr. Jekyll and Mr. Hyde", + "authors": [ + "Stevenson, Robert Louis" + ], + "author_lifespan": [ + "1850-1894" + ], + "isbn": "6351027121864", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Science fiction", + "Horror tales", + "London (England) -- Fiction", + "Physicians -- Fiction", + "Psychological fiction", + "Self-experimentation in medicine -- Fiction", + "Multiple personality -- Fiction" + ], + "locc": "PR" + }, + { + "title": "The Song of the Lark", + "authors": [ + "Cather, Willa" + ], + "author_lifespan": [ + "1873-1947" + ], + "isbn": "8522575370919", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Musical fiction", + "Young women -- Fiction", + "Bildungsromans", + "Chicago (Ill.) -- Fiction", + "Swedish Americans -- Fiction", + "Colorado -- Fiction", + "Opera -- Fiction", + "Children of clergy -- Fiction", + "Women singers -- Fiction" + ], + "locc": "PS" + }, + { + "title": "Anne of Green Gables", + "authors": [ + "Montgomery, L. M. (Lucy Maud)" + ], + "author_lifespan": [ + "1874-1942" + ], + "isbn": "6468706802628", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Orphans -- Fiction", + "Islands -- Fiction", + "Friendship -- Fiction", + "Bildungsromans", + "Girls -- Fiction", + "Country life -- Prince Edward Island -- Fiction", + "Prince Edward Island -- History -- 20th century -- Fiction", + "Canada -- History -- 1867-1914 -- Fiction", + "Shirley, Anne (Fictitious character) -- Fiction" + ], + "locc": "PZ" + }, + { + "title": "A Christmas Carol in Prose; Being a Ghost Story of Christmas", + "authors": [ + "Dickens, Charles", + "Leech, John, [Illustrator]" + ], + "author_lifespan": [ + "1812-1870", + "1817-1864" + ], + "isbn": "8703527800721", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Christmas stories", + "London (England) -- Fiction", + "Poor families -- Fiction", + "Ghost stories", + "Misers -- Fiction", + "Sick children -- Fiction", + "Scrooge, Ebenezer (Fictitious character) -- Fiction" + ], + "locc": "PR" + }, + { + "title": "Anne of Avonlea", + "authors": [ + "Montgomery, L. M. (Lucy Maud)" + ], + "author_lifespan": [ + "1874-1942" + ], + "isbn": "3977497471549", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Orphans -- Fiction", + "Islands -- Fiction", + "Teachers -- Fiction", + "Prince Edward Island -- History -- 20th century -- Fiction", + "Canada -- History -- 1914-1945 -- Fiction", + "Shirley, Anne (Fictitious character) -- Fiction" + ], + "locc": "PZ" + }, + { + "title": "The 1992 CIA World Factbook", + "authors": [ + "United States. Central Intelligence Agency" + ], + "author_lifespan": [], + "isbn": "4611957319519", + "numberOfPages": null, + "genre": null, + "subjects": [ + "World politics -- Handbooks, manuals, etc.", + "Geography -- Handbooks, manuals, etc.", + "Political science -- Handbooks, manuals, etc.", + "Political statistics -- Handbooks, manuals, etc." + ], + "locc": "G" + }, + { + "title": "Surfing the Internet: An Introduction\nVersion 2.0.2", + "authors": [ + "Polly, Jean Armour" + ], + "author_lifespan": [], + "isbn": "8996148242469", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Internet addresses -- Directories", + "Computer network resources -- Directories", + "Electronic mail systems -- Directories" + ], + "locc": "TK" + }, + { + "title": "Pi", + "authors": [ + "Hemphill, Scott" + ], + "author_lifespan": [], + "isbn": "5582366585539", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Mathematics", + "Pi" + ], + "locc": "QA" + }, + { + "title": "Anne of the Island", + "authors": [ + "Montgomery, L. M. (Lucy Maud)" + ], + "author_lifespan": [ + "1874-1942" + ], + "isbn": "3410756933660", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Orphans -- Fiction", + "Prince Edward Island -- History -- 20th century -- Fiction", + "Interpersonal relations -- Fiction", + "Canada -- History -- 1914-1945 -- Fiction", + "Self-perception -- Fiction", + "Universities and colleges -- Fiction", + "Nova Scotia -- History -- 20th century -- Fiction", + "Shirley, Anne (Fictitious character) -- Fiction" + ], + "locc": "PZ" + }, + { + "title": "The Square Root of 2", + "authors": [ + "Kerr, Stan" + ], + "author_lifespan": [], + "isbn": "4107979155307", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Mathematics", + "Square root" + ], + "locc": "QA" + }, + { + "title": "Workshop on Electronic Texts: Proceedings, 9-10 June 1992", + "authors": [ + "Library of Congress", + "Daly, James, 1948- [Editor]" + ], + "author_lifespan": [], + "isbn": "7534381058902", + "numberOfPages": null, + "genre": null, + "subjects": [ + "Text processing (Computer science) -- Congresses", + "Electronic books", + "Electronic publishing -- Congresses" + ], + "locc": "Z" + } +] \ No newline at end of file diff --git a/data/dvdout.json b/data/dvdout.json new file mode 100644 index 0000000..4506d0a --- /dev/null +++ b/data/dvdout.json @@ -0,0 +1,402 @@ +[ + { + "recordID": "000001", + "title": "Toy Story", + "director": "John Lasseter", + "duration": "81.0", + "rating": "G", + "genre": "Animation, Comedy, Family" + }, + { + "recordID": "000002", + "title": "Jumanji", + "director": "Joe Johnston", + "duration": "104.0", + "rating": "PG", + "genre": "Adventure, Fantasy, Family" + }, + { + "recordID": "000003", + "title": "Grumpier Old Men", + "director": "Howard Deutch", + "duration": "101.0", + "rating": "PG-13", + "genre": "Romance, Comedy" + }, + { + "recordID": "000004", + "title": "Waiting to Exhale", + "director": "Forest Whitaker", + "duration": "127.0", + "rating": "R", + "genre": "Comedy, Drama, Romance" + }, + { + "recordID": "000005", + "title": "Father of the Bride Part II", + "director": "Charles Shyer", + "duration": "106.0", + "rating": "PG", + "genre": "Comedy" + }, + { + "recordID": "000006", + "title": "Heat", + "director": "Michael Mann", + "duration": "170.0", + "rating": "R", + "genre": "Action, Crime, Drama, Thriller" + }, + { + "recordID": "000007", + "title": "Sabrina", + "director": "Sydney Pollack", + "duration": "127.0", + "rating": "PG", + "genre": "Comedy, Romance" + }, + { + "recordID": "000008", + "title": "Tom and Huck", + "director": "Peter Hewitt", + "duration": "97.0", + "rating": "PG", + "genre": "Action, Adventure, Drama, Family" + }, + { + "recordID": "000009", + "title": "Sudden Death", + "director": "Peter Hyams", + "duration": "106.0", + "rating": "R", + "genre": "Action, Adventure, Thriller" + }, + { + "recordID": "000010", + "title": "GoldenEye", + "director": "Martin Campbell", + "duration": "130.0", + "rating": "PG-13", + "genre": "Adventure, Action, Thriller" + }, + { + "recordID": "000011", + "title": "The American President", + "director": "Rob Reiner", + "duration": "106.0", + "rating": "PG-13", + "genre": "Comedy, Drama, Romance" + }, + { + "recordID": "000012", + "title": "Dracula: Dead and Loving It", + "director": "Mel Brooks", + "duration": "88.0", + "rating": "PG-13", + "genre": "Comedy, Horror" + }, + { + "recordID": "000013", + "title": "Balto", + "director": "Simon Wells", + "duration": "78.0", + "rating": "G", + "genre": "Family, Animation, Adventure" + }, + { + "recordID": "000014", + "title": "Nixon", + "director": "Oliver Stone", + "duration": "192.0", + "rating": "R", + "genre": "History, Drama" + }, + { + "recordID": "000015", + "title": "Cutthroat Island", + "director": "Renny Harlin", + "duration": "119.0", + "rating": "PG-13", + "genre": "Action, Adventure" + }, + { + "recordID": "000016", + "title": "Casino", + "director": "Martin Scorsese", + "duration": "178.0", + "rating": "R", + "genre": "Drama, Crime" + }, + { + "recordID": "000017", + "title": "Sense and Sensibility", + "director": "Ang Lee", + "duration": "136.0", + "rating": "PG", + "genre": "Drama, Romance" + }, + { + "recordID": "000018", + "title": "Four Rooms", + "director": "Allison Anders", + "duration": "98.0", + "rating": "R", + "genre": "Crime, Comedy" + }, + { + "recordID": "000019", + "title": "Ace Ventura: When Nature Calls", + "director": "Steve Oedekerk", + "duration": "90.0", + "rating": "PG-13", + "genre": "Crime, Comedy, Adventure" + }, + { + "recordID": "000020", + "title": "Money Train", + "director": "Joseph Ruben", + "duration": "103.0", + "rating": "R", + "genre": "Action, Comedy, Crime" + }, + { + "recordID": "000021", + "title": "Get Shorty", + "director": "Barry Sonnenfeld", + "duration": "105.0", + "rating": "R", + "genre": "Comedy, Thriller, Crime" + }, + { + "recordID": "000022", + "title": "Copycat", + "director": "Jon Amiel", + "duration": "124.0", + "rating": "R", + "genre": "Drama, Thriller" + }, + { + "recordID": "000023", + "title": "Assassins", + "director": "Richard Donner", + "duration": "132.0", + "rating": "R", + "genre": "Action, Adventure, Crime, Thriller" + }, + { + "recordID": "000024", + "title": "Powder", + "director": "Victor Salva", + "duration": "111.0", + "rating": "PG-13", + "genre": "Drama, Fantasy, Science Fiction, Thriller" + }, + { + "recordID": "000025", + "title": "Leaving Las Vegas", + "director": "Mike Figgis", + "duration": "112.0", + "rating": "R", + "genre": "Drama, Romance" + }, + { + "recordID": "000026", + "title": "Othello", + "director": "Oliver Parker", + "duration": "123.0", + "rating": "R", + "genre": "Drama" + }, + { + "recordID": "000027", + "title": "Now and Then", + "director": "Lesli Linka Glatter", + "duration": "100.0", + "rating": "PG-13", + "genre": "Comedy, Drama, Family" + }, + { + "recordID": "000028", + "title": "Persuasion", + "director": "Roger Michell", + "duration": "104.0", + "rating": "PG", + "genre": "Drama, Romance" + }, + { + "recordID": "000029", + "title": "The City of Lost Children", + "director": "Jean-Pierre Jeunet", + "duration": "108.0", + "rating": "R", + "genre": "Fantasy, Science Fiction, Adventure" + }, + { + "recordID": "000030", + "title": "Shanghai Triad", + "director": "Zhang Yimou", + "duration": "108.0", + "rating": "R", + "genre": "Drama, Crime" + }, + { + "recordID": "000031", + "title": "Dangerous Minds", + "director": "John N. Smith", + "duration": "99.0", + "rating": "R", + "genre": "Drama, Crime" + }, + { + "recordID": "000032", + "title": "Twelve Monkeys", + "director": "Terry Gilliam", + "duration": "129.0", + "rating": "R", + "genre": "Science Fiction, Thriller, Mystery" + }, + { + "recordID": "000033", + "title": "Wings of Courage", + "director": "Jean-Jacques Annaud", + "duration": "50.0", + "rating": "G", + "genre": "Romance, Adventure" + }, + { + "recordID": "000034", + "title": "Babe", + "director": "Chris Noonan", + "duration": "89.0", + "rating": "G", + "genre": "Fantasy, Drama, Comedy, Family" + }, + { + "recordID": "000035", + "title": "Carrington", + "director": "Christopher Hampton", + "duration": "121.0", + "rating": "R", + "genre": "History, Drama, Romance" + }, + { + "recordID": "000036", + "title": "Dead Man Walking", + "director": "Tim Robbins", + "duration": "122.0", + "rating": "R", + "genre": "Drama" + }, + { + "recordID": "000037", + "title": "Across the Sea of Time", + "director": "Stephen Low", + "duration": "51.0", + "rating": "G", + "genre": "Adventure, History, Drama, Family" + }, + { + "recordID": "000038", + "title": "It Takes Two", + "director": "Andy Tennant", + "duration": "101.0", + "rating": "PG", + "genre": "Comedy, Family, Romance" + }, + { + "recordID": "000039", + "title": "Clueless", + "director": "Amy Heckerling", + "duration": "97.0", + "rating": "PG-13", + "genre": "Comedy, Drama, Romance" + }, + { + "recordID": "000040", + "title": "Cry, the Beloved Country", + "director": "Darrell James Roodt", + "duration": "106.0", + "rating": "PG-13", + "genre": "Drama" + }, + { + "recordID": "000041", + "title": "Richard III", + "director": "Richard Loncraine", + "duration": "104.0", + "rating": "R", + "genre": "Drama, War" + }, + { + "recordID": "000042", + "title": "Dead Presidents", + "director": "Albert Hughes", + "duration": "119.0", + "rating": "R", + "genre": "Action, Crime, Drama, History" + }, + { + "recordID": "000043", + "title": "Restoration", + "director": "Michael Hoffman", + "duration": "117.0", + "rating": "R", + "genre": "Drama, Romance" + }, + { + "recordID": "000044", + "title": "Mortal Kombat", + "director": "Paul W.S. Anderson", + "duration": "101.0", + "rating": "PG-13", + "genre": "Action, Fantasy" + }, + { + "recordID": "000045", + "title": "To Die For", + "director": "Gus Van Sant", + "duration": "106.0", + "rating": "R", + "genre": "Fantasy, Drama, Comedy, Thriller" + }, + { + "recordID": "000046", + "title": "How To Make An American Quilt", + "director": "Jocelyn Moorhouse", + "duration": "116.0", + "rating": "PG-13", + "genre": "Drama, Romance" + }, + { + "recordID": "000047", + "title": "Se7en", + "director": "David Fincher", + "duration": "127.0", + "rating": "R", + "genre": "Crime, Mystery, Thriller" + }, + { + "recordID": "000048", + "title": "Pocahontas", + "director": "Mike Gabriel", + "duration": "81.0", + "rating": "G", + "genre": "Adventure, Animation, Drama, Family" + }, + { + "recordID": "000049", + "title": "When Night Is Falling", + "director": "Patricia Rozema", + "duration": "96.0", + "rating": "Unrated", + "genre": "Drama, Romance" + }, + { + "recordID": "000050", + "title": "The Usual Suspects", + "director": "Bryan Singer", + "duration": "106.0", + "rating": "R", + "genre": "Drama, Crime, Thriller" + } +] \ No newline at end of file diff --git a/src/DVD.py b/src/DVD.py new file mode 100644 index 0000000..1bead66 --- /dev/null +++ b/src/DVD.py @@ -0,0 +1,247 @@ +import ast +import pandas as pd + +def parse_list(value): + """ + Convert a string containing a list of dictionaries + into a real Python list. + + Invalid or missing values become an empty list. + """ + + if pd.isna(value): + return [] + + if isinstance(value, list): + return value + + if isinstance(value, str): + try: + parsed_value = ast.literal_eval(value) + + if isinstance(parsed_value, list): + return parsed_value + + except (ValueError, SyntaxError): + return [] + +def extract_names(items): + """ + Extract the 'name' value from a list of dictionaries. + + Used for genres and keywords. + """ + + names = [] + + for item in items: + if isinstance(item, dict): + name = item.get("name") + + if name: + names.append(name) + + return names + + +def extract_top_cast(cast_list, cast_limit=5): + """ + Extract the first five cast member names. + """ + + cast_names = [] + + for person in cast_list[:cast_limit]: + if isinstance(person, dict): + name = person.get("name") + + if name: + cast_names.append(name) + + return cast_names + + +def extract_director(crew_list): + """ + Extract the director's name from the crew list. + """ + + for person in crew_list: + if not isinstance(person, dict): + continue + + if person.get("job") == "Director": + return person.get("name") + + return "" + +movies_df = pd.read_csv("movies_metadata.csv",low_memory=False) # Read the movies metadata CSV file into a pandas DataFrame +credits_df = pd.read_csv("credits.csv") +keywords_df = pd.read_csv("keywords.csv") + +movies_df = movies_df[ + [ + "id", + "title", + "runtime", + "genres" + ] +].copy() +# Create a copy of the DataFrame with selected columns: 'id', 'title', 'runtime', and 'genres' +credits_df = credits_df[ + [ + "id", + "cast", + "crew" + ] +].copy() + +keywords_df = keywords_df[ + [ + "id", + "keywords" + ] +].copy() + +movies_df["id"] = pd.to_numeric(movies_df["id"],errors="coerce") + +movies_df["runtime"] = pd.to_numeric(movies_df["runtime"], errors="coerce") + +movies_df = movies_df.dropna(subset=["id"]) + +movies_df["id"] = movies_df["id"].astype(int) + +movies_df["title"] = (movies_df["title"].fillna("").astype(str).str.strip()) + +movies_df = movies_df.drop_duplicates(subset=["id"]) + + +# ----------------------------------- +# 4. Clean credits DataFrame +# ----------------------------------- + +credits_df["id"] = pd.to_numeric(credits_df["id"],errors="coerce") + +credits_df = credits_df.dropna(subset=["id"]) + +credits_df["id"] = credits_df["id"].astype(int) + +credits_df = credits_df.drop_duplicates(subset=["id"]) + + +# ----------------------------------- +# 5. Clean keywords DataFrame +# ----------------------------------- + +keywords_df["id"] = pd.to_numeric(keywords_df["id"],errors="coerce") + +keywords_df = keywords_df.dropna(subset=["id"]) + +keywords_df["id"] = keywords_df["id"].astype(int) + +keywords_df = keywords_df.drop_duplicates(subset=["id"]) + +# Join Movies and Credits + +final_df = pd.merge(movies_df, credits_df, on="id", how="left") +final_df = pd.merge(final_df, keywords_df, on="id", how="left") +print("Merge completed.") +print("Rows after merge:", len(final_df)) + +# Check Merge Before Parsing + +print("\nColumns after merge:") + +print(final_df.columns.tolist()) + +print("\nFirst five merged records:") + +print( + final_df[ + [ + "id", + "title", + "runtime", + "genres", + "cast", + "crew", + "keywords" + ] + ].head()) + +# Transform: Parse Complex Columns + +final_df["genres_parsed"] = final_df["genres"].apply(parse_list) + +final_df["cast_parsed"] = final_df["cast"].apply(parse_list) + +final_df["crew_parsed"] = final_df["crew"].apply(parse_list) + +final_df["keywords_parsed"] = final_df["keywords"].apply(parse_list) + +# Transform: Extract Simple Values + +final_df["genre_names"] = final_df["genres_parsed"].apply(extract_names) + +final_df["top_cast"] = final_df["cast_parsed"].apply(lambda cast_list: extract_top_cast(cast_list,cast_limit=5)) + +final_df["director"] = final_df["crew_parsed"].apply(extract_director) + +final_df["keyword_names"] = final_df["keywords_parsed"].apply(extract_names) + +# Create Clean Final Dataframe + +clean_df = final_df[ + [ + "id", + "title", + "runtime", + "genre_names", + "top_cast", + "director", + "keyword_names" + ] +].copy() + +# Validate Final Data + +print("\nFinal DataFrame columns:") + +print(clean_df.columns.tolist()) + +print("\nFinal row count:", len(clean_df)) + +print("\nMissing values:") + +print(clean_df.isnull().sum()) + +print( + "\nDuplicate movie IDs:", + clean_df["id"].duplicated().sum() +) +# Display Toy Story +toy_story_df = clean_df[clean_df["id"] == 862] + +print("\nToy Story parsed record:") + +print(toy_story_df.to_string(index=False)) + +# LOAD: Export Full DATASET + +clean_df.to_csv("final_movies.csv",index=False) + +clean_df.to_json("final_movies.json",orient="records",indent=4,force_ascii=False) + +# Load: Export First 20 Records + +first_20_df = clean_df.head(20) + +first_20_df.to_json("final_movies_20.json",orient="records",indent=4,force_ascii=False) + +# Completion Message + +print("\nETL pipeline completed successfully.") + +print("Created: final_movies.csv") +print("Created: final_movies.json") +print("Created: final_movies_20.json") diff --git a/src/Movies-1.ipynb b/src/Movies-1.ipynb new file mode 100644 index 0000000..514f27a --- /dev/null +++ b/src/Movies-1.ipynb @@ -0,0 +1,700 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "id": "2826d7c0", + "metadata": {}, + "source": [ + "Helper Funtions" + ] + }, + { + "cell_type": "code", + "execution_count": 6, + "id": "f2db46ef", + "metadata": {}, + "outputs": [], + "source": [ + "import ast\n", + "import pandas as pd\n", + "\n", + "\n", + "def parse_list(value):\n", + " \"\"\"\n", + " Convert a string containing a list of dictionaries\n", + " into a real Python list.\n", + "\n", + " Invalid or missing values become an empty list.\n", + " \"\"\"\n", + "\n", + " if pd.isna(value):\n", + " return []\n", + "\n", + " if isinstance(value, list):\n", + " return value\n", + "\n", + " if isinstance(value, str):\n", + " try:\n", + " parsed_value = ast.literal_eval(value)\n", + "\n", + " if isinstance(parsed_value, list):\n", + " return parsed_value\n", + "\n", + " except (ValueError, SyntaxError):\n", + " return []\n", + " \n", + "def extract_names(items):\n", + " \"\"\"\n", + " Extract the 'name' value from a list of dictionaries.\n", + "\n", + " Used for genres and keywords.\n", + " \"\"\"\n", + "\n", + " names = []\n", + "\n", + " for item in items:\n", + " if isinstance(item, dict):\n", + " name = item.get(\"name\")\n", + "\n", + " if name:\n", + " names.append(name)\n", + "\n", + " return names\n", + "\n", + "\n", + "def extract_top_cast(cast_list, cast_limit=5):\n", + " \"\"\"\n", + " Extract the first five cast member names.\n", + " \"\"\"\n", + "\n", + " cast_names = []\n", + "\n", + " for person in cast_list[:cast_limit]:\n", + " if isinstance(person, dict):\n", + " name = person.get(\"name\")\n", + "\n", + " if name:\n", + " cast_names.append(name)\n", + "\n", + " return cast_names\n", + "\n", + "\n", + "def extract_director(crew_list):\n", + " \"\"\"\n", + " Extract the director's name from the crew list.\n", + " \"\"\"\n", + "\n", + " for person in crew_list:\n", + " if not isinstance(person, dict):\n", + " continue\n", + "\n", + " if person.get(\"job\") == \"Director\":\n", + " return person.get(\"name\")\n", + "\n", + " return \"\"" + ] + }, + { + "cell_type": "code", + "execution_count": 7, + "id": "76165d6d", + "metadata": {}, + "outputs": [ + { + "ename": "FileNotFoundError", + "evalue": "[Errno 2] No such file or directory: 'movies_metadata.csv'", + "output_type": "error", + "traceback": [ + "\u001b[31m---------------------------------------------------------------------------\u001b[39m", + "\u001b[31mFileNotFoundError\u001b[39m Traceback (most recent call last)", + "\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[7]\u001b[39m\u001b[32m, line 1\u001b[39m\n\u001b[32m----> \u001b[39m\u001b[32m1\u001b[39m movies_df = \u001b[30;43mpd\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43mread_csv\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43mmovies_metadata.csv\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43m,\u001b[39;49m\u001b[30;43mlow_memory\u001b[39;49m\u001b[30;43m=\u001b[39;49m\u001b[30;43;01mFalse\u001b[39;49;00m\u001b[30;43m)\u001b[39;49m \u001b[38;5;66;03m# Read the movies metadata CSV file into a pandas DataFrame\u001b[39;00m\n\u001b[32m 2\u001b[39m credits_df = pd.read_csv(\u001b[33m\"\u001b[39m\u001b[33mcredits.csv\u001b[39m\u001b[33m\"\u001b[39m)\n\u001b[32m 3\u001b[39m keywords_df = pd.read_csv(\u001b[33m\"\u001b[39m\u001b[33mkeywords.csv\u001b[39m\u001b[33m\"\u001b[39m)\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/.pyenv/versions/3.12.13/lib/python3.12/site-packages/pandas/io/parsers/readers.py:873\u001b[39m, in \u001b[36mread_csv\u001b[39m\u001b[34m(filepath_or_buffer, sep, delimiter, header, names, index_col, usecols, dtype, engine, converters, true_values, false_values, skipinitialspace, skiprows, skipfooter, nrows, na_values, keep_default_na, na_filter, skip_blank_lines, parse_dates, date_format, dayfirst, cache_dates, iterator, chunksize, compression, thousands, decimal, lineterminator, quotechar, quoting, doublequote, escapechar, comment, encoding, encoding_errors, dialect, on_bad_lines, low_memory, memory_map, float_precision, storage_options, dtype_backend)\u001b[39m\n\u001b[32m 861\u001b[39m kwds_defaults = _refine_defaults_read(\n\u001b[32m 862\u001b[39m dialect,\n\u001b[32m 863\u001b[39m delimiter,\n\u001b[32m (...)\u001b[39m\u001b[32m 869\u001b[39m dtype_backend=dtype_backend,\n\u001b[32m 870\u001b[39m )\n\u001b[32m 871\u001b[39m kwds.update(kwds_defaults)\n\u001b[32m--> \u001b[39m\u001b[32m873\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[30;43m_read\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43mfilepath_or_buffer\u001b[39;49m\u001b[30;43m,\u001b[39;49m\u001b[30;43m \u001b[39;49m\u001b[30;43mkwds\u001b[39;49m\u001b[30;43m)\u001b[39;49m\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/.pyenv/versions/3.12.13/lib/python3.12/site-packages/pandas/io/parsers/readers.py:300\u001b[39m, in \u001b[36m_read\u001b[39m\u001b[34m(filepath_or_buffer, kwds)\u001b[39m\n\u001b[32m 297\u001b[39m _validate_names(kwds.get(\u001b[33m\"\u001b[39m\u001b[33mnames\u001b[39m\u001b[33m\"\u001b[39m, \u001b[38;5;28;01mNone\u001b[39;00m))\n\u001b[32m 299\u001b[39m \u001b[38;5;66;03m# Create the parser.\u001b[39;00m\n\u001b[32m--> \u001b[39m\u001b[32m300\u001b[39m parser = \u001b[30;43mTextFileReader\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43mfilepath_or_buffer\u001b[39;49m\u001b[30;43m,\u001b[39;49m\u001b[30;43m \u001b[39;49m\u001b[30;43m*\u001b[39;49m\u001b[30;43m*\u001b[39;49m\u001b[30;43mkwds\u001b[39;49m\u001b[30;43m)\u001b[39;49m\n\u001b[32m 302\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m chunksize \u001b[38;5;129;01mor\u001b[39;00m iterator:\n\u001b[32m 303\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m parser\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/.pyenv/versions/3.12.13/lib/python3.12/site-packages/pandas/io/parsers/readers.py:1645\u001b[39m, in \u001b[36mTextFileReader.__init__\u001b[39m\u001b[34m(self, f, engine, **kwds)\u001b[39m\n\u001b[32m 1642\u001b[39m \u001b[38;5;28mself\u001b[39m.options[\u001b[33m\"\u001b[39m\u001b[33mhas_index_names\u001b[39m\u001b[33m\"\u001b[39m] = kwds[\u001b[33m\"\u001b[39m\u001b[33mhas_index_names\u001b[39m\u001b[33m\"\u001b[39m]\n\u001b[32m 1644\u001b[39m \u001b[38;5;28mself\u001b[39m.handles: IOHandles | \u001b[38;5;28;01mNone\u001b[39;00m = \u001b[38;5;28;01mNone\u001b[39;00m\n\u001b[32m-> \u001b[39m\u001b[32m1645\u001b[39m \u001b[38;5;28mself\u001b[39m._engine = \u001b[30;43mself\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43m_make_engine\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43mf\u001b[39;49m\u001b[30;43m,\u001b[39;49m\u001b[30;43m \u001b[39;49m\u001b[30;43mself\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43mengine\u001b[39;49m\u001b[30;43m)\u001b[39;49m\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/.pyenv/versions/3.12.13/lib/python3.12/site-packages/pandas/io/parsers/readers.py:1904\u001b[39m, in \u001b[36mTextFileReader._make_engine\u001b[39m\u001b[34m(self, f, engine)\u001b[39m\n\u001b[32m 1902\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m \u001b[33m\"\u001b[39m\u001b[33mb\u001b[39m\u001b[33m\"\u001b[39m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;129;01min\u001b[39;00m mode:\n\u001b[32m 1903\u001b[39m mode += \u001b[33m\"\u001b[39m\u001b[33mb\u001b[39m\u001b[33m\"\u001b[39m\n\u001b[32m-> \u001b[39m\u001b[32m1904\u001b[39m \u001b[38;5;28mself\u001b[39m.handles = \u001b[30;43mget_handle\u001b[39;49m\u001b[30;43m(\u001b[39;49m\n\u001b[32m 1905\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43mf\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 1906\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43mmode\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 1907\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43mencoding\u001b[39;49m\u001b[30;43m=\u001b[39;49m\u001b[30;43mself\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43moptions\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43mget\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43mencoding\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43m,\u001b[39;49m\u001b[30;43m \u001b[39;49m\u001b[30;43;01mNone\u001b[39;49;00m\u001b[30;43m)\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 1908\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43mcompression\u001b[39;49m\u001b[30;43m=\u001b[39;49m\u001b[30;43mself\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43moptions\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43mget\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43mcompression\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43m,\u001b[39;49m\u001b[30;43m \u001b[39;49m\u001b[30;43;01mNone\u001b[39;49;00m\u001b[30;43m)\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 1909\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43mmemory_map\u001b[39;49m\u001b[30;43m=\u001b[39;49m\u001b[30;43mself\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43moptions\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43mget\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43mmemory_map\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43m,\u001b[39;49m\u001b[30;43m \u001b[39;49m\u001b[30;43;01mFalse\u001b[39;49;00m\u001b[30;43m)\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 1910\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43mis_text\u001b[39;49m\u001b[30;43m=\u001b[39;49m\u001b[30;43mis_text\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 1911\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43merrors\u001b[39;49m\u001b[30;43m=\u001b[39;49m\u001b[30;43mself\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43moptions\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43mget\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43mencoding_errors\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43m,\u001b[39;49m\u001b[30;43m \u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43mstrict\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43m)\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 1912\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43mstorage_options\u001b[39;49m\u001b[30;43m=\u001b[39;49m\u001b[30;43mself\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43moptions\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43mget\u001b[39;49m\u001b[30;43m(\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43mstorage_options\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43m,\u001b[39;49m\u001b[30;43m \u001b[39;49m\u001b[30;43;01mNone\u001b[39;49;00m\u001b[30;43m)\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 1913\u001b[39m \u001b[30;43m\u001b[39;49m\u001b[30;43m)\u001b[39;49m\n\u001b[32m 1914\u001b[39m \u001b[38;5;28;01massert\u001b[39;00m \u001b[38;5;28mself\u001b[39m.handles \u001b[38;5;129;01mis\u001b[39;00m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;28;01mNone\u001b[39;00m\n\u001b[32m 1915\u001b[39m f = \u001b[38;5;28mself\u001b[39m.handles.handle\n", + "\u001b[36mFile \u001b[39m\u001b[32m~/.pyenv/versions/3.12.13/lib/python3.12/site-packages/pandas/io/common.py:930\u001b[39m, in \u001b[36mget_handle\u001b[39m\u001b[34m(path_or_buf, mode, encoding, compression, memory_map, is_text, errors, storage_options)\u001b[39m\n\u001b[32m 925\u001b[39m \u001b[38;5;28;01melif\u001b[39;00m \u001b[38;5;28misinstance\u001b[39m(handle, \u001b[38;5;28mstr\u001b[39m):\n\u001b[32m 926\u001b[39m \u001b[38;5;66;03m# Check whether the filename is to be opened in binary mode.\u001b[39;00m\n\u001b[32m 927\u001b[39m \u001b[38;5;66;03m# Binary mode does not support 'encoding' and 'newline'.\u001b[39;00m\n\u001b[32m 928\u001b[39m \u001b[38;5;28;01mif\u001b[39;00m ioargs.encoding \u001b[38;5;129;01mand\u001b[39;00m \u001b[33m\"\u001b[39m\u001b[33mb\u001b[39m\u001b[33m\"\u001b[39m \u001b[38;5;129;01mnot\u001b[39;00m \u001b[38;5;129;01min\u001b[39;00m ioargs.mode:\n\u001b[32m 929\u001b[39m \u001b[38;5;66;03m# Encoding\u001b[39;00m\n\u001b[32m--> \u001b[39m\u001b[32m930\u001b[39m handle = \u001b[30;43mopen\u001b[39;49m\u001b[30;43m(\u001b[39;49m\n\u001b[32m 931\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43mhandle\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 932\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43mioargs\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43mmode\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 933\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43mencoding\u001b[39;49m\u001b[30;43m=\u001b[39;49m\u001b[30;43mioargs\u001b[39;49m\u001b[30;43m.\u001b[39;49m\u001b[30;43mencoding\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 934\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43merrors\u001b[39;49m\u001b[30;43m=\u001b[39;49m\u001b[30;43merrors\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 935\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43mnewline\u001b[39;49m\u001b[30;43m=\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43m\"\u001b[39;49m\u001b[30;43m,\u001b[39;49m\n\u001b[32m 936\u001b[39m \u001b[30;43m \u001b[39;49m\u001b[30;43m)\u001b[39;49m\n\u001b[32m 937\u001b[39m \u001b[38;5;28;01melse\u001b[39;00m:\n\u001b[32m 938\u001b[39m \u001b[38;5;66;03m# Binary mode\u001b[39;00m\n\u001b[32m 939\u001b[39m handle = \u001b[38;5;28mopen\u001b[39m(handle, ioargs.mode)\n", + "\u001b[31mFileNotFoundError\u001b[39m: [Errno 2] No such file or directory: 'movies_metadata.csv'" + ] + } + ], + "source": [ + "movies_df = pd.read_csv(\"movies_metadata.csv\",low_memory=False) # Read the movies metadata CSV file into a pandas DataFrame\n", + "credits_df = pd.read_csv(\"credits.csv\")\n", + "keywords_df = pd.read_csv(\"keywords.csv\")" + ] + }, + { + "cell_type": "markdown", + "id": "72043293", + "metadata": {}, + "source": [ + "Selecting Columns" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "4d8b5d2d", + "metadata": {}, + "outputs": [], + "source": [ + "movies_df = movies_df[\n", + " [\n", + " \"id\",\n", + " \"title\",\n", + " \"runtime\",\n", + " \"genres\"\n", + " ]\n", + "].copy()\n", + "# Create a copy of the DataFrame with selected columns: 'id', 'title', 'runtime', and 'genres'\n", + "credits_df = credits_df[\n", + " [\n", + " \"id\",\n", + " \"cast\",\n", + " \"crew\"\n", + " ]\n", + "].copy()\n", + "\n", + "keywords_df = keywords_df[\n", + " [\n", + " \"id\",\n", + " \"keywords\"\n", + " ]\n", + "].copy()" + ] + }, + { + "cell_type": "markdown", + "id": "81dae026", + "metadata": {}, + "source": [ + "3. Clean movies DataFrame" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "bc1c1924", + "metadata": {}, + "outputs": [], + "source": [ + "movies_df[\"id\"] = pd.to_numeric(movies_df[\"id\"],errors=\"coerce\")\n", + "\n", + "movies_df[\"runtime\"] = pd.to_numeric(movies_df[\"runtime\"], errors=\"coerce\")\n", + "\n", + "movies_df = movies_df.dropna(subset=[\"id\"])\n", + "\n", + "movies_df[\"id\"] = movies_df[\"id\"].astype(int)\n", + "\n", + "movies_df[\"title\"] = (movies_df[\"title\"].fillna(\"\").astype(str).str.strip())\n", + "\n", + "movies_df = movies_df.drop_duplicates(subset=[\"id\"])\n", + "\n", + "\n", + "# -----------------------------------\n", + "# 4. Clean credits DataFrame\n", + "# -----------------------------------\n", + "\n", + "credits_df[\"id\"] = pd.to_numeric(credits_df[\"id\"],errors=\"coerce\")\n", + "\n", + "credits_df = credits_df.dropna(subset=[\"id\"])\n", + "\n", + "credits_df[\"id\"] = credits_df[\"id\"].astype(int)\n", + "\n", + "credits_df = credits_df.drop_duplicates(subset=[\"id\"])\n", + "\n", + "\n", + "# -----------------------------------\n", + "# 5. Clean keywords DataFrame\n", + "# -----------------------------------\n", + "\n", + "keywords_df[\"id\"] = pd.to_numeric(keywords_df[\"id\"],errors=\"coerce\")\n", + "\n", + "keywords_df = keywords_df.dropna(subset=[\"id\"])\n", + "\n", + "keywords_df[\"id\"] = keywords_df[\"id\"].astype(int)\n", + "\n", + "keywords_df = keywords_df.drop_duplicates(subset=[\"id\"])" + ] + }, + { + "cell_type": "markdown", + "id": "57ecde3b", + "metadata": {}, + "source": [ + "Join movies and credits" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "c74bcc06", + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Merge completed.\n", + "Rows after merge: 45433\n" + ] + } + ], + "source": [ + "final_df = pd.merge(movies_df, credits_df, on=\"id\", how=\"left\")\n", + "final_df = pd.merge(final_df, keywords_df, on=\"id\", how=\"left\")\n", + "print(\"Merge completed.\")\n", + "print(\"Rows after merge:\", len(final_df))" + ] + }, + { + "cell_type": "markdown", + "id": "871f28a3", + "metadata": {}, + "source": [ + "Check Merge Before Parsing" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "2a3377db", + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\n", + "Columns after merge:\n", + "['id', 'title', 'runtime', 'genres', 'cast', 'crew', 'keywords']\n", + "\n", + "First five merged records:\n", + " id title runtime \\\n", + "0 862 Toy Story 81.0 \n", + "1 8844 Jumanji 104.0 \n", + "2 15602 Grumpier Old Men 101.0 \n", + "3 31357 Waiting to Exhale 127.0 \n", + "4 11862 Father of the Bride Part II 106.0 \n", + "\n", + " genres \\\n", + "0 [{'id': 16, 'name': 'Animation'}, {'id': 35, '... \n", + "1 [{'id': 12, 'name': 'Adventure'}, {'id': 14, '... \n", + "2 [{'id': 10749, 'name': 'Romance'}, {'id': 35, ... \n", + "3 [{'id': 35, 'name': 'Comedy'}, {'id': 18, 'nam... \n", + "4 [{'id': 35, 'name': 'Comedy'}] \n", + "\n", + " cast \\\n", + "0 [{'cast_id': 14, 'character': 'Woody (voice)',... \n", + "1 [{'cast_id': 1, 'character': 'Alan Parrish', '... \n", + "2 [{'cast_id': 2, 'character': 'Max Goldman', 'c... \n", + "3 [{'cast_id': 1, 'character': \"Savannah 'Vannah... \n", + "4 [{'cast_id': 1, 'character': 'George Banks', '... \n", + "\n", + " crew \\\n", + "0 [{'credit_id': '52fe4284c3a36847f8024f49', 'de... \n", + "1 [{'credit_id': '52fe44bfc3a36847f80a7cd1', 'de... \n", + "2 [{'credit_id': '52fe466a9251416c75077a89', 'de... \n", + "3 [{'credit_id': '52fe44779251416c91011acb', 'de... \n", + "4 [{'credit_id': '52fe44959251416c75039ed7', 'de... \n", + "\n", + " keywords \n", + "0 [{'id': 931, 'name': 'jealousy'}, {'id': 4290,... \n", + "1 [{'id': 10090, 'name': 'board game'}, {'id': 1... \n", + "2 [{'id': 1495, 'name': 'fishing'}, {'id': 12392... \n", + "3 [{'id': 818, 'name': 'based on novel'}, {'id':... \n", + "4 [{'id': 1009, 'name': 'baby'}, {'id': 1599, 'n... \n" + ] + } + ], + "source": [ + "print(\"\\nColumns after merge:\")\n", + "\n", + "print(final_df.columns.tolist())\n", + "\n", + "print(\"\\nFirst five merged records:\")\n", + "\n", + "print(\n", + " final_df[\n", + " [\n", + " \"id\",\n", + " \"title\",\n", + " \"runtime\",\n", + " \"genres\",\n", + " \"cast\",\n", + " \"crew\",\n", + " \"keywords\"\n", + " ]\n", + " ].head()\n", + ")" + ] + }, + { + "cell_type": "markdown", + "id": "0d7d52d3", + "metadata": {}, + "source": [ + "Transform: Parse Complex Columns" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "49175342", + "metadata": {}, + "outputs": [], + "source": [ + "final_df[\"genres_parsed\"] = final_df[\"genres\"].apply(parse_list)\n", + "\n", + "final_df[\"cast_parsed\"] = final_df[\"cast\"].apply(parse_list)\n", + "\n", + "final_df[\"crew_parsed\"] = final_df[\"crew\"].apply(parse_list)\n", + "\n", + "final_df[\"keywords_parsed\"] = final_df[\"keywords\"].apply(parse_list)" + ] + }, + { + "cell_type": "markdown", + "id": "08a41b81", + "metadata": {}, + "source": [ + "Transform: Extract Simple Values" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "6f24e64b", + "metadata": {}, + "outputs": [], + "source": [ + "final_df[\"genre_names\"] = final_df[\"genres_parsed\"].apply(extract_names)\n", + "\n", + "final_df[\"top_cast\"] = final_df[\"cast_parsed\"].apply(lambda cast_list: extract_top_cast(cast_list,cast_limit=5))\n", + "\n", + "final_df[\"director\"] = final_df[\"crew_parsed\"].apply(extract_director)\n", + "\n", + "final_df[\"keyword_names\"] = final_df[\"keywords_parsed\"].apply(extract_names)" + ] + }, + { + "cell_type": "markdown", + "id": "20fef88a", + "metadata": {}, + "source": [ + "Create Clean Final Dataframe" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "536bf928", + "metadata": {}, + "outputs": [], + "source": [ + "clean_df = final_df[\n", + " [\n", + " \"id\",\n", + " \"title\",\n", + " \"runtime\",\n", + " \"genre_names\",\n", + " \"top_cast\",\n", + " \"director\",\n", + " \"keyword_names\"\n", + " ]\n", + "].copy()" + ] + }, + { + "cell_type": "markdown", + "id": "26544d96", + "metadata": {}, + "source": [ + "TMDB API Code" + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "id": "2f362e6f", + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "G\n" + ] + }, + { + "ename": "NameError", + "evalue": "name 'final_df' is not defined", + "output_type": "error", + "traceback": [ + "\u001b[31m---------------------------------------------------------------------------\u001b[39m", + "\u001b[31mNameError\u001b[39m Traceback (most recent call last)", + "\u001b[36mCell\u001b[39m\u001b[36m \u001b[39m\u001b[32mIn[4]\u001b[39m\u001b[32m, line 32\u001b[39m\n\u001b[32m 28\u001b[39m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[33m\"\u001b[39m\u001b[33m\"\u001b[39m\n\u001b[32m 30\u001b[39m \u001b[38;5;28mprint\u001b[39m(get_mpa_rating(\u001b[32m862\u001b[39m))\n\u001b[32m---> \u001b[39m\u001b[32m32\u001b[39m test_df = \u001b[30;43mfinal_df\u001b[39;49m.head(\u001b[32m5\u001b[39m).copy()\n\u001b[32m 34\u001b[39m test_df[\u001b[33m\"\u001b[39m\u001b[33mmpa_rating\u001b[39m\u001b[33m\"\u001b[39m] = test_df[\u001b[33m\"\u001b[39m\u001b[33mid\u001b[39m\u001b[33m\"\u001b[39m].apply(get_mpa_rating)\n\u001b[32m 36\u001b[39m test_df[[\u001b[33m\"\u001b[39m\u001b[33mtitle\u001b[39m\u001b[33m\"\u001b[39m, \u001b[33m\"\u001b[39m\u001b[33mmpa_rating\u001b[39m\u001b[33m\"\u001b[39m]]\n", + "\u001b[31mNameError\u001b[39m: name 'final_df' is not defined" + ] + } + ], + "source": [ + "import os\n", + "import requests\n", + "\n", + "TMDB_TOKEN = os.getenv(\"TMDB_TOKEN\")\n", + "\n", + "HEADERS = {\n", + " \"Authorization\": f\"Bearer {TMDB_TOKEN}\",\n", + " \"accept\": \"application/json\"\n", + "}\n", + "\n", + "def get_mpa_rating(tmdb_id):\n", + " url = f\"https://api.themoviedb.org/3/movie/{tmdb_id}/release_dates\"\n", + "\n", + " response = requests.get(url, headers=HEADERS)\n", + "\n", + " if response.status_code != 200:\n", + " return \"\"\n", + "\n", + " data = response.json()\n", + "\n", + " for country in data.get(\"results\", []):\n", + " if country.get(\"iso_3166_1\") == \"US\":\n", + " for release in country.get(\"release_dates\", []):\n", + " rating = release.get(\"certification\", \"\")\n", + " if rating:\n", + " return rating\n", + "\n", + " return \"\"\n", + "\n", + "print(get_mpa_rating(862))\n", + "\n", + "test_df = final_df.head(5).copy()\n", + "\n", + "test_df[\"mpa_rating\"] = test_df[\"id\"].apply(get_mpa_rating)\n", + "\n", + "test_df[[\"title\", \"mpa_rating\"]]" + ] + }, + { + "cell_type": "markdown", + "id": "8e68905f", + "metadata": {}, + "source": [ + "Test it with Toy Story" + ] + }, + { + "cell_type": "markdown", + "id": "8a834a47", + "metadata": {}, + "source": [ + "Validate Final Data" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "feebbf11", + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\n", + "Final DataFrame columns:\n", + "['id', 'title', 'runtime', 'genre_names', 'top_cast', 'director', 'keyword_names']\n", + "\n", + "Final row count: 45433\n", + "\n", + "Missing values:\n", + "id 0\n", + "title 0\n", + "runtime 260\n", + "genre_names 0\n", + "top_cast 0\n", + "director 0\n", + "keyword_names 0\n", + "dtype: int64\n", + "\n", + "Duplicate movie IDs: 0\n" + ] + } + ], + "source": [ + "# print(\"\\nFinal DataFrame columns:\")\n", + "\n", + "# print(clean_df.columns.tolist())\n", + "\n", + "# print(\"\\nFinal row count:\", len(clean_df))\n", + "\n", + "# print(\"\\nMissing values:\")\n", + "\n", + "# print(clean_df.isnull().sum())\n", + "\n", + "# print(\n", + "# \"\\nDuplicate movie IDs:\",\n", + "# clean_df[\"id\"].duplicated().sum()\n", + "# )" + ] + }, + { + "cell_type": "markdown", + "id": "a830728a", + "metadata": {}, + "source": [ + "Display Toy Story" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "79feed93", + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\n", + "Toy Story parsed record:\n", + " id title runtime genre_names top_cast director keyword_names\n", + "862 Toy Story 81.0 [Animation, Comedy, Family] [Tom Hanks, Tim Allen, Don Rickles, Jim Varney, Wallace Shawn] John Lasseter [jealousy, toy, boy, friendship, friends, rivalry, boy next door, new toy, toy comes to life]\n" + ] + } + ], + "source": [ + "# toy_story_df = clean_df[clean_df[\"id\"] == 862]\n", + "\n", + "# print(\"\\nToy Story parsed record:\")\n", + "\n", + "# print(toy_story_df.to_string(index=False))" + ] + }, + { + "cell_type": "markdown", + "id": "c7ce158b", + "metadata": {}, + "source": [ + "LOAD: Export Full DATASET" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "d50b5e8a", + "metadata": {}, + "outputs": [], + "source": [ + "# clean_df.to_csv(\"final_movies.csv\",index=False)\n", + "\n", + "# clean_df.to_json(\"final_movies.json\",orient=\"records\",indent=4,force_ascii=False)" + ] + }, + { + "cell_type": "markdown", + "id": "fe1e668f", + "metadata": {}, + "source": [ + "Load: Export First 20 Records" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "680af149", + "metadata": {}, + "outputs": [], + "source": [ + "# first_20_df = clean_df.head(20)\n", + "\n", + "# first_20_df.to_json(\"final_movies_20.json\",orient=\"records\",indent=4,force_ascii=False)" + ] + }, + { + "cell_type": "markdown", + "id": "849db261", + "metadata": {}, + "source": [ + "Completion Message" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "0caf1cfd", + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "\n", + "ETL pipeline completed successfully.\n", + "Created: final_movies.csv\n", + "Created: final_movies.json\n", + "Created: final_movies_20.json\n" + ] + } + ], + "source": [ + "# print(\"\\nETL pipeline completed successfully.\")\n", + "\n", + "# print(\"Created: final_movies.csv\")\n", + "# print(\"Created: final_movies.json\")\n", + "# print(\"Created: final_movies_20.json\")" + ] + }, + { + "cell_type": "markdown", + "id": "02c20144", + "metadata": {}, + "source": [] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "456d0ded", + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "eyJhbGciOiJIUzI1NiJ9.eyJhdWQiOiIxN2RmNWYyNzhiZTExOTQ5ZGVmZmI1NmY2ZTk3YjczMCIsIm5iZiI6MTc4NDg5NTIwOS41MjQ5OTk5LCJzdWIiOiI2YTYzNTZlOWRlZjAyYmE4NTk0MGU0YTQiLCJzY29wZXMiOlsiYXBpX3JlYWQiXSwidmVyc2lvbiI6MX0.DqOORblbbHamaK92toKGASnDQ1kZk0kR9udv10PiIcE\n" + ] + } + ], + "source": [ + "import os\n", + "\n", + "TMDB_TOKEN = os.getenv(\"TMDB_TOKEN\")\n", + "\n", + "print(TMDB_TOKEN)" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "3.12.13", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.12.13" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} diff --git a/src/books.py b/src/books.py new file mode 100644 index 0000000..b3128fa --- /dev/null +++ b/src/books.py @@ -0,0 +1,219 @@ +import csv +import json +import random +import re + + +from pathlib import Path + +csv_file = Path("data/pg_catalog.csv") +json_file = Path("data/booksout.json") +used_isbns = set() + +def generate_isbn(): #generate a unique 13 digit isbn-like number + + while True: + isbn = str(random.randint(1000000000000, 9999999999999)) + + if isbn not in used_isbns: + used_isbns.add(isbn) + return isbn + +# def get_genre(title,author): +# paramaters = { +# "title": title, +# "author": author, +# "limit": 1 +# } +# response = requests.get(url, params=paramaters, timeout=10) +# if response.status_code != 200: +# return "Unknown" + +# data = response.json() + +def read_books(csv_file): + with open(csv_file, 'r', encoding="utf-8") as file: + reader = csv.DictReader(file) + books = [] + + for row in reader: + books.append(row) + + return books + +def get_genre(bookshelves_field): + + if not bookshelves_field: + return "Unknown" + + bookshelves = bookshelves_field.split(";") + + for shelf in bookshelves: + shelf = shelf.strip() + + if shelf.startswith("Browsing:"): + continue + + if shelf: + return shelf + + return "Unknown" + +def clean_books(books): + """1. Add an isbn to each book/row + """ + cleaned_books = [] + + for book in books: + parsed_authors = clean_authors(book.get("Authors", "")) + parsed_subjects = clean_subjects(book.get("Subjects", "")) + + cleaned_book = { + # "text_number": book.get("Text#", "").strip(), + # "type": book.get("Type", "").strip(), + + "title": book.get("Title", "").strip(), + # "issued": book.get("Issued", "").strip(), + # "language": book.get("Language", "").strip(), + "authors": [ + author["name"] + for author in parsed_authors + ], + "author_lifespan": [ + author["lifespan"] + for author in parsed_authors + if author["lifespan"] + ], + "isbn": generate_isbn(), + "numberOfPages": None, + "genre": get_genre(book.get("Bookshelves", "")), + "subjects": clean_subjects(book.get("Subjects", "")), + "locc": book.get("LoCC", "").strip(), + # "bookshelves": book.get("Bookshelves", "").strip(), + + } + + cleaned_books.append(cleaned_book) + + return cleaned_books + +def clean_authors(author_field): + if not author_field: + return[] + + authors = author_field.split(";") + cleaned_authors = [] + + for author in authors: + author = author.strip() + + lifespan_match = re.search( + r"\b\d{3,4}\??(?:\s+BCE)?-\d{3,4}\??(?:\s+BCE)?\b", + author + ) + if lifespan_match: + lifespan = lifespan_match.group() + + name = author.replace(lifespan, "") + name = name.strip(" ,") + else: + lifespan = "" + name = author + + cleaned_authors.append({ + "name": name, + "lifespan": lifespan + }) + + return cleaned_authors + +def clean_subjects(subject_field): + if not subject_field: + return [] + + subjects = subject_field.split(";") + cleaned_subjects = [] + + for subject in subjects: + subject = subject.strip() + + if subject: + cleaned_subjects.append(subject) + + return cleaned_subjects + +def validate_books(cleaned_books): + valid_books = [] + missing_title = 0 + missing_authors = 0 + invalid_isbn = 0 + record_number = 1 + + for book in cleaned_books: + + title = book.get("title", "").strip() + authors = book.get("authors", []) + isbn = book.get("isbn", "").strip() + + if not title: + missing_title += 1 + continue + + if not authors: + missing_authors += 1 + continue + + if len(isbn) != 13: + invalid_isbn += 1 + continue + + # Create a new dictionary with recordID as the first field + new_book = { + "recordID": f"{record_number:06d}" + } + + # Add the rest of the book fields + new_book.update(book) + + valid_books.append(new_book) + record_number += 1 + + print("Missing title:", missing_title) + print("Missing authors:", missing_authors) + print("Invalid ISBN:", invalid_isbn) + print("Books validated:", len(valid_books)) + + return valid_books + +def write_json(valid_books, json_file): + with open(json_file, "w", encoding="utf-8") as file: + json.dump(valid_books, file, indent=4) + + return json_file + + +if __name__ == "__main__": + books = read_books(csv_file) + + cleaned_books = clean_books(books) + + print(cleaned_books[0]) + print(cleaned_books[0].keys()) + + valid_books = validate_books(cleaned_books) + + books_to_process = valid_books[:50] + + + print("Books read: ", len(books)) + print("Books cleaned: ", len(cleaned_books)) + print("Books validated: ", len(valid_books)) + + books_to_write = valid_books[:50] + print("Books being written to JSON file: ", len(books_to_write)) + write_json(valid_books[:50], json_file) + + #this is so only the first 5 rows are returned for testing purposes + # for book in books[:5]: + # print(book) + diff --git a/src/dvd.py b/src/dvd.py new file mode 100644 index 0000000..63bbdc1 --- /dev/null +++ b/src/dvd.py @@ -0,0 +1,398 @@ +import ast +import os +from pathlib import Path + +import pandas as pd +import requests + + +# ============================================================ +# 1. Helper Functions +# ============================================================ + +def parse_list(value): + """ + Convert a string containing a list of dictionaries + into a Python list. + """ + if pd.isna(value): + return [] + + if isinstance(value, list): + return value + + if isinstance(value, str): + try: + parsed_value = ast.literal_eval(value) + + if isinstance(parsed_value, list): + return parsed_value + + except (ValueError, SyntaxError): + return [] + + return [] + + +def extract_names(items): + """ + Extract the 'name' value from a list of dictionaries. + Used for movie genres. + """ + names = [] + + for item in items: + if isinstance(item, dict): + name = item.get("name") + + if name: + names.append(name) + + return names + + +def extract_top_cast(cast_list, cast_limit=5): + """ + Extract the first cast-member names, + up to the specified cast limit. + """ + cast_names = [] + + for person in cast_list[:cast_limit]: + if isinstance(person, dict): + name = person.get("name") + + if name: + cast_names.append(name) + + return cast_names + + +def extract_director(crew_list): + """ + Extract the director's name from the crew list. + """ + for person in crew_list: + if ( + isinstance(person, dict) + and person.get("job") == "Director" + ): + return person.get("name", "") + + return "" + + +# ============================================================ +# 2. Extract: Read CSV Files +# ============================================================ + +movies_df = pd.read_csv( + "movies_metadata.csv", + low_memory=False +) + +credits_df = pd.read_csv("credits.csv") + + +# ============================================================ +# 3. Select Required Columns +# ============================================================ + +movies_df = movies_df[ + [ + "id", + "title", + "runtime", + "genres" + ] +].copy() + +credits_df = credits_df[ + [ + "id", + "cast", + "crew" + ] +].copy() + + +# ============================================================ +# 4. Clean Movies DataFrame +# ============================================================ + +movies_df["id"] = pd.to_numeric( + movies_df["id"], + errors="coerce" +) + +movies_df["runtime"] = pd.to_numeric( + movies_df["runtime"], + errors="coerce" +) + +# Remove rows without a valid movie ID +movies_df = movies_df.dropna(subset=["id"]) + +movies_df["id"] = movies_df["id"].astype(int) + +movies_df["title"] = ( + movies_df["title"] + .fillna("") + .astype(str) + .str.strip() +) + +# Remove duplicate TMDB movie IDs +movies_df = movies_df.drop_duplicates( + subset=["id"] +) + + +# ============================================================ +# 5. Clean Credits DataFrame +# ============================================================ + +credits_df["id"] = pd.to_numeric( + credits_df["id"], + errors="coerce" +) + +credits_df = credits_df.dropna( + subset=["id"] +) + +credits_df["id"] = credits_df["id"].astype(int) + +credits_df = credits_df.drop_duplicates( + subset=["id"] +) + + +# ============================================================ +# 6. Merge Movies and Credits +# ============================================================ + +final_df = pd.merge( + movies_df, + credits_df, + on="id", + how="left" +) + + +# ============================================================ +# 7. Parse Complex Columns +# ============================================================ + +final_df["genres_parsed"] = ( + final_df["genres"].apply(parse_list) +) + +final_df["cast_parsed"] = ( + final_df["cast"].apply(parse_list) +) + +final_df["crew_parsed"] = ( + final_df["crew"].apply(parse_list) +) + + +# ============================================================ +# 8. Create Simplified Columns +# ============================================================ + +final_df["genre_names"] = ( + final_df["genres_parsed"].apply(extract_names) +) + +final_df["top_cast"] = ( + final_df["cast_parsed"].apply( + lambda cast_list: extract_top_cast( + cast_list, + cast_limit=5 + ) + ) +) + +final_df["director"] = ( + final_df["crew_parsed"].apply(extract_director) +) + + +# ============================================================ +# 9. Create Clean DataFrame +# ============================================================ + +clean_df = final_df[ + [ + "id", + "title", + "runtime", + "genre_names", + "top_cast", + "director" + ] +].copy() + + +# ============================================================ +# 10. TMDB API Configuration +# ============================================================ + +TMDB_TOKEN = os.getenv("TMDB_TOKEN") + +if not TMDB_TOKEN: + raise ValueError( + "TMDB_TOKEN was not found. " + "Add it to your environment and restart VS Code." + ) + +HEADERS = { + "Authorization": f"Bearer {TMDB_TOKEN}", + "accept": "application/json" +} + +# Reuse the same connection for multiple requests +session = requests.Session() +session.headers.update(HEADERS) + + +def get_mpa_rating(tmdb_id): + """ + Return the first available US MPA certification + for a movie from the TMDB API. + """ + url = ( + f"https://api.themoviedb.org/3/movie/" + f"{tmdb_id}/release_dates" + ) + + try: + response = session.get( + url, + timeout=15 + ) + + response.raise_for_status() + + except requests.RequestException as error: + print( + f"TMDB request failed for movie " + f"{tmdb_id}: {error}" + ) + return "" + + try: + data = response.json() + + except requests.JSONDecodeError: + print( + f"TMDB returned invalid JSON " + f"for movie {tmdb_id}." + ) + return "" + + for country in data.get("results", []): + if country.get("iso_3166_1") == "US": + + for release in country.get( + "release_dates", + [] + ): + rating = ( + release + .get("certification", "") + .strip() + ) + + if rating: + return rating + + return "" + + +# ============================================================ +# 11. Test the TMDB Function +# ============================================================ + +# ============================================================ +# Process only the first 100 movies +# ============================================================ + +clean_df = clean_df.head(100).copy() + +print(f"Processing {len(clean_df)} movies...") + + +# ============================================================ +# 12. Add MPA Ratings +# ============================================================ + +# This makes one API request for each movie. +clean_df["mpa_rating"] = ( + clean_df["id"].apply(get_mpa_rating) +) + + +# Put mpa_rating next to runtime +clean_df = clean_df[ + [ + "id", + "title", + "runtime", + "mpa_rating", + "genre_names", + "top_cast", + "director" + ] +].copy() + +print("\nMPA ratings added to clean_df.") + +print( + clean_df[ + [ + "title", + "mpa_rating" + ] + ].head() +) + + +# ============================================================ +# 13. Validate Final Data +# ============================================================ + +print("\nFinal DataFrame columns:") +print(clean_df.columns.tolist()) + +print("\nFinal row count:") +print(len(clean_df)) + +print("\nMissing values:") +print(clean_df.isnull().sum()) + +print("\nDuplicate movie IDs:") +print(clean_df["id"].duplicated().sum()) + +print("\nMPA rating distribution:") + +print( + clean_df["mpa_rating"] + .replace("", "Not available") + .value_counts(dropna=False) +) + +# ============================================================ +# 14. Export Full Dataset as JSON +# ============================================================ +# Export the JSON file + +clean_df.to_json( + "final_movies.json", + orient="records", + indent=4, + force_ascii=False +) + +print("final_movies.json created successfully!") diff --git a/src/dvd_pandas.py b/src/dvd_pandas.py new file mode 100644 index 0000000..bcb7469 --- /dev/null +++ b/src/dvd_pandas.py @@ -0,0 +1,86 @@ +import json +import os +import requests +import pandas as pd +from pathlib import Path + +API_KEY = os.getenv("OMDB_API_KEY") +rating_cache = {} + +credit_file = Path("data/movies-archive/credits.csv") +movies_file = Path("data/movies-archive/movies_metadata.csv") +json_file = Path("data/dvdout_panda.json") + +def get_movie_rating(imdb_id, api_key): + if not imdb_id: + return "Not Rated" + + if imdb_id in rating_cache: + return rating_cache[imdb_id] + + url = "https://www.omdbapi.com/" + parameters = {"apikey": api_key, "i": imdb_id} + try: + response = requests.get(url, params=parameters, timeout=10) + response.raise_for_status() + rating = response.json().get("Rated", "NR") + except requests.RequestException: + rating = "NR" + + rating_cache[imdb_id] = rating + return rating + + +def get_director(crew_string): + try: + crew_list = pd.io.json.ujson_loads(crew_string) if False else __import__("ast").literal_eval(crew_string) + except (ValueError, SyntaxError): + return "Unknown" + + for member in crew_list: + if member.get("job") == "Director": + return member.get("name") + return "Unknown" + + +def get_genres(genres_string): + try: + genres_list = __import__("ast").literal_eval(genres_string) + + if not isinstance(genres_list, list): + return "Unknown" + names = [] + for g in genres_list: + # each genre item is expected to be a dict with a 'name' key + if isinstance(g, dict) and "name" in g: + names.append(str(g["name"])) + return ", ".join(names) if names else "Unknown" + except (ValueError, SyntaxError): + return "Unknown" + + +if __name__ == "__main__": + movies_df = pd.read_csv(movies_file, low_memory=False).head(50) + credits_df = pd.read_csv(credit_file) + + # id columns must match types to merge cleanly + movies_df["id"] = movies_df["id"].astype(str) + credits_df["id"] = credits_df["id"].astype(str) + + credits_df["director"] = credits_df["crew"].apply(get_director) + + merged_df = movies_df.merge( + credits_df[["id", "director"]], + on="id", + how="left" + ) + + merged_df["director"] = merged_df["director"].fillna("Unknown") + merged_df["genre"] = merged_df["genres"].apply(get_genres) + merged_df["rating"] = merged_df["imdb_id"].apply(lambda x: get_movie_rating(x, API_KEY)) + merged_df["recordID"] = [f"{i:06d}" for i in range(1, len(merged_df) + 1)] + + output_df = merged_df[["recordID", "title", "director", "runtime", "rating", "genre"]] + output_df = output_df.rename(columns={"runtime": "duration"}) + + output_df.to_json(json_file, orient="records", indent=4) \ No newline at end of file diff --git a/src/music.py b/src/music.py new file mode 100644 index 0000000..8a0469a --- /dev/null +++ b/src/music.py @@ -0,0 +1,77 @@ +import json +from pathlib import Path + +import pandas as pd + + +INPUT_FILE = Path("tcc_ceds_music.csv") +OUTPUT_FILE = Path("music_data.json") +OUTPUT_FILE_20 = Path("music_data_20.json") + +REQUIRED_COLUMNS = [ + "artist_name", + "track_name", + "genre", + "release_date", +] + + +def extract_music_data(input_file): + + if not input_file.exists(): + raise FileNotFoundError(f"Input file was not found: {input_file}") + + # Read CSV + music_df = pd.read_csv(input_file, dtype=str) + + # Check required columns + missing_columns = [] + + for column in REQUIRED_COLUMNS: + if column not in music_df.columns: + missing_columns.append(column) + + if missing_columns: + raise ValueError(f"Missing required columns: {missing_columns}") + + # Keep only required columns + music_df = music_df[REQUIRED_COLUMNS] + + # Replace missing values with empty strings + music_df = music_df.fillna("") + + # Remove leading/trailing spaces + for column in REQUIRED_COLUMNS: + music_df[column] = music_df[column].str.strip() + + # Convert DataFrame to list of dictionaries + music_data = music_df.to_dict(orient="records") + + return music_data + + +def write_json(data, output_file): + + with output_file.open(mode="w", encoding="utf-8") as json_file: + json.dump(data, json_file, indent=4, ensure_ascii=False) + + +def main(): + + music_data = extract_music_data(INPUT_FILE) + + # Write all records + write_json(music_data, OUTPUT_FILE) + + # Write first 20 records (index 0-19) + first_20_records = music_data[:20] + write_json(first_20_records, OUTPUT_FILE_20) + + print("Extraction completed successfully.") + print(f"Total records extracted: {len(music_data)}") + print(f"Full JSON file: {OUTPUT_FILE}") + print(f"First 20 records JSON file: {OUTPUT_FILE_20}") + + +if __name__ == "__main__": + main()