Skip to content

Add our archive.org homebrew - #1

Merged
sharkwouter merged 3 commits into
mainfrom
add-archive
Sep 21, 2026
Merged

sharkwouter merged 3 commits into
mainfrom
add-archive

Conversation

@sharkwouter

Copy link
Copy Markdown
Member

This is the script I used:

#!/usr/bin/env python3

import os
import json
import time
from dataclasses import dataclass
import xml.etree.ElementTree as ET
import urllib.parse
import datetime

import requests

from homebrew_database.homebrew import Homebrew, Release


@dataclass
class DocDownloadData():
  download_url: str
  download_sha1: str
  screenshots: list[str]


@dataclass
class Doc():
  creator: str
  date: str
  description: str
  identifier: str
  title: str
  version: str
  subject: list[str] | str
  format: str


def create_doc_list_from_data(data: dict):
  docs = []
  doc_count_in_data = data["response"]["numFound"]
  print(f"Processing {doc_count_in_data} entries")
  for doc_dict in data["response"]["docs"]:
    identifier = doc_dict["identifier"]
    title = doc_dict["title"]
    description = doc_dict["description"]
    format = doc_dict["format"]
    subject = doc_dict["subject"]
    if type(subject) is not list:
      continue
    if not doc_dict.get("creator"):
      print(f"Could not get creator for {title}")
      continue
    creator = doc_dict["creator"]
    if not doc_dict.get("date"):
      print(f"Could not get date for {title}")
      continue
    date = doc_dict["date"]
    if not doc_dict.get("version"):
      print(f"Could not get version for {title}")
      continue
    version = doc_dict["version"]

    docs.append(
      Doc(
        creator=creator,
        date=date,
        description=description,
        identifier=identifier,
        title=title,
        version=version,
        subject=subject,
        format=format,
      )
    )

  print(f"Was able to process {len(docs)}/{doc_count_in_data} entries")

  return docs

def get_data() -> dict:
  file_name = "psp-archive.json"
  if os.path.exists(file_name):
    with open(file_name, "r") as fd:
      return json.loads(fd.read())

  url = "https://archive.org/advancedsearch.php?q=collection%3Apsp-homebrew-library&fl[]=identifier&fl[]=title&fl[]=creator&fl[]=description&fl[]=version&fl[]=date&fl[]=subject&fl[]=format&sort[]=date asc&rows=10000&page=1&output=json"

  response = requests.get(url=url)
  response.raise_for_status()
  with open(file_name, "w") as fd:
    fd.write(response.text)
  return response.json()


def convert_name_to_identifier(name: str) -> str:
  identifier = name.lower()

  identifier = identifier.replace(" ", "_")
  identifier = identifier.replace(":", "")
  identifier = identifier.replace("/", "_")

  return identifier


def enrich_homebrew_with_xml_data(homebrew: Homebrew, doc: Doc, xml_data: str) -> None:
  root = ET.fromstring(xml_data)
  print(f"{homebrew.name}:")
  screenshots = []
  download_url = ""
  for file in root.iter("file"):
    if not file.attrib or len(file.attrib.keys()) == 0:
      continue
    if file.attrib.get("source") != "original":
      continue
    name = file.attrib.get("name")
    if not name or name == "__ia_thumb.jpg":
      continue
    if name.lower().endswith(".png"):
      screenshots.append(name)
    if name.lower().endswith(".jpg"):
      screenshots.append(name)
    if name.lower().endswith(".jpeg"):
      screenshots.append(name)
    if name.lower().endswith(".bmp"):
      screenshots.append(name)
    if name.lower().endswith(".zip"):
      download_url = name

  if not download_url:
    raise Exception(f"No download_url found for {homebrew.name} in xml: {xml_data}")
  if not screenshots:
    raise Exception(f"No screenshots found for {homebrew.name} in xml: {xml_data}")

  print(doc.identifier)
  base_url = f"https://archive.org/download/{urllib.parse.quote(doc.identifier)}/"
  release = Release(
    tag=doc.version,
    url=str(urllib.parse.urljoin(base_url, urllib.parse.quote(download_url))),
    published_at=datetime.date.fromisoformat(doc.date.split("T")[0]),
  )
  homebrew.releases.append(release)
  for screenshot in screenshots:
    homebrew.screenshots.append(str(urllib.parse.urljoin(base_url, urllib.parse.quote(screenshot))))


def get_download_xml(doc: Doc) -> str:
  file_name = f"{convert_name_to_identifier(doc.title).replace("/","_")}.xml"
  file_path = os.path.join("temp", file_name)
  if os.path.exists(file_path):
    with open(file_path, "r") as fd:
      return fd.read()
  url = f"https://archive.org/download/{urllib.parse.quote(doc.identifier)}/{urllib.parse.quote(doc.identifier)}_files.xml"
  response = requests.get(url)
  response.raise_for_status()

  if not os.path.isdir("temp"):
    os.mkdir("temp")
  with open(file_path, "w") as fd:
    fd.write(response.text)

  time.sleep(0.1)
  return response.text


def create_homebrew_list(doc_list: list[Doc]) -> list[Homebrew]:
  homebrew_list = []
  for doc in doc_list:
    if "not working" in doc.subject:
      continue
    if "1.50 FW only" in doc.subject:
      continue
    if "demo" in doc.subject:
      continue
    if "ZIP" not in doc.format:
      print(f"{doc.title} has no zip")
      continue
    identifier = convert_name_to_identifier(doc.title)
    homebrew = Homebrew(
      name=doc.title,
      id=identifier,
      summary=doc.description,
      author=doc.creator,
      screenshots=[],
      ai_used=False,
      requires_additional_files="game data required" in doc.subject,
      category="",
    )

    for subject in doc.subject:
      if subject == "PSP homebrew":
        continue
      if subject == "game":
        homebrew.category = subject
        continue
      if subject == "emulator":
        homebrew.category = subject
        continue
      if subject == "application":
        homebrew.category = subject
        continue
      homebrew.tags.append(subject)

    if not homebrew.category:
      continue

    download_xml = get_download_xml(doc)
    try:
      enrich_homebrew_with_xml_data(homebrew=homebrew, doc=doc, xml_data=download_xml)
    except Exception as e:
      print(f"Could not process homebrew {homebrew.name} {type(e).__name__}: {e}")
      continue
    homebrew_list.append(homebrew)

  return homebrew_list


def main():
  data = get_data()
  doc_list = create_doc_list_from_data(data=data)
  homebrew_list = create_homebrew_list(doc_list)
  for homebrew in homebrew_list:
    file_path = os.path.join("data", f"{homebrew.id}.json")
    if os.path.exists(file_path):
      continue
    with open(file_path, "w") as fd:
      fd.write(json.dumps(homebrew.to_dict(), indent=2))


if __name__ == "__main__":
  main()

It was somewhat painful to do because there are way too many files. It seems to work okay. We can fix small issues with it later still.

@sharkwouter
sharkwouter merged commit 1ceeea4 into main Sep 21, 2026
2 checks passed
@sharkwouter
sharkwouter deleted the add-archive branch September 21, 2026 00:21
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

Labels

None yet

Projects

None yet

Development

Successfully merging this pull request may close these issues.

1 participant