From 502f9b93c993e7660f74002b979a144f750c8118 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Wed, 15 Oct 2025 22:25:20 +0300 Subject: [PATCH 01/30] Added Europeana integration --- Pipfile | 1 + Pipfile.lock | 88 ++++--- data/2025Q4/1-fetch/europeana_1_count.csv | 31 +++ env.example | 4 +- scripts/1-fetch/europeana_fetch.py | 265 ++++++++++++++++++++++ 5 files changed, 359 insertions(+), 30 deletions(-) create mode 100644 data/2025Q4/1-fetch/europeana_1_count.csv create mode 100644 scripts/1-fetch/europeana_fetch.py diff --git a/Pipfile b/Pipfile index 3fc0133d..889b706f 100644 --- a/Pipfile +++ b/Pipfile @@ -22,6 +22,7 @@ requests = ">=2.31.0" seaborn = "*" urllib3 = ">=2.5.0" wordcloud = "*" +pre-commit = "*" [dev-packages] black = "*" diff --git a/Pipfile.lock b/Pipfile.lock index 741d5da4..17810845 100644 --- a/Pipfile.lock +++ b/Pipfile.lock @@ -1,7 +1,7 @@ { "_meta": { "hash": { - "sha256": "ffa89d9d058b31c7b28aca6ed62f13f1273b3f8b92f5fa4d279dd3737844d8e8" + "sha256": "444f6f70a55f14caf5149c4c9b487780805c162fb5d0add1433c8bf74769004c" }, "pipfile-spec": 6, "requires": { @@ -24,14 +24,6 @@ "markers": "python_version >= '3.9'", "version": "==4.11.0" }, - "appnope": { - "hashes": [ - "sha256:1de3860566df9caf38f01f86f65e0e13e379af54f9e4bee1e66b48f2efffd1ee", - "sha256:502575ee11cd7a28c0205f379b525beefebab9d161b7c964670864014ed7213c" - ], - "markers": "python_version >= '3.6'", - "version": "==0.1.4" - }, "argon2-cffi": { "hashes": [ "sha256:694ae5cc8a42f4c4e2bf2ca0e64e51e23a040c6a517a85074683d3959e1346c1", @@ -237,6 +229,14 @@ "markers": "python_version >= '3.9'", "version": "==2.0.0" }, + "cfgv": { + "hashes": [ + "sha256:b7265b1f29fd3316bfcd2b330d63d024f2bfd8bcb8b0272f8e19a504856c48f9", + "sha256:e52591d4c5f5dead8e0f673fb16db7949d2cfb3f7da4582893288f0ded8fe560" + ], + "markers": "python_version >= '3.8'", + "version": "==3.4.0" + }, "charset-normalizer": { "hashes": [ "sha256:00237675befef519d9af72169d8604a067d92755e84fe76492fef5441db05b91", @@ -468,6 +468,13 @@ "markers": "python_version >= '2.7' and python_version not in '3.0, 3.1, 3.2, 3.3, 3.4'", "version": "==0.7.1" }, + "distlib": { + "hashes": [ + "sha256:9659f7d87e46584a30b5780e43ac7a2143098441670ff0a49d5f9034c54a6c16", + "sha256:feec40075be03a04501a973d81f633735b4b69f98b05450592310c0f401a4e0d" + ], + "version": "==0.4.0" + }, "executing": { "hashes": [ "sha256:3632cc370565f6648cc328b32435bd120a1e4ebb20c77e3fdde9a13cd1e533c4", @@ -483,6 +490,14 @@ ], "version": "==2.21.2" }, + "filelock": { + "hashes": [ + "sha256:339b4732ffda5cd79b13f4e2711a31b0365ce445d95d243bb996273d072546a2", + "sha256:711e943b4ec6be42e1d4e6690b48dc175c822967466bb31c0c293f34334c13f4" + ], + "markers": "python_version >= '3.10'", + "version": "==3.20.0" + }, "flickrapi": { "hashes": [ "sha256:28e6d0ebafc83b79c58d5056b3370fb2ae6618e386fa691246f3240dbcd8b967", @@ -653,6 +668,14 @@ "markers": "python_version >= '3.8'", "version": "==0.28.1" }, + "identify": { + "hashes": [ + "sha256:1181ef7608e00704db228516541eb83a88a9f94433a8c80bb9b5bd54b1d81757", + "sha256:e4f4864b96c6557ef2a1e1c951771838f4edc9df3a72ec7118b338801b11c7bf" + ], + "markers": "python_version >= '3.9'", + "version": "==2.6.15" + }, "idna": { "hashes": [ "sha256:12f65c9b470abda6dc35cf8e63cc574b1c52b11df2c86030af0ac09b01b13ea9", @@ -1162,6 +1185,14 @@ "markers": "python_version >= '3.5'", "version": "==1.6.0" }, + "nodeenv": { + "hashes": [ + "sha256:6ec12890a2dab7946721edbfbcd91f3319c6ccc9aec47be7c7e6b7011ee6645f", + "sha256:ba11c9782d29c27c70ffbdda2d7415098754709be8a7056d79a737cd901155c9" + ], + "markers": "python_version >= '2.7' and python_version not in '3.0, 3.1, 3.2, 3.3, 3.4, 3.5, 3.6'", + "version": "==1.9.1" + }, "notebook-shim": { "hashes": [ "sha256:411a5be4e9dc882a074ccbcae671eda64cceb068767e9a3419096986560e1cef", @@ -1259,14 +1290,6 @@ "markers": "python_version >= '3.8'", "version": "==3.3.1" }, - "overrides": { - "hashes": [ - "sha256:55158fa3d93b98cc75299b1e67078ad9003ca27945c76162c1c0766d6f91820a", - "sha256:c7ed9d062f78b8e4c1a7b70bd8796b35ead4d9f510227ef9c5dc7626c60d7e49" - ], - "markers": "python_version >= '3.6'", - "version": "==7.7.0" - }, "packaging": { "hashes": [ "sha256:29572ef2b1f17581046b3a2227d5c611fb25ec70ca1ba8554b24b0e69331a484", @@ -1491,6 +1514,15 @@ "markers": "python_version >= '3.8'", "version": "==6.3.1" }, + "pre-commit": { + "hashes": [ + "sha256:2b0747ad7e6e967169136edffee14c16e148a778a54e4f967921aa1ebf2308d8", + "sha256:499fe450cc9d42e9d58e606262795ecb64dd05438943c62b66f6a8673da30b16" + ], + "index": "pypi", + "markers": "python_version >= '3.9'", + "version": "==4.3.0" + }, "prometheus-client": { "hashes": [ "sha256:6ae8f9081eaaaf153a2e959d2e6c4f4fb57b12ef76c8c7980202f1e57b48b2ce", @@ -2248,6 +2280,14 @@ "markers": "python_version >= '3.9'", "version": "==2.5.0" }, + "virtualenv": { + "hashes": [ + "sha256:4f1a845d131133bdff10590489610c98c168ff99dc75d6c96853801f7f67af44", + "sha256:63d106565078d8c8d0b206d48080f938a8b25361e19432d2c9db40d2899c810a" + ], + "markers": "python_version >= '3.8'", + "version": "==20.35.3" + }, "wcwidth": { "hashes": [ "sha256:4d478375d31bc5395a3c55c40ccdf3354688364cd61c4f6adacaa9215d0b3605", @@ -2734,21 +2774,13 @@ "markers": "python_version >= '3.8'", "version": "==5.14.3" }, - "typing-extensions": { - "hashes": [ - "sha256:0cea48d173cc12fa28ecabc3b837ea3cf6f38c6d1136f85cbaaf598984861466", - "sha256:f0fa19c6845758ab08074a0cfa8b7aecb71c999ca73d62883bc25cc018c4e548" - ], - "markers": "python_version >= '3.9'", - "version": "==4.15.0" - }, "virtualenv": { "hashes": [ - "sha256:341f5afa7eee943e4984a9207c025feedd768baff6753cd660c857ceb3e36026", - "sha256:44815b2c9dee7ed86e387b842a84f20b93f7f417f95886ca1996a72a4138eb1a" + "sha256:4f1a845d131133bdff10590489610c98c168ff99dc75d6c96853801f7f67af44", + "sha256:63d106565078d8c8d0b206d48080f938a8b25361e19432d2c9db40d2899c810a" ], "markers": "python_version >= '3.8'", - "version": "==20.34.0" + "version": "==20.35.3" }, "wcwidth": { "hashes": [ diff --git a/data/2025Q4/1-fetch/europeana_1_count.csv b/data/2025Q4/1-fetch/europeana_1_count.csv new file mode 100644 index 00000000..bed629ad --- /dev/null +++ b/data/2025Q4/1-fetch/europeana_1_count.csv @@ -0,0 +1,31 @@ +"DATA_PROVIDER","LEGAL_TOOL","COUNT" +"Museum of Gothenburg","CC ZERO 1.0","1" +"Internet Culturale","INC 1.0","14" +"Hellenic Literary and Historical Archive - Cultural Foundation of the National Bank Of Greece","CC BY 4.0","1" +"Open Society Archives at Central European University","INC 1.0","1" +"National Library of France","INC 1.0","1" +"National Audiovisual Institute France","INC 1.0","6" +"European Library of Information and Culture","CC ZERO 1.0","2" +"European Library of Information and Culture","CC BY-SA 4.0","1" +"National Museum of Art of Catalonia","CC BY-NC 4.0","1" +"German National Library","INC 1.0","7" +"Deutsche Fotothek","CC BY-SA 4.0","2" +"Deutsche Fotothek","INC-EDU 1.0","3" +"Royal Museums of Fine Arts of Belgium","INC 1.0","1" +"DFF – German Film Institute & Film Museum","INC 1.0","10" +"MAK – Museum of Applied Arts","INC 1.0","2" +"GESIS - Leibniz Institute for the Social Sciences. Library Cologne","CC BY-NC-ND 4.0","2" +"GESIS - Leibniz Institute for the Social Sciences. Library Cologne","INC 1.0","2" +"The Albertina Museum","CC MARK 1.0","8" +"Braidense National Library","NOC-OKLR 1.0","1" +"Bauhaus University Weimar. University Library","CC MARK 1.0","2" +"Department of Life Sciences, University of Trieste","CC BY-SA 3.0","5" +"Foundation Virtual Library Miguel de Cervantes","INC 1.0","2" +"Austrian National Library","NOC-NC 1.0","2" +"Interuniversity Health Library","INC 1.0","1" +"National Library of Romania","CC BY-SA 4.0","1" +"Provincial Library Magna Capitana","NOC-OKLR 1.0","1" +"Croatian Academy of Sciences and Arts","INC 1.0","1" +"National Conservatory of Arts and Crafts","CC MARK 1.0","1" +"National Library of Israel","CC BY 4.0","14" +"Library of the Wroclaw University","CC MARK 1.0","4" diff --git a/env.example b/env.example index b53d4f03..f62736e9 100644 --- a/env.example +++ b/env.example @@ -10,7 +10,7 @@ # https://googleapis.github.io/google-api-python-client/docs/epy/index.html # "string, key obtained from https://code.google.com/apis/console" -# GCS_DEVELOPER_KEY = +# GCS_DEVELOPER_KEY = AIzaSyBcFgVx6oPcBu7sBkYIuaQeYg-aLttBqmo # https://developers.google.com/custom-search/v1/reference/rest/v1/Search # "The identifier of an engine created using the Programmable Search Engine @@ -19,7 +19,7 @@ # https://googleapis.github.io/google-api-python-client/docs/dyn/customsearch_v1.cse.html # "string, The Programmable Search Engine ID to use for this request." -# GCS_CX = +# GCS_CX # GitHub diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py new file mode 100644 index 00000000..df8c4573 --- /dev/null +++ b/scripts/1-fetch/europeana_fetch.py @@ -0,0 +1,265 @@ +#!/usr/bin/env python +""" +Fetch high-level Europeana statistics for Openverse integration. +Aggregates data by DATA_PROVIDER, LEGAL_TOOL, and COUNT. +""" + +# Standard library +import argparse +import csv +import os +import sys +import textwrap +import time +import traceback +from collections import defaultdict + +# Third-party +import requests +from dotenv import load_dotenv +from pygments import highlight +from pygments.formatters import TerminalFormatter +from pygments.lexers import PythonTracebackLexer + +# Add parent directory so shared can be imported +sys.path.append(os.path.join(os.path.dirname(__file__), "..")) + +# First-party/Local +import shared # noqa: E402 + +# Setup +LOGGER, PATHS = shared.setup(__file__) + +# Load environment variables +load_dotenv(PATHS["dotenv"]) + +# Constants +EUROPEANA_API_KEY = os.getenv("EUROPEANA_API_KEY") +BASE_URL = "https://api.europeana.eu/record/v2/search.json" +FILE_STATS = shared.path_join(PATHS["data_phase"], "europeana_1_count.csv") +HEADER_STATS = ["DATA_PROVIDER", "LEGAL_TOOL", "COUNT"] +PLAN_COMPLETED_INDEX = 1 # Placeholder, not used in this script +QUARTER = os.path.basename(PATHS["data_quarter"]) + +# Log the start of script execution +LOGGER.info("Europeana high-level stats script execution started.") + + +def parse_arguments(): + """ + Parse command-line options, returns parsed argument namespace. + """ + LOGGER.info("Parsing command-line options.") + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--limit", + type=int, + default=100, + help="Limit number of results to fetch (default: 100).", + ) + parser.add_argument( + "--enable-save", + action="store_true", + help="Enable saving aggregated results to CSV.", + ) + parser.add_argument( + "--enable-git", + action="store_true", + help="Enable git actions (fetch, merge, add, commit, push).", + ) + args = parser.parse_args() + if not args.enable_save and args.enable_git: + parser.error("--enable-git requires --enable-save") + return args + + +def initialize_data_file(file_path, header): + """Initialize the data file with a header if it doesn't exist.""" + if not os.path.isfile(file_path): + with open(file_path, "w", newline="") as file_obj: + writer = csv.DictWriter( + file_obj, fieldnames=header, dialect="unix" + ) + writer.writeheader() + + +def initialize_all_data_files(args): + """Ensure data directories and files exist.""" + if not args.enable_save: + return + os.makedirs(PATHS["data_phase"], exist_ok=True) + initialize_data_file(FILE_STATS, HEADER_STATS) + + +def fetch_europeana_data(args): + """ + Fetch and aggregate data from the Europeana Search API + by DATA_PROVIDER and LEGAL_TOOL. + """ + LOGGER.info("Fetching aggregated Europeana data.") + + if not EUROPEANA_API_KEY: + LOGGER.error("EUROPEANA_API_KEY not found in environment variables") + return [] + + # Try different queries to get diverse content + queries = ["art", "history", "science", "music", "photography"] + items_per_query = max(20, args.limit // len(queries)) + all_items = [] + + for query in queries: + params = { + "wskey": EUROPEANA_API_KEY, + "rows": min(items_per_query, 20), + "profile": "rich", + "query": query, + } + + try: + LOGGER.info( + f"Fetching {params['rows']} records for query: '{query}'" + ) + response = requests.get(BASE_URL, params=params, timeout=30) + response.raise_for_status() + results = response.json() + items = results.get("items", []) + all_items.extend(items) + LOGGER.info(f"Retrieved {len(items)} items for '{query}'") + time.sleep(1) # Be nice to the API + except requests.RequestException as e: + LOGGER.warning(f"Failed to fetch data for query '{query}': {e}") + continue + + if not all_items: + LOGGER.error("No items retrieved from any query") + return [] + + LOGGER.info(f"Total items retrieved: {len(all_items)}") + + # Aggregate by data provider and legal tool + aggregation = defaultdict(lambda: defaultdict(int)) + + for item in all_items: + # Handle dataProvider (can be array or string) + data_providers = item.get("dataProvider", []) + if isinstance(data_providers, str): + data_provider = data_providers + elif data_providers and isinstance(data_providers, list): + data_provider = data_providers[0] if data_providers else "Unknown" + else: + data_provider = "Unknown" + + # Handle rights/license information - extract only the license code + rights = item.get("rights", []) + if isinstance(rights, str): + legal_tool = rights + elif rights and isinstance(rights, list): + legal_tool = rights[0] if rights else "Unknown" + else: + legal_tool = "Unknown" + + # Simplify legal tool (e.g., extract 'by/4.0/' → 'CC BY 4.0') + if ( + legal_tool + and legal_tool != "Unknown" + and legal_tool.startswith("http") + ): + parts = legal_tool.strip("/").split("/") + last_parts = parts[-2:] # e.g., ['by', '4.0'] or ['InC', '1.0'] + if last_parts: + # Join neatly with spaces and add CC if + # it’s a Creative Commons license + joined = " ".join(part.upper() for part in last_parts if part) + if "creativecommons.org" in legal_tool: + legal_tool = f"CC {joined}" + else: + legal_tool = joined + else: + legal_tool = "Unknown" + + aggregation[data_provider][legal_tool] += 1 + + # Convert to flat list + output = [] + for provider, licenses in aggregation.items(): + for legal_tool, count in licenses.items(): + output.append( + { + "DATA_PROVIDER": provider, + "LEGAL_TOOL": legal_tool, + "COUNT": count, + } + ) + + LOGGER.info( + f"Aggregated data into {len(output)} provider-license combinations" + ) + return output + + +def save_to_csv(args, data): + """Save aggregated data to CSV.""" + if not args.enable_save: + LOGGER.info("Save disabled - skipping file write") + return + if not data: + LOGGER.warning("No data to save") + return + + with open(FILE_STATS, "w", newline="") as file_obj: + writer = csv.DictWriter( + file_obj, fieldnames=HEADER_STATS, dialect="unix" + ) + writer.writeheader() + for row in data: + writer.writerow(row) + LOGGER.info(f"Saved {len(data)} aggregated rows to {FILE_STATS}.") + + +def main(): + args = parse_arguments() + shared.paths_log(LOGGER, PATHS) + shared.git_fetch_and_merge(args, PATHS["repo"]) + initialize_all_data_files(args) + + data = fetch_europeana_data(args) + save_to_csv(args, data) + + args = shared.git_add_and_commit( + args, + PATHS["repo"], + PATHS["data_quarter"], + f"Add and commit Europeana high-level statistics for {QUARTER}", + ) + shared.git_push_changes(args, PATHS["repo"]) + + LOGGER.info("Europeana high-level stats script completed successfully.") + + +if __name__ == "__main__": + try: + main() + except shared.QuantifyingException as e: + if e.exit_code == 0: + LOGGER.info(e.message) + else: + LOGGER.error(e.message) + sys.exit(e.exit_code) + except SystemExit as e: + if e.code != 0: + LOGGER.error(f"System exit with code: {e.code}") + sys.exit(e.code) + except KeyboardInterrupt: + LOGGER.info("(130) Halted via KeyboardInterrupt.") + sys.exit(130) + except Exception: + traceback_formatted = textwrap.indent( + highlight( + traceback.format_exc(), + PythonTracebackLexer(), + TerminalFormatter(), + ), + " ", + ) + LOGGER.critical(f"(1) Unhandled exception:\n{traceback_formatted}") + sys.exit(1) From 7a7766d5bf7e95897183d75eb32adc6956cc938d Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Wed, 15 Oct 2025 23:08:02 +0300 Subject: [PATCH 02/30] Remove unnecessary CSV file --- data/2025Q4/1-fetch/europeana_1_count.csv | 31 ----------------------- env.example | 7 +++-- 2 files changed, 5 insertions(+), 33 deletions(-) delete mode 100644 data/2025Q4/1-fetch/europeana_1_count.csv diff --git a/data/2025Q4/1-fetch/europeana_1_count.csv b/data/2025Q4/1-fetch/europeana_1_count.csv deleted file mode 100644 index bed629ad..00000000 --- a/data/2025Q4/1-fetch/europeana_1_count.csv +++ /dev/null @@ -1,31 +0,0 @@ -"DATA_PROVIDER","LEGAL_TOOL","COUNT" -"Museum of Gothenburg","CC ZERO 1.0","1" -"Internet Culturale","INC 1.0","14" -"Hellenic Literary and Historical Archive - Cultural Foundation of the National Bank Of Greece","CC BY 4.0","1" -"Open Society Archives at Central European University","INC 1.0","1" -"National Library of France","INC 1.0","1" -"National Audiovisual Institute France","INC 1.0","6" -"European Library of Information and Culture","CC ZERO 1.0","2" -"European Library of Information and Culture","CC BY-SA 4.0","1" -"National Museum of Art of Catalonia","CC BY-NC 4.0","1" -"German National Library","INC 1.0","7" -"Deutsche Fotothek","CC BY-SA 4.0","2" -"Deutsche Fotothek","INC-EDU 1.0","3" -"Royal Museums of Fine Arts of Belgium","INC 1.0","1" -"DFF – German Film Institute & Film Museum","INC 1.0","10" -"MAK – Museum of Applied Arts","INC 1.0","2" -"GESIS - Leibniz Institute for the Social Sciences. Library Cologne","CC BY-NC-ND 4.0","2" -"GESIS - Leibniz Institute for the Social Sciences. Library Cologne","INC 1.0","2" -"The Albertina Museum","CC MARK 1.0","8" -"Braidense National Library","NOC-OKLR 1.0","1" -"Bauhaus University Weimar. University Library","CC MARK 1.0","2" -"Department of Life Sciences, University of Trieste","CC BY-SA 3.0","5" -"Foundation Virtual Library Miguel de Cervantes","INC 1.0","2" -"Austrian National Library","NOC-NC 1.0","2" -"Interuniversity Health Library","INC 1.0","1" -"National Library of Romania","CC BY-SA 4.0","1" -"Provincial Library Magna Capitana","NOC-OKLR 1.0","1" -"Croatian Academy of Sciences and Arts","INC 1.0","1" -"National Conservatory of Arts and Crafts","CC MARK 1.0","1" -"National Library of Israel","CC BY 4.0","14" -"Library of the Wroclaw University","CC MARK 1.0","4" diff --git a/env.example b/env.example index f62736e9..81a724b5 100644 --- a/env.example +++ b/env.example @@ -10,7 +10,7 @@ # https://googleapis.github.io/google-api-python-client/docs/epy/index.html # "string, key obtained from https://code.google.com/apis/console" -# GCS_DEVELOPER_KEY = AIzaSyBcFgVx6oPcBu7sBkYIuaQeYg-aLttBqmo +# GCS_DEVELOPER_KEY = # https://developers.google.com/custom-search/v1/reference/rest/v1/Search # "The identifier of an engine created using the Programmable Search Engine @@ -19,7 +19,7 @@ # https://googleapis.github.io/google-api-python-client/docs/dyn/customsearch_v1.cse.html # "string, The Programmable Search Engine ID to use for this request." -# GCS_CX +# GCS_CX = # GitHub @@ -29,3 +29,6 @@ # https://docs.github.com/en/rest/authentication/authenticating-to-the-rest-api # GH_TOKEN = +# "The flickr developer guide: https://www.flickr.com/services/developer/" +# FLICKR_API_KEY = +# FLICKR_API_SECRET = From 9e52f73c041ab5990d183f813a98e4da125c4a15 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Sat, 18 Oct 2025 10:41:16 +0300 Subject: [PATCH 03/30] Revert "Done the necessary changes" This reverts commit 3190e3c8682c659d2429226960b46a73eda9020b. --- Pipfile | 1 - Pipfile.lock | 88 +++------- scripts/1-fetch/europeana_fetch.py | 265 ----------------------------- 3 files changed, 28 insertions(+), 326 deletions(-) delete mode 100644 scripts/1-fetch/europeana_fetch.py diff --git a/Pipfile b/Pipfile index 889b706f..3fc0133d 100644 --- a/Pipfile +++ b/Pipfile @@ -22,7 +22,6 @@ requests = ">=2.31.0" seaborn = "*" urllib3 = ">=2.5.0" wordcloud = "*" -pre-commit = "*" [dev-packages] black = "*" diff --git a/Pipfile.lock b/Pipfile.lock index 17810845..741d5da4 100644 --- a/Pipfile.lock +++ b/Pipfile.lock @@ -1,7 +1,7 @@ { "_meta": { "hash": { - "sha256": "444f6f70a55f14caf5149c4c9b487780805c162fb5d0add1433c8bf74769004c" + "sha256": "ffa89d9d058b31c7b28aca6ed62f13f1273b3f8b92f5fa4d279dd3737844d8e8" }, "pipfile-spec": 6, "requires": { @@ -24,6 +24,14 @@ "markers": "python_version >= '3.9'", "version": "==4.11.0" }, + "appnope": { + "hashes": [ + "sha256:1de3860566df9caf38f01f86f65e0e13e379af54f9e4bee1e66b48f2efffd1ee", + "sha256:502575ee11cd7a28c0205f379b525beefebab9d161b7c964670864014ed7213c" + ], + "markers": "python_version >= '3.6'", + "version": "==0.1.4" + }, "argon2-cffi": { "hashes": [ "sha256:694ae5cc8a42f4c4e2bf2ca0e64e51e23a040c6a517a85074683d3959e1346c1", @@ -229,14 +237,6 @@ "markers": "python_version >= '3.9'", "version": "==2.0.0" }, - "cfgv": { - "hashes": [ - "sha256:b7265b1f29fd3316bfcd2b330d63d024f2bfd8bcb8b0272f8e19a504856c48f9", - "sha256:e52591d4c5f5dead8e0f673fb16db7949d2cfb3f7da4582893288f0ded8fe560" - ], - "markers": "python_version >= '3.8'", - "version": "==3.4.0" - }, "charset-normalizer": { "hashes": [ "sha256:00237675befef519d9af72169d8604a067d92755e84fe76492fef5441db05b91", @@ -468,13 +468,6 @@ "markers": "python_version >= '2.7' and python_version not in '3.0, 3.1, 3.2, 3.3, 3.4'", "version": "==0.7.1" }, - "distlib": { - "hashes": [ - "sha256:9659f7d87e46584a30b5780e43ac7a2143098441670ff0a49d5f9034c54a6c16", - "sha256:feec40075be03a04501a973d81f633735b4b69f98b05450592310c0f401a4e0d" - ], - "version": "==0.4.0" - }, "executing": { "hashes": [ "sha256:3632cc370565f6648cc328b32435bd120a1e4ebb20c77e3fdde9a13cd1e533c4", @@ -490,14 +483,6 @@ ], "version": "==2.21.2" }, - "filelock": { - "hashes": [ - "sha256:339b4732ffda5cd79b13f4e2711a31b0365ce445d95d243bb996273d072546a2", - "sha256:711e943b4ec6be42e1d4e6690b48dc175c822967466bb31c0c293f34334c13f4" - ], - "markers": "python_version >= '3.10'", - "version": "==3.20.0" - }, "flickrapi": { "hashes": [ "sha256:28e6d0ebafc83b79c58d5056b3370fb2ae6618e386fa691246f3240dbcd8b967", @@ -668,14 +653,6 @@ "markers": "python_version >= '3.8'", "version": "==0.28.1" }, - "identify": { - "hashes": [ - "sha256:1181ef7608e00704db228516541eb83a88a9f94433a8c80bb9b5bd54b1d81757", - "sha256:e4f4864b96c6557ef2a1e1c951771838f4edc9df3a72ec7118b338801b11c7bf" - ], - "markers": "python_version >= '3.9'", - "version": "==2.6.15" - }, "idna": { "hashes": [ "sha256:12f65c9b470abda6dc35cf8e63cc574b1c52b11df2c86030af0ac09b01b13ea9", @@ -1185,14 +1162,6 @@ "markers": "python_version >= '3.5'", "version": "==1.6.0" }, - "nodeenv": { - "hashes": [ - "sha256:6ec12890a2dab7946721edbfbcd91f3319c6ccc9aec47be7c7e6b7011ee6645f", - "sha256:ba11c9782d29c27c70ffbdda2d7415098754709be8a7056d79a737cd901155c9" - ], - "markers": "python_version >= '2.7' and python_version not in '3.0, 3.1, 3.2, 3.3, 3.4, 3.5, 3.6'", - "version": "==1.9.1" - }, "notebook-shim": { "hashes": [ "sha256:411a5be4e9dc882a074ccbcae671eda64cceb068767e9a3419096986560e1cef", @@ -1290,6 +1259,14 @@ "markers": "python_version >= '3.8'", "version": "==3.3.1" }, + "overrides": { + "hashes": [ + "sha256:55158fa3d93b98cc75299b1e67078ad9003ca27945c76162c1c0766d6f91820a", + "sha256:c7ed9d062f78b8e4c1a7b70bd8796b35ead4d9f510227ef9c5dc7626c60d7e49" + ], + "markers": "python_version >= '3.6'", + "version": "==7.7.0" + }, "packaging": { "hashes": [ "sha256:29572ef2b1f17581046b3a2227d5c611fb25ec70ca1ba8554b24b0e69331a484", @@ -1514,15 +1491,6 @@ "markers": "python_version >= '3.8'", "version": "==6.3.1" }, - "pre-commit": { - "hashes": [ - "sha256:2b0747ad7e6e967169136edffee14c16e148a778a54e4f967921aa1ebf2308d8", - "sha256:499fe450cc9d42e9d58e606262795ecb64dd05438943c62b66f6a8673da30b16" - ], - "index": "pypi", - "markers": "python_version >= '3.9'", - "version": "==4.3.0" - }, "prometheus-client": { "hashes": [ "sha256:6ae8f9081eaaaf153a2e959d2e6c4f4fb57b12ef76c8c7980202f1e57b48b2ce", @@ -2280,14 +2248,6 @@ "markers": "python_version >= '3.9'", "version": "==2.5.0" }, - "virtualenv": { - "hashes": [ - "sha256:4f1a845d131133bdff10590489610c98c168ff99dc75d6c96853801f7f67af44", - "sha256:63d106565078d8c8d0b206d48080f938a8b25361e19432d2c9db40d2899c810a" - ], - "markers": "python_version >= '3.8'", - "version": "==20.35.3" - }, "wcwidth": { "hashes": [ "sha256:4d478375d31bc5395a3c55c40ccdf3354688364cd61c4f6adacaa9215d0b3605", @@ -2774,13 +2734,21 @@ "markers": "python_version >= '3.8'", "version": "==5.14.3" }, + "typing-extensions": { + "hashes": [ + "sha256:0cea48d173cc12fa28ecabc3b837ea3cf6f38c6d1136f85cbaaf598984861466", + "sha256:f0fa19c6845758ab08074a0cfa8b7aecb71c999ca73d62883bc25cc018c4e548" + ], + "markers": "python_version >= '3.9'", + "version": "==4.15.0" + }, "virtualenv": { "hashes": [ - "sha256:4f1a845d131133bdff10590489610c98c168ff99dc75d6c96853801f7f67af44", - "sha256:63d106565078d8c8d0b206d48080f938a8b25361e19432d2c9db40d2899c810a" + "sha256:341f5afa7eee943e4984a9207c025feedd768baff6753cd660c857ceb3e36026", + "sha256:44815b2c9dee7ed86e387b842a84f20b93f7f417f95886ca1996a72a4138eb1a" ], "markers": "python_version >= '3.8'", - "version": "==20.35.3" + "version": "==20.34.0" }, "wcwidth": { "hashes": [ diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py deleted file mode 100644 index df8c4573..00000000 --- a/scripts/1-fetch/europeana_fetch.py +++ /dev/null @@ -1,265 +0,0 @@ -#!/usr/bin/env python -""" -Fetch high-level Europeana statistics for Openverse integration. -Aggregates data by DATA_PROVIDER, LEGAL_TOOL, and COUNT. -""" - -# Standard library -import argparse -import csv -import os -import sys -import textwrap -import time -import traceback -from collections import defaultdict - -# Third-party -import requests -from dotenv import load_dotenv -from pygments import highlight -from pygments.formatters import TerminalFormatter -from pygments.lexers import PythonTracebackLexer - -# Add parent directory so shared can be imported -sys.path.append(os.path.join(os.path.dirname(__file__), "..")) - -# First-party/Local -import shared # noqa: E402 - -# Setup -LOGGER, PATHS = shared.setup(__file__) - -# Load environment variables -load_dotenv(PATHS["dotenv"]) - -# Constants -EUROPEANA_API_KEY = os.getenv("EUROPEANA_API_KEY") -BASE_URL = "https://api.europeana.eu/record/v2/search.json" -FILE_STATS = shared.path_join(PATHS["data_phase"], "europeana_1_count.csv") -HEADER_STATS = ["DATA_PROVIDER", "LEGAL_TOOL", "COUNT"] -PLAN_COMPLETED_INDEX = 1 # Placeholder, not used in this script -QUARTER = os.path.basename(PATHS["data_quarter"]) - -# Log the start of script execution -LOGGER.info("Europeana high-level stats script execution started.") - - -def parse_arguments(): - """ - Parse command-line options, returns parsed argument namespace. - """ - LOGGER.info("Parsing command-line options.") - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument( - "--limit", - type=int, - default=100, - help="Limit number of results to fetch (default: 100).", - ) - parser.add_argument( - "--enable-save", - action="store_true", - help="Enable saving aggregated results to CSV.", - ) - parser.add_argument( - "--enable-git", - action="store_true", - help="Enable git actions (fetch, merge, add, commit, push).", - ) - args = parser.parse_args() - if not args.enable_save and args.enable_git: - parser.error("--enable-git requires --enable-save") - return args - - -def initialize_data_file(file_path, header): - """Initialize the data file with a header if it doesn't exist.""" - if not os.path.isfile(file_path): - with open(file_path, "w", newline="") as file_obj: - writer = csv.DictWriter( - file_obj, fieldnames=header, dialect="unix" - ) - writer.writeheader() - - -def initialize_all_data_files(args): - """Ensure data directories and files exist.""" - if not args.enable_save: - return - os.makedirs(PATHS["data_phase"], exist_ok=True) - initialize_data_file(FILE_STATS, HEADER_STATS) - - -def fetch_europeana_data(args): - """ - Fetch and aggregate data from the Europeana Search API - by DATA_PROVIDER and LEGAL_TOOL. - """ - LOGGER.info("Fetching aggregated Europeana data.") - - if not EUROPEANA_API_KEY: - LOGGER.error("EUROPEANA_API_KEY not found in environment variables") - return [] - - # Try different queries to get diverse content - queries = ["art", "history", "science", "music", "photography"] - items_per_query = max(20, args.limit // len(queries)) - all_items = [] - - for query in queries: - params = { - "wskey": EUROPEANA_API_KEY, - "rows": min(items_per_query, 20), - "profile": "rich", - "query": query, - } - - try: - LOGGER.info( - f"Fetching {params['rows']} records for query: '{query}'" - ) - response = requests.get(BASE_URL, params=params, timeout=30) - response.raise_for_status() - results = response.json() - items = results.get("items", []) - all_items.extend(items) - LOGGER.info(f"Retrieved {len(items)} items for '{query}'") - time.sleep(1) # Be nice to the API - except requests.RequestException as e: - LOGGER.warning(f"Failed to fetch data for query '{query}': {e}") - continue - - if not all_items: - LOGGER.error("No items retrieved from any query") - return [] - - LOGGER.info(f"Total items retrieved: {len(all_items)}") - - # Aggregate by data provider and legal tool - aggregation = defaultdict(lambda: defaultdict(int)) - - for item in all_items: - # Handle dataProvider (can be array or string) - data_providers = item.get("dataProvider", []) - if isinstance(data_providers, str): - data_provider = data_providers - elif data_providers and isinstance(data_providers, list): - data_provider = data_providers[0] if data_providers else "Unknown" - else: - data_provider = "Unknown" - - # Handle rights/license information - extract only the license code - rights = item.get("rights", []) - if isinstance(rights, str): - legal_tool = rights - elif rights and isinstance(rights, list): - legal_tool = rights[0] if rights else "Unknown" - else: - legal_tool = "Unknown" - - # Simplify legal tool (e.g., extract 'by/4.0/' → 'CC BY 4.0') - if ( - legal_tool - and legal_tool != "Unknown" - and legal_tool.startswith("http") - ): - parts = legal_tool.strip("/").split("/") - last_parts = parts[-2:] # e.g., ['by', '4.0'] or ['InC', '1.0'] - if last_parts: - # Join neatly with spaces and add CC if - # it’s a Creative Commons license - joined = " ".join(part.upper() for part in last_parts if part) - if "creativecommons.org" in legal_tool: - legal_tool = f"CC {joined}" - else: - legal_tool = joined - else: - legal_tool = "Unknown" - - aggregation[data_provider][legal_tool] += 1 - - # Convert to flat list - output = [] - for provider, licenses in aggregation.items(): - for legal_tool, count in licenses.items(): - output.append( - { - "DATA_PROVIDER": provider, - "LEGAL_TOOL": legal_tool, - "COUNT": count, - } - ) - - LOGGER.info( - f"Aggregated data into {len(output)} provider-license combinations" - ) - return output - - -def save_to_csv(args, data): - """Save aggregated data to CSV.""" - if not args.enable_save: - LOGGER.info("Save disabled - skipping file write") - return - if not data: - LOGGER.warning("No data to save") - return - - with open(FILE_STATS, "w", newline="") as file_obj: - writer = csv.DictWriter( - file_obj, fieldnames=HEADER_STATS, dialect="unix" - ) - writer.writeheader() - for row in data: - writer.writerow(row) - LOGGER.info(f"Saved {len(data)} aggregated rows to {FILE_STATS}.") - - -def main(): - args = parse_arguments() - shared.paths_log(LOGGER, PATHS) - shared.git_fetch_and_merge(args, PATHS["repo"]) - initialize_all_data_files(args) - - data = fetch_europeana_data(args) - save_to_csv(args, data) - - args = shared.git_add_and_commit( - args, - PATHS["repo"], - PATHS["data_quarter"], - f"Add and commit Europeana high-level statistics for {QUARTER}", - ) - shared.git_push_changes(args, PATHS["repo"]) - - LOGGER.info("Europeana high-level stats script completed successfully.") - - -if __name__ == "__main__": - try: - main() - except shared.QuantifyingException as e: - if e.exit_code == 0: - LOGGER.info(e.message) - else: - LOGGER.error(e.message) - sys.exit(e.exit_code) - except SystemExit as e: - if e.code != 0: - LOGGER.error(f"System exit with code: {e.code}") - sys.exit(e.code) - except KeyboardInterrupt: - LOGGER.info("(130) Halted via KeyboardInterrupt.") - sys.exit(130) - except Exception: - traceback_formatted = textwrap.indent( - highlight( - traceback.format_exc(), - PythonTracebackLexer(), - TerminalFormatter(), - ), - " ", - ) - LOGGER.critical(f"(1) Unhandled exception:\n{traceback_formatted}") - sys.exit(1) From 73f9aa10db02fe844b0b7ff42c033ea69e8b21ef Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Sat, 18 Oct 2025 11:10:11 +0300 Subject: [PATCH 04/30] Done the necessary changes --- env.example | 1 + scripts/1-fetch/europeana_fetch.py | 265 +++++++++++++++++++++++++++++ 2 files changed, 266 insertions(+) create mode 100755 scripts/1-fetch/europeana_fetch.py diff --git a/env.example b/env.example index 81a724b5..6727e8da 100644 --- a/env.example +++ b/env.example @@ -30,5 +30,6 @@ # GH_TOKEN = # "The flickr developer guide: https://www.flickr.com/services/developer/" + # FLICKR_API_KEY = # FLICKR_API_SECRET = diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py new file mode 100755 index 00000000..328ce8e9 --- /dev/null +++ b/scripts/1-fetch/europeana_fetch.py @@ -0,0 +1,265 @@ +#!/usr/bin/env python +""" +Fetch high-level Europeana statistics for Quantifying the Commons. +Aggregates data by DATA_PROVIDER, LEGAL_TOOL, and COUNT. +""" + +# Standard library +import argparse +import csv +import os +import sys +import textwrap +import time +import traceback +from collections import defaultdict + +# Third-party +import requests +from dotenv import load_dotenv +from pygments import highlight +from pygments.formatters import TerminalFormatter +from pygments.lexers import PythonTracebackLexer + +# Add parent directory so shared can be imported +sys.path.append(os.path.join(os.path.dirname(__file__), "..")) + +# First-party/Local +import shared # noqa: E402 + +# Setup +LOGGER, PATHS = shared.setup(__file__) + +# Load environment variables +load_dotenv(PATHS["dotenv"]) + +# Constants +EUROPEANA_API_KEY = os.getenv("EUROPEANA_API_KEY") +BASE_URL = "https://api.europeana.eu/record/v2/search.json" +FILE_STATS = shared.path_join(PATHS["data_phase"], "europeana_1_count.csv") +HEADER_STATS = ["DATA_PROVIDER", "LEGAL_TOOL", "COUNT"] +QUARTER = os.path.basename(PATHS["data_quarter"]) + +# Log the start of script execution +LOGGER.info("Europeana high-level stats script execution started.") + + +def parse_arguments(): + """ + Parse command-line options, returns parsed argument namespace. + """ + LOGGER.info("Parsing command-line options.") + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--limit", + type=int, + default=100, + help="Limit number of results to fetch (default: 100).", + ) + parser.add_argument( + "--enable-save", + action="store_true", + help="Enable saving aggregated results to CSV.", + ) + parser.add_argument( + "--enable-git", + action="store_true", + help="Enable git actions (fetch, merge, add, commit, push).", + ) + args = parser.parse_args() + if not args.enable_save and args.enable_git: + parser.error("--enable-git requires --enable-save") + return args + + +def initialize_data_file(file_path, header): + """Initialize the data file with a header if it doesn't exist.""" + if not os.path.isfile(file_path): + with open(file_path, "w", newline="") as file_obj: + writer = csv.DictWriter( + file_obj, fieldnames=header, dialect="unix" + ) + writer.writeheader() + + +def initialize_all_data_files(args): + """Ensure data directories and files exist.""" + if not args.enable_save: + return + os.makedirs(PATHS["data_phase"], exist_ok=True) + initialize_data_file(FILE_STATS, HEADER_STATS) + + +def fetch_europeana_data(args): + """ + Fetch and aggregate data from the Europeana Search API + by DATA_PROVIDER and LEGAL_TOOL. + """ + LOGGER.info("Fetching aggregated Europeana data.") + + if not EUROPEANA_API_KEY: + raise shared.QuantifyingException( + "EUROPEANA_API_KEY not found in environment variables", 1 + ) + + # Try different queries to get diverse content + queries = ["art", "history", "science", "music", "photography"] + items_per_query = max(20, args.limit // len(queries)) + all_items = [] + + for query in queries: + params = { + "wskey": EUROPEANA_API_KEY, + "rows": min(items_per_query, 20), + "profile": "rich", + "query": query, + } + + try: + LOGGER.info( + f"Fetching {params['rows']} records for query: '{query}'" + ) + response = requests.get(BASE_URL, params=params, timeout=30) + response.raise_for_status() + results = response.json() + items = results.get("items", []) + all_items.extend(items) + LOGGER.info(f"Retrieved {len(items)} items for '{query}'") + time.sleep(1) # Be nice to the API + except requests.RequestException as e: + LOGGER.warning(f"Failed to fetch data for query '{query}': {e}") + continue + + if not all_items: + LOGGER.error("No items retrieved from any query") + return [] + + LOGGER.info(f"Total items retrieved: {len(all_items)}") + + # Aggregate by data provider and legal tool + aggregation = defaultdict(lambda: defaultdict(int)) + + for item in all_items: + # Handle dataProvider (can be array or string) + data_providers = item.get("dataProvider", []) + if isinstance(data_providers, str): + data_provider = data_providers + elif data_providers and isinstance(data_providers, list): + data_provider = data_providers[0] if data_providers else "Unknown" + else: + data_provider = "Unknown" + + # Handle rights/license information - extract only the license code + rights = item.get("rights", []) + if isinstance(rights, str): + legal_tool = rights + elif rights and isinstance(rights, list): + legal_tool = rights[0] if rights else "Unknown" + else: + legal_tool = "Unknown" + + # Simplify legal tool (e.g., extract 'by/4.0/' → 'CC BY 4.0') + if ( + legal_tool + and legal_tool != "Unknown" + and legal_tool.startswith("http") + ): + parts = legal_tool.strip("/").split("/") + last_parts = parts[-2:] # e.g., ['by', '4.0'] or ['InC', '1.0'] + if last_parts: + # Join neatly with spaces and add CC if + # it’s a Creative Commons license + joined = " ".join(part.upper() for part in last_parts if part) + if "creativecommons.org" in legal_tool: + legal_tool = f"CC {joined}" + else: + legal_tool = joined + else: + legal_tool = "Unknown" + + aggregation[data_provider][legal_tool] += 1 + + # Convert to flat list + output = [] + for provider, licenses in aggregation.items(): + for legal_tool, count in licenses.items(): + output.append( + { + "DATA_PROVIDER": provider, + "LEGAL_TOOL": legal_tool, + "COUNT": count, + } + ) + + LOGGER.info( + f"Aggregated data into {len(output)} provider-license combinations" + ) + return output + + +def save_to_csv(args, data): + """Save aggregated data to CSV.""" + if not args.enable_save: + LOGGER.info("Save disabled - skipping file write") + return + if not data: + LOGGER.warning("No data to save") + return + + with open(FILE_STATS, "w", newline="") as file_obj: + writer = csv.DictWriter( + file_obj, fieldnames=HEADER_STATS, dialect="unix" + ) + writer.writeheader() + for row in data: + writer.writerow(row) + LOGGER.info(f"Saved {len(data)} aggregated rows to {FILE_STATS}.") + + +def main(): + args = parse_arguments() + shared.paths_log(LOGGER, PATHS) + shared.git_fetch_and_merge(args, PATHS["repo"]) + initialize_all_data_files(args) + + data = fetch_europeana_data(args) + save_to_csv(args, data) + + args = shared.git_add_and_commit( + args, + PATHS["repo"], + PATHS["data_quarter"], + f"Add and commit Europeana high-level statistics for {QUARTER}", + ) + shared.git_push_changes(args, PATHS["repo"]) + + LOGGER.info("Europeana high-level stats script completed successfully.") + + +if __name__ == "__main__": + try: + main() + except shared.QuantifyingException as e: + if e.exit_code == 0: + LOGGER.info(e.message) + else: + LOGGER.error(e.message) + sys.exit(e.exit_code) + except SystemExit as e: + if e.code != 0: + LOGGER.error(f"System exit with code: {e.code}") + sys.exit(e.code) + except KeyboardInterrupt: + LOGGER.info("(130) Halted via KeyboardInterrupt.") + sys.exit(130) + except Exception: + traceback_formatted = textwrap.indent( + highlight( + traceback.format_exc(), + PythonTracebackLexer(), + TerminalFormatter(), + ), + " ", + ) + LOGGER.critical(f"(1) Unhandled exception:\n{traceback_formatted}") + sys.exit(1) From cc2e742ca192e864deba0a39975d2a7f41d03c3b Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Mon, 20 Oct 2025 17:09:18 +0300 Subject: [PATCH 05/30] Fix formatting and linting issues in europeana_fetch.py to comply with pre-commit hooks --- scripts/1-fetch/europeana_fetch.py | 119 +++++++++++++++++++---------- 1 file changed, 77 insertions(+), 42 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index 328ce8e9..4e1dd479 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -1,7 +1,7 @@ #!/usr/bin/env python """ Fetch high-level Europeana statistics for Quantifying the Commons. -Aggregates data by DATA_PROVIDER, LEGAL_TOOL, and COUNT. +Aggregates data by DATA_PROVIDER, LEGAL_TOOL, THEME, and COUNT. """ # Standard library @@ -20,6 +20,7 @@ from pygments import highlight from pygments.formatters import TerminalFormatter from pygments.lexers import PythonTracebackLexer +from requests.adapters import HTTPAdapter, Retry # Add parent directory so shared can be imported sys.path.append(os.path.join(os.path.dirname(__file__), "..")) @@ -37,7 +38,7 @@ EUROPEANA_API_KEY = os.getenv("EUROPEANA_API_KEY") BASE_URL = "https://api.europeana.eu/record/v2/search.json" FILE_STATS = shared.path_join(PATHS["data_phase"], "europeana_1_count.csv") -HEADER_STATS = ["DATA_PROVIDER", "LEGAL_TOOL", "COUNT"] +HEADER_STATS = ["DATA_PROVIDER", "LEGAL_TOOL", "THEME", "COUNT"] QUARTER = os.path.basename(PATHS["data_quarter"]) # Log the start of script execution @@ -45,9 +46,7 @@ def parse_arguments(): - """ - Parse command-line options, returns parsed argument namespace. - """ + """Parse command-line options.""" LOGGER.info("Parsing command-line options.") parser = argparse.ArgumentParser(description=__doc__) parser.add_argument( @@ -90,10 +89,28 @@ def initialize_all_data_files(args): initialize_data_file(FILE_STATS, HEADER_STATS) +def get_requests_session(): + """Create a requests session with retry and headers.""" + max_retries = Retry( + total=5, + backoff_factor=5, + status_forcelist=shared.RETRY_STATUS_FORCELIST, + ) + session = requests.Session() + session.mount("https://", HTTPAdapter(max_retries=max_retries)) + session.headers.update( + { + "accept": "application/json", + "User-Agent": shared.USER_AGENT, + } + ) + return session + + def fetch_europeana_data(args): """ Fetch and aggregate data from the Europeana Search API - by DATA_PROVIDER and LEGAL_TOOL. + by DATA_PROVIDER, LEGAL_TOOL, and THEME. """ LOGGER.info("Fetching aggregated Europeana data.") @@ -102,32 +119,50 @@ def fetch_europeana_data(args): "EUROPEANA_API_KEY not found in environment variables", 1 ) - # Try different queries to get diverse content - queries = ["art", "history", "science", "music", "photography"] - items_per_query = max(20, args.limit // len(queries)) + # Define Europeana themes to query + # Provided in Europeana's site + themes = [ + "art", + "fashion", + "music", + "industrial", + "sport", + "photography", + "archaeology", + ] + + items_per_query = max(20, args.limit // len(themes)) all_items = [] + # Initialize a session for efficient and reliable requests + session = get_requests_session() - for query in queries: + for theme in themes: params = { "wskey": EUROPEANA_API_KEY, "rows": min(items_per_query, 20), "profile": "rich", - "query": query, + "query": "*", + "theme": theme, } try: LOGGER.info( - f"Fetching {params['rows']} records for query: '{query}'" + f"Fetching {params['rows']} records for theme: '{theme}'" ) - response = requests.get(BASE_URL, params=params, timeout=30) - response.raise_for_status() - results = response.json() - items = results.get("items", []) + with session.get(BASE_URL, params=params, timeout=30) as response: + response.raise_for_status() + results = response.json() + items = results.get("items", []) + + # Tag each item with the theme used for easy tracking + for item in items: + item["theme_used"] = theme + all_items.extend(items) - LOGGER.info(f"Retrieved {len(items)} items for '{query}'") - time.sleep(1) # Be nice to the API + LOGGER.info(f"Retrieved {len(items)} items for theme '{theme}'") + time.sleep(1) except requests.RequestException as e: - LOGGER.warning(f"Failed to fetch data for query '{query}': {e}") + LOGGER.warning(f"Failed to fetch data for theme '{theme}': {e}") continue if not all_items: @@ -136,8 +171,8 @@ def fetch_europeana_data(args): LOGGER.info(f"Total items retrieved: {len(all_items)}") - # Aggregate by data provider and legal tool - aggregation = defaultdict(lambda: defaultdict(int)) + # Aggregate by data provider, legal tool, and theme + aggregation = defaultdict(lambda: defaultdict(lambda: defaultdict(int))) for item in all_items: # Handle dataProvider (can be array or string) @@ -145,30 +180,24 @@ def fetch_europeana_data(args): if isinstance(data_providers, str): data_provider = data_providers elif data_providers and isinstance(data_providers, list): - data_provider = data_providers[0] if data_providers else "Unknown" + data_provider = data_providers[0] else: data_provider = "Unknown" - # Handle rights/license information - extract only the license code + # Handle rights/license information rights = item.get("rights", []) if isinstance(rights, str): legal_tool = rights elif rights and isinstance(rights, list): - legal_tool = rights[0] if rights else "Unknown" + legal_tool = rights[0] else: legal_tool = "Unknown" # Simplify legal tool (e.g., extract 'by/4.0/' → 'CC BY 4.0') - if ( - legal_tool - and legal_tool != "Unknown" - and legal_tool.startswith("http") - ): + if legal_tool and legal_tool.startswith("http"): parts = legal_tool.strip("/").split("/") - last_parts = parts[-2:] # e.g., ['by', '4.0'] or ['InC', '1.0'] + last_parts = parts[-2:] if last_parts: - # Join neatly with spaces and add CC if - # it’s a Creative Commons license joined = " ".join(part.upper() for part in last_parts if part) if "creativecommons.org" in legal_tool: legal_tool = f"CC {joined}" @@ -177,22 +206,28 @@ def fetch_europeana_data(args): else: legal_tool = "Unknown" - aggregation[data_provider][legal_tool] += 1 + # Use the theme from the query loop + theme = item.get("theme_used", "Unknown") + + aggregation[data_provider][legal_tool][theme] += 1 # Convert to flat list output = [] for provider, licenses in aggregation.items(): - for legal_tool, count in licenses.items(): - output.append( - { - "DATA_PROVIDER": provider, - "LEGAL_TOOL": legal_tool, - "COUNT": count, - } - ) + for legal_tool, themes_dict in licenses.items(): + for theme, count in themes_dict.items(): + output.append( + { + "DATA_PROVIDER": provider, + "LEGAL_TOOL": legal_tool, + "THEME": theme, + "COUNT": count, + } + ) LOGGER.info( - f"Aggregated data into {len(output)} provider-license combinations" + f"Aggregated data into {len(output)} " + f"provider-license-theme combinations" ) return output From a7a1402f73d32ff3f8409afcec53e85b72baa044 Mon Sep 17 00:00:00 2001 From: Timid Robot Zehta Date: Tue, 21 Oct 2025 10:53:36 +0200 Subject: [PATCH 06/30] update variable name (due to merge with main) --- scripts/1-fetch/europeana_fetch.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index 4e1dd479..d7d5ed1d 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -94,7 +94,7 @@ def get_requests_session(): max_retries = Retry( total=5, backoff_factor=5, - status_forcelist=shared.RETRY_STATUS_FORCELIST, + status_forcelist=shared.STATUS_FORCELIST, ) session = requests.Session() session.mount("https://", HTTPAdapter(max_retries=max_retries)) From e26de976f2f0def8edf4550fce36eb610ab96e17 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Tue, 21 Oct 2025 16:05:52 +0300 Subject: [PATCH 07/30] Add functionality to generate separate Europeana data files with and without themes --- scripts/1-fetch/europeana_fetch.py | 245 +++++++++++++++++++---------- 1 file changed, 158 insertions(+), 87 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index d7d5ed1d..96abe579 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -1,7 +1,9 @@ #!/usr/bin/env python """ Fetch high-level Europeana statistics for Quantifying the Commons. -Aggregates data by DATA_PROVIDER, LEGAL_TOOL, THEME, and COUNT. +Generates two datasets: +1) Without themes (aggregated by DATA_PROVIDER, LEGAL_TOOL) +2) With all themes (aggregated by DATA_PROVIDER, LEGAL_TOOL, THEME) """ # Standard library @@ -37,12 +39,18 @@ # Constants EUROPEANA_API_KEY = os.getenv("EUROPEANA_API_KEY") BASE_URL = "https://api.europeana.eu/record/v2/search.json" -FILE_STATS = shared.path_join(PATHS["data_phase"], "europeana_1_count.csv") -HEADER_STATS = ["DATA_PROVIDER", "LEGAL_TOOL", "THEME", "COUNT"] +FILE_WITH_THEMES = shared.path_join( + PATHS["data_phase"], "europeana_with_themes.csv" +) +FILE_WITHOUT_THEMES = shared.path_join( + PATHS["data_phase"], "europeana_without_themes.csv" +) +HEADER_WITH_THEMES = ["DATA_PROVIDER", "LEGAL_TOOL", "THEME", "COUNT"] +HEADER_WITHOUT_THEMES = ["DATA_PROVIDER", "LEGAL_TOOL", "COUNT"] QUARTER = os.path.basename(PATHS["data_quarter"]) -# Log the start of script execution -LOGGER.info("Europeana high-level stats script execution started.") +# Log start +LOGGER.info("Europeana dual-fetch (with & without themes) script started.") def parse_arguments(): @@ -71,24 +79,6 @@ def parse_arguments(): return args -def initialize_data_file(file_path, header): - """Initialize the data file with a header if it doesn't exist.""" - if not os.path.isfile(file_path): - with open(file_path, "w", newline="") as file_obj: - writer = csv.DictWriter( - file_obj, fieldnames=header, dialect="unix" - ) - writer.writeheader() - - -def initialize_all_data_files(args): - """Ensure data directories and files exist.""" - if not args.enable_save: - return - os.makedirs(PATHS["data_phase"], exist_ok=True) - initialize_data_file(FILE_STATS, HEADER_STATS) - - def get_requests_session(): """Create a requests session with retry and headers.""" max_retries = Retry( @@ -107,20 +97,103 @@ def get_requests_session(): return session -def fetch_europeana_data(args): - """ - Fetch and aggregate data from the Europeana Search API - by DATA_PROVIDER, LEGAL_TOOL, and THEME. - """ - LOGGER.info("Fetching aggregated Europeana data.") +def fetch_europeana_data_without_themes(args): + """Fetch and aggregate Europeana data without specifying themes.""" + LOGGER.info("Fetching Europeana data without themes.") if not EUROPEANA_API_KEY: raise shared.QuantifyingException( "EUROPEANA_API_KEY not found in environment variables", 1 ) - # Define Europeana themes to query - # Provided in Europeana's site + params = { + "wskey": EUROPEANA_API_KEY, + "rows": args.limit, + "profile": "rich", + "query": "*", + } + + session = get_requests_session() + try: + with session.get(BASE_URL, params=params, timeout=30) as response: + response.raise_for_status() + results = response.json() + items = results.get("items", []) + except requests.RequestException as e: + LOGGER.error(f"Failed to fetch data without themes: {e}") + return [] + + LOGGER.info(f"Retrieved {len(items)} items without themes.") + + # --- Aggregate by DATA_PROVIDER + LEGAL_TOOL --- + aggregation = defaultdict(lambda: defaultdict(int)) + + for item in items: + data_providers = item.get("dataProvider", []) + data_provider = ( + data_providers + if isinstance(data_providers, str) + else ( + data_providers[0] + if isinstance(data_providers, list) and data_providers + else "Unknown" + ) + ) + + rights = item.get("rights", []) + legal_tool = ( + rights + if isinstance(rights, str) + else ( + rights[0] if isinstance(rights, list) and rights else "Unknown" + ) + ) + + # Simplify license format if it’s a Creative Commons URL + if ( + legal_tool + and isinstance(legal_tool, str) + and legal_tool.startswith("http") + ): + parts = legal_tool.strip("/").split("/") + last_parts = parts[-2:] + if last_parts: + joined = " ".join(part.upper() for part in last_parts if part) + if "creativecommons.org" in legal_tool: + legal_tool = f"CC {joined}" + else: + legal_tool = joined + else: + legal_tool = "Unknown" + + aggregation[data_provider][legal_tool] += 1 + + # Convert to flat list + output = [] + for provider, licenses in aggregation.items(): + for legal_tool, count in licenses.items(): + output.append( + { + "DATA_PROVIDER": provider, + "LEGAL_TOOL": legal_tool, + "COUNT": count, + } + ) + + LOGGER.info(f"Aggregated data without themes into {len(output)} records.") + return output + + +def fetch_europeana_data_with_themes(args): + """Fetch and aggregate data by DATA_PROVIDER, LEGAL_TOOL, and THEME.""" + LOGGER.info("Fetching aggregated Europeana data with themes.") + + if not EUROPEANA_API_KEY: + raise shared.QuantifyingException( + "EUROPEANA_API_KEY not found in environment variables", 1 + ) + + # Themes from Europeana site themes = [ "art", "fashion", @@ -133,7 +206,6 @@ def fetch_europeana_data(args): items_per_query = max(20, args.limit // len(themes)) all_items = [] - # Initialize a session for efficient and reliable requests session = get_requests_session() for theme in themes: @@ -153,11 +225,8 @@ def fetch_europeana_data(args): response.raise_for_status() results = response.json() items = results.get("items", []) - - # Tag each item with the theme used for easy tracking for item in items: item["theme_used"] = theme - all_items.extend(items) LOGGER.info(f"Retrieved {len(items)} items for theme '{theme}'") time.sleep(1) @@ -166,35 +235,40 @@ def fetch_europeana_data(args): continue if not all_items: - LOGGER.error("No items retrieved from any query") + LOGGER.error("No items retrieved for any theme.") return [] - LOGGER.info(f"Total items retrieved: {len(all_items)}") + LOGGER.info(f"Total items retrieved across all themes: {len(all_items)}") - # Aggregate by data provider, legal tool, and theme + # Aggregate by DATA_PROVIDER + LEGAL_TOOL + THEME aggregation = defaultdict(lambda: defaultdict(lambda: defaultdict(int))) for item in all_items: - # Handle dataProvider (can be array or string) data_providers = item.get("dataProvider", []) - if isinstance(data_providers, str): - data_provider = data_providers - elif data_providers and isinstance(data_providers, list): - data_provider = data_providers[0] - else: - data_provider = "Unknown" + data_provider = ( + data_providers + if isinstance(data_providers, str) + else ( + data_providers[0] + if isinstance(data_providers, list) and data_providers + else "Unknown" + ) + ) - # Handle rights/license information rights = item.get("rights", []) - if isinstance(rights, str): - legal_tool = rights - elif rights and isinstance(rights, list): - legal_tool = rights[0] - else: - legal_tool = "Unknown" + legal_tool = ( + rights + if isinstance(rights, str) + else ( + rights[0] if isinstance(rights, list) and rights else "Unknown" + ) + ) - # Simplify legal tool (e.g., extract 'by/4.0/' → 'CC BY 4.0') - if legal_tool and legal_tool.startswith("http"): + if ( + legal_tool + and isinstance(legal_tool, str) + and legal_tool.startswith("http") + ): parts = legal_tool.strip("/").split("/") last_parts = parts[-2:] if last_parts: @@ -206,9 +280,7 @@ def fetch_europeana_data(args): else: legal_tool = "Unknown" - # Use the theme from the query loop theme = item.get("theme_used", "Unknown") - aggregation[data_provider][legal_tool][theme] += 1 # Convert to flat list @@ -225,50 +297,49 @@ def fetch_europeana_data(args): } ) - LOGGER.info( - f"Aggregated data into {len(output)} " - f"provider-license-theme combinations" - ) + LOGGER.info(f"Aggregated data with themes into {len(output)} records.") return output -def save_to_csv(args, data): - """Save aggregated data to CSV.""" - if not args.enable_save: - LOGGER.info("Save disabled - skipping file write") - return - if not data: - LOGGER.warning("No data to save") - return - - with open(FILE_STATS, "w", newline="") as file_obj: - writer = csv.DictWriter( - file_obj, fieldnames=HEADER_STATS, dialect="unix" - ) +def save_csv(filepath, header, data): + """Save data to a CSV file.""" + with open(filepath, "w", newline="") as f: + writer = csv.DictWriter(f, fieldnames=header) writer.writeheader() - for row in data: - writer.writerow(row) - LOGGER.info(f"Saved {len(data)} aggregated rows to {FILE_STATS}.") + writer.writerows(data) + LOGGER.info(f"Saved {len(data)} rows to {filepath}.") def main(): args = parse_arguments() shared.paths_log(LOGGER, PATHS) shared.git_fetch_and_merge(args, PATHS["repo"]) - initialize_all_data_files(args) - data = fetch_europeana_data(args) - save_to_csv(args, data) + os.makedirs(PATHS["data_phase"], exist_ok=True) + + # Fetch and save data WITHOUT themes (aggregated) + data_no_theme = fetch_europeana_data_without_themes(args) + if args.enable_save and data_no_theme: + save_csv(FILE_WITHOUT_THEMES, HEADER_WITHOUT_THEMES, data_no_theme) + + # Fetch and save data WITH themes (aggregated) + data_with_theme = fetch_europeana_data_with_themes(args) + if args.enable_save and data_with_theme: + save_csv(FILE_WITH_THEMES, HEADER_WITH_THEMES, data_with_theme) + + # Git commit & push + if args.enable_git and args.enable_save: + args = shared.git_add_and_commit( + args, + PATHS["repo"], + PATHS["data_quarter"], + f"Add Europeana files (with and without themes) for {QUARTER}", + ) + shared.git_push_changes(args, PATHS["repo"]) - args = shared.git_add_and_commit( - args, - PATHS["repo"], - PATHS["data_quarter"], - f"Add and commit Europeana high-level statistics for {QUARTER}", + LOGGER.info( + "Europeana dual-fetch (with & without themes) completed successfully." ) - shared.git_push_changes(args, PATHS["repo"]) - - LOGGER.info("Europeana high-level stats script completed successfully.") if __name__ == "__main__": From ee36206ce8c33ea31d0e77de5b5d88348f31a063 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Tue, 21 Oct 2025 17:49:16 +0300 Subject: [PATCH 08/30] =?UTF-8?q?Updated=20script=20to=20use=20Europeana?= =?UTF-8?q?=20API=E2=80=99s=20totalResults?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- scripts/1-fetch/europeana_fetch.py | 348 +++++++++++------------------ 1 file changed, 132 insertions(+), 216 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index 96abe579..52af4475 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -4,6 +4,7 @@ Generates two datasets: 1) Without themes (aggregated by DATA_PROVIDER, LEGAL_TOOL) 2) With all themes (aggregated by DATA_PROVIDER, LEGAL_TOOL, THEME) +Uses totalResults instead of looping through pages for efficiency. """ # Standard library @@ -14,7 +15,6 @@ import textwrap import time import traceback -from collections import defaultdict # Third-party import requests @@ -24,16 +24,13 @@ from pygments.lexers import PythonTracebackLexer from requests.adapters import HTTPAdapter, Retry -# Add parent directory so shared can be imported +# Add parent directory for shared imports sys.path.append(os.path.join(os.path.dirname(__file__), "..")) - # First-party/Local import shared # noqa: E402 # Setup LOGGER, PATHS = shared.setup(__file__) - -# Load environment variables load_dotenv(PATHS["dotenv"]) # Constants @@ -50,19 +47,14 @@ QUARTER = os.path.basename(PATHS["data_quarter"]) # Log start -LOGGER.info("Europeana dual-fetch (with & without themes) script started.") +LOGGER.info( + "Optimized Europeana dual-fetch (using totalResults) script started." +) def parse_arguments(): """Parse command-line options.""" - LOGGER.info("Parsing command-line options.") parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument( - "--limit", - type=int, - default=100, - help="Limit number of results to fetch (default: 100).", - ) parser.add_argument( "--enable-save", action="store_true", @@ -80,120 +72,107 @@ def parse_arguments(): def get_requests_session(): - """Create a requests session with retry and headers.""" + """Create a requests session with retry.""" max_retries = Retry( - total=5, - backoff_factor=5, - status_forcelist=shared.STATUS_FORCELIST, + total=5, backoff_factor=5, status_forcelist=shared.STATUS_FORCELIST ) session = requests.Session() session.mount("https://", HTTPAdapter(max_retries=max_retries)) session.headers.update( - { - "accept": "application/json", - "User-Agent": shared.USER_AGENT, - } + {"accept": "application/json", "User-Agent": shared.USER_AGENT} ) return session -def fetch_europeana_data_without_themes(args): - """Fetch and aggregate Europeana data without specifying themes.""" - LOGGER.info("Fetching Europeana data without themes.") - - if not EUROPEANA_API_KEY: - raise shared.QuantifyingException( - "EUROPEANA_API_KEY not found in environment variables", 1 - ) - +def get_facet_list(session, facet_field): + """Fetch unique values for a facet (e.g., DATA_PROVIDER or RIGHTS).""" params = { "wskey": EUROPEANA_API_KEY, - "rows": args.limit, - "profile": "rich", "query": "*", + "rows": 0, + "facet": facet_field, + "profile": "facets", } + resp = session.get(BASE_URL, params=params, timeout=30) + resp.raise_for_status() + data = resp.json() + facet_values = [ + f["label"] for f in data.get("facets", [])[0].get("fields", []) + ] + return facet_values + + +def simplify_legal_tool(legal_tool): + """Simplify and standardize Creative Commons or license URLs.""" + if ( + legal_tool + and isinstance(legal_tool, str) + and legal_tool.startswith("http") + ): + parts = legal_tool.strip("/").split("/") + last_parts = parts[-2:] + if last_parts: + joined = " ".join(part.upper() for part in last_parts if part) + if "creativecommons.org" in legal_tool: + return f"CC {joined}" + else: + return joined + else: + return "Unknown" + return legal_tool - session = get_requests_session() - try: - with session.get(BASE_URL, params=params, timeout=30) as response: - response.raise_for_status() - results = response.json() - items = results.get("items", []) - except requests.RequestException as e: - LOGGER.error(f"Failed to fetch data without themes: {e}") - return [] - - LOGGER.info(f"Retrieved {len(items)} items without themes.") - - # --- Aggregate by DATA_PROVIDER + LEGAL_TOOL --- - aggregation = defaultdict(lambda: defaultdict(int)) - - for item in items: - data_providers = item.get("dataProvider", []) - data_provider = ( - data_providers - if isinstance(data_providers, str) - else ( - data_providers[0] - if isinstance(data_providers, list) and data_providers - else "Unknown" - ) - ) - - rights = item.get("rights", []) - legal_tool = ( - rights - if isinstance(rights, str) - else ( - rights[0] if isinstance(rights, list) and rights else "Unknown" - ) - ) - # Simplify license format if it’s a Creative Commons URL - if ( - legal_tool - and isinstance(legal_tool, str) - and legal_tool.startswith("http") - ): - parts = legal_tool.strip("/").split("/") - last_parts = parts[-2:] - if last_parts: - joined = " ".join(part.upper() for part in last_parts if part) - if "creativecommons.org" in legal_tool: - legal_tool = f"CC {joined}" - else: - legal_tool = joined - else: - legal_tool = "Unknown" +def fetch_europeana_data_without_themes(session): + """Fetch aggregated counts by DATA_PROVIDER and LEGAL_TOOL (no theme).""" + LOGGER.info( + "Fetching Europeana totalResults aggregated " + "by provider and rights (without themes)." + ) - aggregation[data_provider][legal_tool] += 1 + providers = get_facet_list(session, "DATA_PROVIDER") + rights_list = get_facet_list(session, "RIGHTS") - # Convert to flat list output = [] - for provider, licenses in aggregation.items(): - for legal_tool, count in licenses.items(): - output.append( - { - "DATA_PROVIDER": provider, - "LEGAL_TOOL": legal_tool, - "COUNT": count, - } - ) + for provider in providers: + for rights in rights_list: + params = { + "wskey": EUROPEANA_API_KEY, + "rows": 0, + "query": f'DATA_PROVIDER:"{provider}" AND RIGHTS:"{rights}"', + } + try: + resp = session.get(BASE_URL, params=params, timeout=30) + resp.raise_for_status() + count = resp.json().get("totalResults", 0) + + if count > 0: + simplified_rights = simplify_legal_tool(rights) + output.append( + { + "DATA_PROVIDER": provider, + "LEGAL_TOOL": simplified_rights, + "COUNT": count, + } + ) + except requests.RequestException as e: + LOGGER.warning( + f"Failed for provider={provider}, rights={rights}: {e}" + ) + time.sleep(0.5) - LOGGER.info(f"Aggregated data without themes into {len(output)} records.") + LOGGER.info(f"Aggregated {len(output)} records (without themes).") return output -def fetch_europeana_data_with_themes(args): - """Fetch and aggregate data by DATA_PROVIDER, LEGAL_TOOL, and THEME.""" - LOGGER.info("Fetching aggregated Europeana data with themes.") - - if not EUROPEANA_API_KEY: - raise shared.QuantifyingException( - "EUROPEANA_API_KEY not found in environment variables", 1 - ) +def fetch_europeana_data_with_themes(session): + """Fetch aggregated counts by DATA_PROVIDER, LEGAL_TOOL, and THEME.""" + LOGGER.info( + "Fetching Europeana totalResults " + "aggregated by provider, rights, and theme." + ) - # Themes from Europeana site + providers = get_facet_list(session, "DATA_PROVIDER") + rights_list = get_facet_list(session, "RIGHTS") themes = [ "art", "fashion", @@ -204,106 +183,46 @@ def fetch_europeana_data_with_themes(args): "archaeology", ] - items_per_query = max(20, args.limit // len(themes)) - all_items = [] - session = get_requests_session() - - for theme in themes: - params = { - "wskey": EUROPEANA_API_KEY, - "rows": min(items_per_query, 20), - "profile": "rich", - "query": "*", - "theme": theme, - } - - try: - LOGGER.info( - f"Fetching {params['rows']} records for theme: '{theme}'" - ) - with session.get(BASE_URL, params=params, timeout=30) as response: - response.raise_for_status() - results = response.json() - items = results.get("items", []) - for item in items: - item["theme_used"] = theme - all_items.extend(items) - LOGGER.info(f"Retrieved {len(items)} items for theme '{theme}'") - time.sleep(1) - except requests.RequestException as e: - LOGGER.warning(f"Failed to fetch data for theme '{theme}': {e}") - continue - - if not all_items: - LOGGER.error("No items retrieved for any theme.") - return [] - - LOGGER.info(f"Total items retrieved across all themes: {len(all_items)}") - - # Aggregate by DATA_PROVIDER + LEGAL_TOOL + THEME - aggregation = defaultdict(lambda: defaultdict(lambda: defaultdict(int))) - - for item in all_items: - data_providers = item.get("dataProvider", []) - data_provider = ( - data_providers - if isinstance(data_providers, str) - else ( - data_providers[0] - if isinstance(data_providers, list) and data_providers - else "Unknown" - ) - ) - - rights = item.get("rights", []) - legal_tool = ( - rights - if isinstance(rights, str) - else ( - rights[0] if isinstance(rights, list) and rights else "Unknown" - ) - ) - - if ( - legal_tool - and isinstance(legal_tool, str) - and legal_tool.startswith("http") - ): - parts = legal_tool.strip("/").split("/") - last_parts = parts[-2:] - if last_parts: - joined = " ".join(part.upper() for part in last_parts if part) - if "creativecommons.org" in legal_tool: - legal_tool = f"CC {joined}" - else: - legal_tool = joined - else: - legal_tool = "Unknown" - - theme = item.get("theme_used", "Unknown") - aggregation[data_provider][legal_tool][theme] += 1 - - # Convert to flat list output = [] - for provider, licenses in aggregation.items(): - for legal_tool, themes_dict in licenses.items(): - for theme, count in themes_dict.items(): - output.append( - { - "DATA_PROVIDER": provider, - "LEGAL_TOOL": legal_tool, - "THEME": theme, - "COUNT": count, - } - ) - - LOGGER.info(f"Aggregated data with themes into {len(output)} records.") + for provider in providers: + for rights in rights_list: + simplified_rights = simplify_legal_tool(rights) + for theme in themes: + params = { + "wskey": EUROPEANA_API_KEY, + "rows": 0, + "query": ( + f'DATA_PROVIDER:"{provider}" AND RIGHTS:"{rights}"' + ), + "theme": theme, + } + try: + resp = session.get(BASE_URL, params=params, timeout=30) + resp.raise_for_status() + count = resp.json().get("totalResults", 0) + if count > 0: + output.append( + { + "DATA_PROVIDER": provider, + "LEGAL_TOOL": simplified_rights, + "THEME": theme, + "COUNT": count, + } + ) + except requests.RequestException as e: + LOGGER.warning( + f"Failed for provider={provider}, rights={rights}, " + f"theme={theme}: {e}" + ) + time.sleep(0.5) + + LOGGER.info(f"Aggregated {len(output)} records (with themes).") return output def save_csv(filepath, header, data): - """Save data to a CSV file.""" - with open(filepath, "w", newline="") as f: + """Save aggregated data to CSV.""" + with open(filepath, "w", newline="", encoding="utf-8") as f: writer = csv.DictWriter(f, fieldnames=header) writer.writeheader() writer.writerows(data) @@ -314,20 +233,22 @@ def main(): args = parse_arguments() shared.paths_log(LOGGER, PATHS) shared.git_fetch_and_merge(args, PATHS["repo"]) - os.makedirs(PATHS["data_phase"], exist_ok=True) - # Fetch and save data WITHOUT themes (aggregated) - data_no_theme = fetch_europeana_data_without_themes(args) - if args.enable_save and data_no_theme: - save_csv(FILE_WITHOUT_THEMES, HEADER_WITHOUT_THEMES, data_no_theme) + session = get_requests_session() + + # Fetch data + data_no_theme = fetch_europeana_data_without_themes(session) + data_with_theme = fetch_europeana_data_with_themes(session) - # Fetch and save data WITH themes (aggregated) - data_with_theme = fetch_europeana_data_with_themes(args) - if args.enable_save and data_with_theme: - save_csv(FILE_WITH_THEMES, HEADER_WITH_THEMES, data_with_theme) + # Save if enabled + if args.enable_save: + if data_no_theme: + save_csv(FILE_WITHOUT_THEMES, HEADER_WITHOUT_THEMES, data_no_theme) + if data_with_theme: + save_csv(FILE_WITH_THEMES, HEADER_WITH_THEMES, data_with_theme) - # Git commit & push + # Git actions if args.enable_git and args.enable_save: args = shared.git_add_and_commit( args, @@ -337,19 +258,14 @@ def main(): ) shared.git_push_changes(args, PATHS["repo"]) - LOGGER.info( - "Europeana dual-fetch (with & without themes) completed successfully." - ) + LOGGER.info("Optimized Europeana dual-fetch completed successfully.") if __name__ == "__main__": try: main() except shared.QuantifyingException as e: - if e.exit_code == 0: - LOGGER.info(e.message) - else: - LOGGER.error(e.message) + LOGGER.error(e.message) sys.exit(e.exit_code) except SystemExit as e: if e.code != 0: From add3aaf0d54a884c36cab71caced57be21840b93 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Fri, 24 Oct 2025 11:35:27 +0300 Subject: [PATCH 09/30] Added updated files --- env.example | 3 + scripts/1-fetch/europeana_fetch.py | 238 ++++++++++++++++++++++------- sources.md | 81 ++++++++++ 3 files changed, 270 insertions(+), 52 deletions(-) diff --git a/env.example b/env.example index 6727e8da..8c66ebd3 100644 --- a/env.example +++ b/env.example @@ -33,3 +33,6 @@ # FLICKR_API_KEY = # FLICKR_API_SECRET = + +# "Europeana Search API Documentation: https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation#Request +# EUROPEANA_API_KEY = diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index 52af4475..0db68217 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -85,53 +85,150 @@ def get_requests_session(): def get_facet_list(session, facet_field): - """Fetch unique values for a facet (e.g., DATA_PROVIDER or RIGHTS).""" - params = { - "wskey": EUROPEANA_API_KEY, - "query": "*", - "rows": 0, - "facet": facet_field, - "profile": "facets", - } - resp = session.get(BASE_URL, params=params, timeout=30) - resp.raise_for_status() - data = resp.json() - facet_values = [ - f["label"] for f in data.get("facets", [])[0].get("fields", []) - ] - return facet_values + """Complete facet fetching using cursor pagination""" + all_values = [] + cursor = "*" + page = 1 + + print(f"\n=== Fetching {facet_field} values (cursor-based) ===") + + while cursor: + params = { + "wskey": EUROPEANA_API_KEY, + "query": "*", + "rows": 0, + "facet": facet_field, + "facet.limit": 100, + "cursor": cursor, + "profile": "facets", + } + + resp = session.get(BASE_URL, params=params, timeout=30) + data = resp.json() + + facets = data.get("facets", []) + if facets and facets[0].get("fields"): + fields = facets[0]["fields"] + new_count = 0 + + for field in fields: + if field.get("label") and field["label"] not in all_values: + all_values.append(field["label"]) + new_count += 1 + + print(f"P{page}: {len(fields)} total, +{new_count}") + + # Show sample of new values + if new_count > 0: + new_values = [ + field["label"] + for field in fields + if field.get("label") + and field["label"] not in all_values[:-new_count] + ] + sample = new_values[:3] + sample_display = ", ".join(sample) + if new_count > 3: + sample_display += f" ... and {new_count - 3} more" + print(f" New: {sample_display}") + else: + print(f"Page {page}: No fields returned") + + # Get next cursor or break + next_cursor = data.get("nextCursor") + if next_cursor == cursor or not next_cursor: + print(f"→ Reached end after {page} pages") + break + + cursor = next_cursor + page += 1 + time.sleep(0.5) + + print(f"✅ Completed: {len(all_values)} total unique {facet_field} values") + return all_values def simplify_legal_tool(legal_tool): - """Simplify and standardize Creative Commons or license URLs.""" - if ( - legal_tool - and isinstance(legal_tool, str) - and legal_tool.startswith("http") - ): + """ + Simplify and standardize license URLs (especially Creative Commons). + + This function converts long or complex license URLs into + short, human-readable labels like "CC BY-SA 4.0" or "CC BY-NC-ND 3.0 AT". + + It handles both: + - Non-ported Creative Commons licenses (no jurisdiction) + - Ported licenses (with jurisdiction codes, e.g., 'AT', 'DE', etc.) + - Other license URLs gracefully (returns simplified or original form) + + Examples + -------- + >>> simplify_legal_tool("http://creativecommons.org/licenses/by-sa/4.0/") + 'CC BY-SA 4.0' + + >>> simplify_legal_tool + >>> ("http://creativecommons.org/licenses/by-nc-nd/3.0/at/") + 'CC BY-NC-ND 3.0 AT' + + >>> simplify_legal_tool("http://rightsstatements.org/vocab/InC/1.0/") + 'VOCAB INC 1.0' + + >>> simplify_legal_tool("Public Domain") + 'Public Domain' + + Parameters + ---------- + legal_tool : str + A license string or URL, e.g., Creative Commons or RightsStatement. + + Returns + ------- + str + A short, standardized form of the license. + """ + if not (legal_tool and isinstance(legal_tool, str)): + return legal_tool + + if legal_tool.startswith("http"): parts = legal_tool.strip("/").split("/") - last_parts = parts[-2:] - if last_parts: - joined = " ".join(part.upper() for part in last_parts if part) - if "creativecommons.org" in legal_tool: - return f"CC {joined}" - else: - return joined + last_parts = parts[-3:] # allow for jurisdiction at the end + + # Detect jurisdiction (2-letter code or known suffix) + jurisdiction = "" + if len(last_parts[-1]) == 2 and last_parts[-1].isalpha(): + jurisdiction = last_parts[-1].upper() + last_parts = last_parts[:-1] + + # Join and format + joined = " ".join( + part.upper() for part in last_parts if part and part != "licenses" + ) + + if "creativecommons.org" in legal_tool: + return f"CC {joined} {jurisdiction}".strip() else: - return "Unknown" + return joined or "Unknown" + return legal_tool -def fetch_europeana_data_without_themes(session): - """Fetch aggregated counts by DATA_PROVIDER and LEGAL_TOOL (no theme).""" +def fetch_europeana_data_without_themes(session, providers, rights_list): + """Fetch aggregated counts by DATA_PROVIDER and LEGAL_TOOL (no theme) + + Parameters + ---------- + session : requests.Session + A configured requests session. + providers : list[str] + List of DATA_PROVIDER names. + rights_list : list[str] + List of license/rights strings. + + """ LOGGER.info( "Fetching Europeana totalResults aggregated " "by provider and rights (without themes)." ) - providers = get_facet_list(session, "DATA_PROVIDER") - rights_list = get_facet_list(session, "RIGHTS") - output = [] for provider in providers: for rights in rights_list: @@ -164,35 +261,42 @@ def fetch_europeana_data_without_themes(session): return output -def fetch_europeana_data_with_themes(session): - """Fetch aggregated counts by DATA_PROVIDER, LEGAL_TOOL, and THEME.""" +def fetch_europeana_data_with_themes(session, providers, rights_list, themes): + """ + Fetch aggregated counts by DATA_PROVIDER, LEGAL_TOOL, and THEME. + + Parameters + ---------- + session : requests.Session + A configured requests session. + providers : list[str] + List of DATA_PROVIDER names. + rights_list : list[str] + List of license/rights strings. + themes : list[str] + List of themes to query. + + Returns + ------- + list[dict] + Aggregated counts with keys: DATA_PROVIDER, LEGAL_TOOL, THEME, COUNT. + """ LOGGER.info( "Fetching Europeana totalResults " "aggregated by provider, rights, and theme." ) - providers = get_facet_list(session, "DATA_PROVIDER") - rights_list = get_facet_list(session, "RIGHTS") - themes = [ - "art", - "fashion", - "music", - "industrial", - "sport", - "photography", - "archaeology", - ] - output = [] + for provider in providers: for rights in rights_list: simplified_rights = simplify_legal_tool(rights) - for theme in themes: + for theme in themes: # use the themes passed from main() params = { "wskey": EUROPEANA_API_KEY, "rows": 0, "query": ( - f'DATA_PROVIDER:"{provider}" AND RIGHTS:"{rights}"' + f'DATA_PROVIDER:"{provider}" ' f'AND RIGHTS:"{rights}"' ), "theme": theme, } @@ -235,11 +339,41 @@ def main(): shared.git_fetch_and_merge(args, PATHS["repo"]) os.makedirs(PATHS["data_phase"], exist_ok=True) + # --- Environment check for Europeana API key --- + if not EUROPEANA_API_KEY: + raise shared.QuantifyingException( + "EUROPEANA_API_KEY not found in environment variables", 1 + ) + session = get_requests_session() + providers = get_facet_list(session, "DATA_PROVIDER") + rights_list = get_facet_list(session, "RIGHTS") + + # Define themes here (alphabetically for consistency) + themes = [ + "archaeology", + "art", + "fashion", + "industrial", + "manuscript", + "maps", + "migration", + "music", + "nature", + "newspaper", + "photography", + "sport", + "ww1", + ] + # Fetch data - data_no_theme = fetch_europeana_data_without_themes(session) - data_with_theme = fetch_europeana_data_with_themes(session) + data_no_theme = fetch_europeana_data_without_themes( + session, providers, rights_list + ) + data_with_theme = fetch_europeana_data_with_themes( + session, providers, rights_list, themes + ) # Save if enabled if args.enable_save: diff --git a/sources.md b/sources.md index a4119ee5..37c26fa5 100644 --- a/sources.md +++ b/sources.md @@ -102,3 +102,84 @@ language edition of wikipedia. It runs on the Meta-Wiki API. - No API key required - Query limit: It is rate-limited only to prevent abuse - Data available through XML or JSON format +- No query limits + +[ia-search]: https://internetarchive.readthedocs.io/en/stable/internetarchive.html#internetarchive.Search + + +## MediaWiki Action API + +**Description:** _The MediaWiki Action API is a web service that allows access +to some wiki features like authentication, page operations, and search. It can +provide meta information about the wiki and the logged-in user._ ([API:Main +page - MediaWiki](https://www.mediawiki.org/wiki/API:Main_page)) + +**API documentation link:** +- [MediaWiki Action API](https://www.mediawiki.org/wiki/API:Main_page) + +**API information:** + - No API key required + - Query limit: depends on user status and request type + - Data available through XML or JSON format + + +## The Metropolitan Museum of Art Collection API + +**Description:** _The Met’s Open Access datasets are available through our API. +The API (RESTful web service in JSON format) gives access to all of The Met’s +Open Access data and to corresponding high resolution images (JPEG format) that +are in the public domain._ ([The Metropolitan Museum of Art Collection +API](https://metmuseum.github.io/)) + +**API documentation link:** +- [Latest Updates | The Metropolitan Museum of Art Collection + API](https://metmuseum.github.io/) + +**API information:** + - No API key required + - 80 queries per second + + +## Vimeo API + +**Description:** The Vimeo API allows users to perform filtered, advanced +search on Vimeo videos. + +**API documentation link:** +- [Getting Started with the Vimeo API](https://developer.vimeo.com/api/start) + +**API information:** + - API key required + - Query limit: 5000 authenticated requests per day + - Data available through JSON format + + +## YouTube Data API + +**Description:** An API from YouTube for platform users to upload videos, +adjust video parameters, and obtain search results. + +**API documentation link:** +- [Search: list | YouTube Data API | Google + Developers](https://developers.google.com/youtube/v3/docs/search/list) + +**API information:** + - API key required + - Query limit: depends on the type and number of requests + - Data available through JSON format + + +## EUROPEANA DATA API + +**Description:** The Europeana Search API provides access to digital cultural heritage records from museums, libraries, and archives across Europe + +**API Documentation link:** +- [Search API](https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation) + +**API information:** + - API key required + - Query parameters allow: + - Full-text searching (query) + - Retrieving only metadata facets (e.g. via profile=facets) + - Data returned in JSON format + - Uses facet pagination From 57cc578c4983b42c0b93e12f5cdf5b3dd8ff4ed0 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Fri, 24 Oct 2025 14:24:59 +0300 Subject: [PATCH 10/30] Removed print statements --- scripts/1-fetch/europeana_fetch.py | 38 ++++++++++++++---------------- 1 file changed, 18 insertions(+), 20 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index 0db68217..4eece92a 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -85,12 +85,12 @@ def get_requests_session(): def get_facet_list(session, facet_field): - """Complete facet fetching using cursor pagination""" + """Complete facet fetching using cursor pagination.""" all_values = [] cursor = "*" page = 1 - print(f"\n=== Fetching {facet_field} values (cursor-based) ===") + LOGGER.info(f"Fetching {facet_field} values using cursor pagination.") while cursor: params = { @@ -110,41 +110,39 @@ def get_facet_list(session, facet_field): if facets and facets[0].get("fields"): fields = facets[0]["fields"] new_count = 0 + new_values = [] for field in fields: - if field.get("label") and field["label"] not in all_values: - all_values.append(field["label"]) + label = field.get("label") + if label and label not in all_values: + all_values.append(label) + new_values.append(label) new_count += 1 - print(f"P{page}: {len(fields)} total, +{new_count}") + LOGGER.debug( + f"Page {page}: {len(fields)} total, +{new_count} new." + ) - # Show sample of new values - if new_count > 0: - new_values = [ - field["label"] - for field in fields - if field.get("label") - and field["label"] not in all_values[:-new_count] - ] + if new_values: sample = new_values[:3] - sample_display = ", ".join(sample) if new_count > 3: - sample_display += f" ... and {new_count - 3} more" - print(f" New: {sample_display}") + sample.append(f"... +{new_count - 3} more") + LOGGER.debug(f"New sample values: {', '.join(sample)}") else: - print(f"Page {page}: No fields returned") + LOGGER.debug(f"Page {page}: No fields returned.") - # Get next cursor or break next_cursor = data.get("nextCursor") if next_cursor == cursor or not next_cursor: - print(f"→ Reached end after {page} pages") + LOGGER.info( + f"Cursor exhausted after {page} pages. " + f"Collected {len(all_values)} unique {facet_field} values." + ) break cursor = next_cursor page += 1 time.sleep(0.5) - print(f"✅ Completed: {len(all_values)} total unique {facet_field} values") return all_values From 000addca62eb17799b495df81691248442a30b4c Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Fri, 24 Oct 2025 15:20:26 +0300 Subject: [PATCH 11/30] Used facet pagination instead of cursor --- scripts/1-fetch/europeana_fetch.py | 58 +++++++++++++----------------- 1 file changed, 25 insertions(+), 33 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index 4eece92a..faf5327f 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -85,21 +85,22 @@ def get_requests_session(): def get_facet_list(session, facet_field): - """Complete facet fetching using cursor pagination.""" + """Fetch complete facet list using offset-based pagination.""" all_values = [] - cursor = "*" + offset = 0 + limit = 100 page = 1 - LOGGER.info(f"Fetching {facet_field} values using cursor pagination.") + LOGGER.info(f"Fetching {facet_field} values with offset pagination.") - while cursor: + while True: params = { "wskey": EUROPEANA_API_KEY, "query": "*", "rows": 0, "facet": facet_field, - "facet.limit": 100, - "cursor": cursor, + f"f.{facet_field}.facet.limit": limit, + f"f.{facet_field}.facet.offset": offset, "profile": "facets", } @@ -107,39 +108,30 @@ def get_facet_list(session, facet_field): data = resp.json() facets = data.get("facets", []) - if facets and facets[0].get("fields"): - fields = facets[0]["fields"] - new_count = 0 - new_values = [] - - for field in fields: - label = field.get("label") - if label and label not in all_values: - all_values.append(label) - new_values.append(label) - new_count += 1 - - LOGGER.debug( - f"Page {page}: {len(fields)} total, +{new_count} new." - ) + if not facets or not facets[0].get("fields"): + LOGGER.info(f"No more values after page {page}.") + break - if new_values: - sample = new_values[:3] - if new_count > 3: - sample.append(f"... +{new_count - 3} more") - LOGGER.debug(f"New sample values: {', '.join(sample)}") - else: - LOGGER.debug(f"Page {page}: No fields returned.") + fields = facets[0]["fields"] + new_values = [f["label"] for f in fields if f.get("label")] + + for v in new_values: + if v not in all_values: + all_values.append(v) + + LOGGER.debug( + f"Page {page}: Received {len(new_values)} facet values. " + f"Total so far: {len(all_values)}" + ) - next_cursor = data.get("nextCursor") - if next_cursor == cursor or not next_cursor: + if len(new_values) < limit: LOGGER.info( - f"Cursor exhausted after {page} pages. " - f"Collected {len(all_values)} unique {facet_field} values." + f"Completed fetching {facet_field}. " + f"Total unique: {len(all_values)}" ) break - cursor = next_cursor + offset += limit page += 1 time.sleep(0.5) From 71c8b58cf5b325f3243240debac3bb3c8bbf2e3e Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Fri, 24 Oct 2025 15:27:32 +0300 Subject: [PATCH 12/30] Uses offset based pagination and updates sources --- scripts/1-fetch/europeana_fetch.py | 2 +- sources.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index faf5327f..669b5d92 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -119,7 +119,7 @@ def get_facet_list(session, facet_field): if v not in all_values: all_values.append(v) - LOGGER.debug( + LOGGER.info( f"Page {page}: Received {len(new_values)} facet values. " f"Total so far: {len(all_values)}" ) diff --git a/sources.md b/sources.md index 37c26fa5..79352fb3 100644 --- a/sources.md +++ b/sources.md @@ -182,4 +182,4 @@ adjust video parameters, and obtain search results. - Full-text searching (query) - Retrieving only metadata facets (e.g. via profile=facets) - Data returned in JSON format - - Uses facet pagination + - Uses offset based pagination From 87c3c1ed85e584e496b47280a76c07b8bee70a5a Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Fri, 24 Oct 2025 15:46:31 +0300 Subject: [PATCH 13/30] Removed unnecessary logger info --- scripts/1-fetch/europeana_fetch.py | 20 ++++++-------------- 1 file changed, 6 insertions(+), 14 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index 669b5d92..6ff432d3 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -85,13 +85,12 @@ def get_requests_session(): def get_facet_list(session, facet_field): - """Fetch complete facet list using offset-based pagination.""" + """Fetch complete facet list""" all_values = [] offset = 0 limit = 100 - page = 1 - LOGGER.info(f"Fetching {facet_field} values with offset pagination.") + LOGGER.info(f"Fetching {facet_field} facet values.") while True: params = { @@ -109,7 +108,6 @@ def get_facet_list(session, facet_field): facets = data.get("facets", []) if not facets or not facets[0].get("fields"): - LOGGER.info(f"No more values after page {page}.") break fields = facets[0]["fields"] @@ -119,22 +117,16 @@ def get_facet_list(session, facet_field): if v not in all_values: all_values.append(v) - LOGGER.info( - f"Page {page}: Received {len(new_values)} facet values. " - f"Total so far: {len(all_values)}" - ) - if len(new_values) < limit: - LOGGER.info( - f"Completed fetching {facet_field}. " - f"Total unique: {len(all_values)}" - ) break offset += limit - page += 1 time.sleep(0.5) + LOGGER.info( + f"Completed fetching {facet_field}. Total unique: {len(all_values)}" + ) + return all_values From c5a3abd5adedf7b031783f2028762cc20f4b26c2 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Tue, 28 Oct 2025 11:41:54 +0300 Subject: [PATCH 14/30] Add limit parser to allow restricting providers during testing --- scripts/1-fetch/europeana_fetch.py | 438 +++++++++++++++++------------ 1 file changed, 260 insertions(+), 178 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index 6ff432d3..c0393c80 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -45,11 +45,30 @@ HEADER_WITH_THEMES = ["DATA_PROVIDER", "LEGAL_TOOL", "THEME", "COUNT"] HEADER_WITHOUT_THEMES = ["DATA_PROVIDER", "LEGAL_TOOL", "COUNT"] QUARTER = os.path.basename(PATHS["data_quarter"]) - -# Log start -LOGGER.info( - "Optimized Europeana dual-fetch (using totalResults) script started." -) +THEMES = [ + "archaeology", + "art", + "fashion", + "industrial", + "manuscript", + "map", + "migration", + "music", + "nature", + "newspaper", + "photography", + "sport", + "ww1", +] + +RIGHTS_LABEL_MAP = { + "InC": "In Copyright", + "InC-EDU": "In Copyright – Educational Use Only", + "InC-OW-EU": "In Copyright – EU Only", + "CNE": "No Copyright – Contractual Restrictions", + "NoC-NC": "No Copyright – Non-Commercial Use Only", + "NoC-OKLR": "No Copyright – Other Known Legal Restrictions", +} def parse_arguments(): @@ -58,12 +77,18 @@ def parse_arguments(): parser.add_argument( "--enable-save", action="store_true", - help="Enable saving aggregated results to CSV.", + help="Enable saving results", ) parser.add_argument( "--enable-git", action="store_true", - help="Enable git actions (fetch, merge, add, commit, push).", + help="Enable git actions (fetch, merge, add, commit, push)", + ) + parser.add_argument( + "--limit", + type=int, + default=5, + help="Limit number of providers for testing.", ) args = parser.parse_args() if not args.enable_save and args.enable_git: @@ -84,11 +109,100 @@ def get_requests_session(): return session +def simplify_legal_tool(legal_tool): + """Simplify license URLs into human-readable labels + + + This function converts long or complex license URLs into + short, human-readable labels such as: + - "CC BY-SA 4.0" + - "CC BY-NC-ND 3.0 AT" + - "Public Domain (CC0 1.0)" + - "Public Domain (PDM 1.0)" + - "Rights Statement: In copyright" + + It handles: + - Non-ported Creative Commons licenses (no jurisdiction) + - Ported Creative Commons licenses (with jurisdiction codes) + - Public Domain identifiers (CC0, PDM) + - RightsStatements.org URLs + - Other license URLs (returns simplified or original form) + + Examples + -------- + simplify_legal_tool("http://creativecommons.org/licenses/by-sa/4.0/") + 'CC BY-SA 4.0' + + simplify_legal_tool("http://creativecommons.org/licenses/by-nc-nd/3.0/at/") + 'CC BY-NC-ND 3.0 AT' + + simplify_legal_tool("http://rightsstatements.org/vocab/InC/1.0/") + 'Rights Statement: In copyright' + + simplify_legal_tool("http://creativecommons.org/publicdomain/zero/1.0/") + 'Public Domain (CC0 1.0)' + + simplify_legal_tool("http://creativecommons.org/publicdomain/mark/1.0/") + 'Public Domain (PDM 1.0)' + + simplify_legal_tool("Public Domain") + 'Public Domain' + + Parameters + ---------- + legal_tool : str + A license string or URL, e.g., a Creative Commons license, + a RightsStatement.org URL, + or a Public Domain identifier. + + Returns + ------- + str + A short, standardized form of the license. + """ + + if not isinstance(legal_tool, str): + return legal_tool + if not legal_tool.startswith("http"): + return legal_tool + + # Public domain handling + if "publicdomain" in legal_tool: + if "zero" in legal_tool.lower(): + return "Public Domain (CC0 1.0)" + if "mark" in legal_tool.lower(): + return "Public Domain (PDM 1.0)" + return "Public Domain" + + # RightsStatements.org handling + if "rightsstatements.org" in legal_tool: + parts = legal_tool.strip("/").split("/") + code = parts[-2] + if code in RIGHTS_LABEL_MAP: + return f"Rights Statement: {RIGHTS_LABEL_MAP[code]}" + return f"Rights Statement: {code}" + + # Creative Commons handling + if "creativecommons.org" in legal_tool: + parts = legal_tool.strip("/").split("/") + last_parts = parts[-3:] + jurisdiction = "" + if len(last_parts[-1]) == 2 and last_parts[-1].isalpha(): + jurisdiction = last_parts[-1].upper() + last_parts = last_parts[:-1] + joined = " ".join( + part.upper() for part in last_parts if part and part != "licenses" + ) + return f"CC {joined} {jurisdiction}".strip() + + return legal_tool + + def get_facet_list(session, facet_field): - """Fetch complete facet list""" + """Fetch complete facet list from Europeana API for a given facet field.""" all_values = [] offset = 0 - limit = 100 + limit = 1000 LOGGER.info(f"Fetching {facet_field} facet values.") @@ -103,8 +217,15 @@ def get_facet_list(session, facet_field): "profile": "facets", } - resp = session.get(BASE_URL, params=params, timeout=30) - data = resp.json() + try: + resp = session.get(BASE_URL, params=params, timeout=30) + resp.raise_for_status() + data = resp.json() + except requests.RequestException as e: + LOGGER.warning( + f"Failed fetching facet {facet_field} at offset {offset}: {e}" + ) + break facets = data.get("facets", []) if not facets or not facets[0].get("fields"): @@ -126,166 +247,130 @@ def get_facet_list(session, facet_field): LOGGER.info( f"Completed fetching {facet_field}. Total unique: {len(all_values)}" ) - + all_values.sort() return all_values -def simplify_legal_tool(legal_tool): - """ - Simplify and standardize license URLs (especially Creative Commons). - - This function converts long or complex license URLs into - short, human-readable labels like "CC BY-SA 4.0" or "CC BY-NC-ND 3.0 AT". - - It handles both: - - Non-ported Creative Commons licenses (no jurisdiction) - - Ported licenses (with jurisdiction codes, e.g., 'AT', 'DE', etc.) - - Other license URLs gracefully (returns simplified or original form) - - Examples - -------- - >>> simplify_legal_tool("http://creativecommons.org/licenses/by-sa/4.0/") - 'CC BY-SA 4.0' - - >>> simplify_legal_tool - >>> ("http://creativecommons.org/licenses/by-nc-nd/3.0/at/") - 'CC BY-NC-ND 3.0 AT' +def fetch_europeana_data_without_themes(session, limit=None): + """Fetch counts by DATA_PROVIDER and RIGHTS using facets.""" + LOGGER.info("Fetching Europeana counts without themes.") - >>> simplify_legal_tool("http://rightsstatements.org/vocab/InC/1.0/") - 'VOCAB INC 1.0' + params = { + "wskey": EUROPEANA_API_KEY, + "query": "*", + "rows": 0, + "profile": "facets", + "facet": ["DATA_PROVIDER", "RIGHTS"], + "f.DATA_PROVIDER.facet.limit": 1000, + "f.RIGHTS.facet.limit": 100, + } - >>> simplify_legal_tool("Public Domain") - 'Public Domain' - - Parameters - ---------- - legal_tool : str - A license string or URL, e.g., Creative Commons or RightsStatement. - - Returns - ------- - str - A short, standardized form of the license. - """ - if not (legal_tool and isinstance(legal_tool, str)): - return legal_tool - - if legal_tool.startswith("http"): - parts = legal_tool.strip("/").split("/") - last_parts = parts[-3:] # allow for jurisdiction at the end - - # Detect jurisdiction (2-letter code or known suffix) - jurisdiction = "" - if len(last_parts[-1]) == 2 and last_parts[-1].isalpha(): - jurisdiction = last_parts[-1].upper() - last_parts = last_parts[:-1] - - # Join and format - joined = " ".join( - part.upper() for part in last_parts if part and part != "licenses" - ) - - if "creativecommons.org" in legal_tool: - return f"CC {joined} {jurisdiction}".strip() - else: - return joined or "Unknown" - - return legal_tool - - -def fetch_europeana_data_without_themes(session, providers, rights_list): - """Fetch aggregated counts by DATA_PROVIDER and LEGAL_TOOL (no theme) + try: + resp = session.get(BASE_URL, params=params, timeout=30) + resp.raise_for_status() + data = resp.json() + except requests.RequestException as e: + LOGGER.error(f"Failed to fetch facets: {e}") + return [] - Parameters - ---------- - session : requests.Session - A configured requests session. - providers : list[str] - List of DATA_PROVIDER names. - rights_list : list[str] - List of license/rights strings. - - """ - LOGGER.info( - "Fetching Europeana totalResults aggregated " - "by provider and rights (without themes)." - ) + facets = {f["name"]: f["fields"] for f in data.get("facets", [])} + provider_fields = facets.get("DATA_PROVIDER", []) + rights_fields = facets.get("RIGHTS", []) + if limit: + provider_fields = provider_fields[:limit] output = [] - for provider in providers: - for rights in rights_list: - params = { + for provider_entry in provider_fields: + provider = provider_entry["label"] + provider_count = provider_entry["count"] + if provider_count == 0: + continue + + for rights_entry in rights_fields: + rights = rights_entry["label"] + query = f'DATA_PROVIDER:"{provider}" AND RIGHTS:"{rights}"' + params_detail = { "wskey": EUROPEANA_API_KEY, "rows": 0, - "query": f'DATA_PROVIDER:"{provider}" AND RIGHTS:"{rights}"', + "query": query, } try: - resp = session.get(BASE_URL, params=params, timeout=30) - resp.raise_for_status() - count = resp.json().get("totalResults", 0) - + resp_detail = session.get( + BASE_URL, params=params_detail, timeout=60 + ) + resp_detail.raise_for_status() + count = resp_detail.json().get("totalResults", 0) if count > 0: - simplified_rights = simplify_legal_tool(rights) output.append( { "DATA_PROVIDER": provider, - "LEGAL_TOOL": simplified_rights, + "LEGAL_TOOL": simplify_legal_tool(rights), "COUNT": count, } ) + except requests.RequestException as e: LOGGER.warning( f"Failed for provider={provider}, rights={rights}: {e}" ) - time.sleep(0.5) - + time.sleep(0.2) LOGGER.info(f"Aggregated {len(output)} records (without themes).") return output -def fetch_europeana_data_with_themes(session, providers, rights_list, themes): - """ - Fetch aggregated counts by DATA_PROVIDER, LEGAL_TOOL, and THEME. +def fetch_europeana_data_with_themes(session, themes, limit=None): + """Fetch counts by DATA_PROVIDER, RIGHTS, and THEME using facets.""" + LOGGER.info("Fetching Europeana counts with themes") - Parameters - ---------- - session : requests.Session - A configured requests session. - providers : list[str] - List of DATA_PROVIDER names. - rights_list : list[str] - List of license/rights strings. - themes : list[str] - List of themes to query. + params = { + "wskey": EUROPEANA_API_KEY, + "query": "*", + "rows": 0, + "profile": "facets", + "facet": ["DATA_PROVIDER", "RIGHTS"], + "f.DATA_PROVIDER.facet.limit": 1000, + "f.RIGHTS.facet.limit": 100, + } - Returns - ------- - list[dict] - Aggregated counts with keys: DATA_PROVIDER, LEGAL_TOOL, THEME, COUNT. - """ - LOGGER.info( - "Fetching Europeana totalResults " - "aggregated by provider, rights, and theme." - ) + try: + resp = session.get(BASE_URL, params=params, timeout=60) + resp.raise_for_status() + data = resp.json() + except requests.RequestException as e: + LOGGER.error(f"Failed to fetch facets: {e}") + return [] - output = [] + facets = {f["name"]: f["fields"] for f in data.get("facets", [])} + provider_fields = facets.get("DATA_PROVIDER", []) + rights_fields = facets.get("RIGHTS", []) + if limit: + provider_fields = provider_fields[:limit] - for provider in providers: - for rights in rights_list: + output = [] + for provider_entry in provider_fields: + provider = provider_entry["label"] + provider_count = provider_entry["count"] + if provider_count == 0: + continue + + for rights_entry in rights_fields: + rights = rights_entry["label"] simplified_rights = simplify_legal_tool(rights) - for theme in themes: # use the themes passed from main() - params = { + + for theme in themes: + query = f'DATA_PROVIDER:"{provider}" AND RIGHTS:"{rights}"' + params_detail = { "wskey": EUROPEANA_API_KEY, "rows": 0, - "query": ( - f'DATA_PROVIDER:"{provider}" ' f'AND RIGHTS:"{rights}"' - ), + "query": query, "theme": theme, } try: - resp = session.get(BASE_URL, params=params, timeout=30) - resp.raise_for_status() - count = resp.json().get("totalResults", 0) + resp_detail = session.get( + BASE_URL, params=params_detail, timeout=30 + ) + resp_detail.raise_for_status() + count = resp_detail.json().get("totalResults", 0) if count > 0: output.append( { @@ -295,77 +380,74 @@ def fetch_europeana_data_with_themes(session, providers, rights_list, themes): "COUNT": count, } ) + except requests.RequestException as e: LOGGER.warning( - f"Failed for provider={provider}, rights={rights}, " - f"theme={theme}: {e}" + f"Failed for provider={provider}, " + f"rights={rights}, " + f"theme={theme}: " + f"{e}" ) time.sleep(0.5) - LOGGER.info(f"Aggregated {len(output)} records (with themes).") return output -def save_csv(filepath, header, data): - """Save aggregated data to CSV.""" - with open(filepath, "w", newline="", encoding="utf-8") as f: - writer = csv.DictWriter(f, fieldnames=header) - writer.writeheader() - writer.writerows(data) - LOGGER.info(f"Saved {len(data)} rows to {filepath}.") +def write_data(args, data_no_theme, data_with_theme): + """Write Europeana data to CSV files.""" + if not args.enable_save: + return args + + os.makedirs(PATHS["data_phase"], exist_ok=True) + + if data_no_theme: + with open(FILE_WITHOUT_THEMES, "w", newline="") as f: + writer = csv.DictWriter( + f, fieldnames=HEADER_WITHOUT_THEMES, dialect="unix" + ) + writer.writeheader() + writer.writerows(data_no_theme) + LOGGER.info( + f"Saved {len(data_no_theme)} rows to {FILE_WITHOUT_THEMES}." + ) + + if data_with_theme: + with open(FILE_WITH_THEMES, "w", newline="") as f: + writer = csv.DictWriter( + f, fieldnames=HEADER_WITH_THEMES, dialect="unix" + ) + writer.writeheader() + writer.writerows(data_with_theme) + LOGGER.info( + f"Saved {len(data_with_theme)} rows to {FILE_WITH_THEMES}." + ) + + return args def main(): args = parse_arguments() + LOGGER.info("Beginning fetch from Europeana") shared.paths_log(LOGGER, PATHS) shared.git_fetch_and_merge(args, PATHS["repo"]) os.makedirs(PATHS["data_phase"], exist_ok=True) - # --- Environment check for Europeana API key --- if not EUROPEANA_API_KEY: raise shared.QuantifyingException( "EUROPEANA_API_KEY not found in environment variables", 1 ) session = get_requests_session() - - providers = get_facet_list(session, "DATA_PROVIDER") - rights_list = get_facet_list(session, "RIGHTS") - - # Define themes here (alphabetically for consistency) - themes = [ - "archaeology", - "art", - "fashion", - "industrial", - "manuscript", - "maps", - "migration", - "music", - "nature", - "newspaper", - "photography", - "sport", - "ww1", - ] - - # Fetch data data_no_theme = fetch_europeana_data_without_themes( - session, providers, rights_list + session, limit=args.limit ) data_with_theme = fetch_europeana_data_with_themes( - session, providers, rights_list, themes + session, THEMES, limit=args.limit ) - # Save if enabled - if args.enable_save: - if data_no_theme: - save_csv(FILE_WITHOUT_THEMES, HEADER_WITHOUT_THEMES, data_no_theme) - if data_with_theme: - save_csv(FILE_WITH_THEMES, HEADER_WITH_THEMES, data_with_theme) + args = write_data(args, data_no_theme, data_with_theme) - # Git actions - if args.enable_git and args.enable_save: + if args.enable_git: args = shared.git_add_and_commit( args, PATHS["repo"], @@ -374,7 +456,7 @@ def main(): ) shared.git_push_changes(args, PATHS["repo"]) - LOGGER.info("Optimized Europeana dual-fetch completed successfully.") + LOGGER.info("Europeana fetch completed successfully.") if __name__ == "__main__": From 49e85f20af88fd88ac00ab27ad9f5e38e7b1e1ba Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Tue, 28 Oct 2025 11:55:47 +0300 Subject: [PATCH 15/30] Added comments to theme --- scripts/1-fetch/europeana_fetch.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index c0393c80..a891a801 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -45,6 +45,11 @@ HEADER_WITH_THEMES = ["DATA_PROVIDER", "LEGAL_TOOL", "THEME", "COUNT"] HEADER_WITHOUT_THEMES = ["DATA_PROVIDER", "LEGAL_TOOL", "COUNT"] QUARTER = os.path.basename(PATHS["data_quarter"]) +# Define themes here (alphabetically for consistency) +# Themes are listed at +# https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation#Request +# (in the Search API Request Parameter accordion) + THEMES = [ "archaeology", "art", From 008ec814b3deb56b299e13333d887f686dda264a Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Tue, 28 Oct 2025 12:12:07 +0300 Subject: [PATCH 16/30] Add additional logging for provider-level rights fetching --- scripts/1-fetch/europeana_fetch.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index a891a801..e1877c91 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -290,7 +290,7 @@ def fetch_europeana_data_without_themes(session, limit=None): provider_count = provider_entry["count"] if provider_count == 0: continue - + LOGGER.info(f"Fetching rights data for provider={provider}") for rights_entry in rights_fields: rights = rights_entry["label"] query = f'DATA_PROVIDER:"{provider}" AND RIGHTS:"{rights}"' @@ -357,7 +357,7 @@ def fetch_europeana_data_with_themes(session, themes, limit=None): provider_count = provider_entry["count"] if provider_count == 0: continue - + LOGGER.info(f"Fetching theme+rights data for provider={provider}") for rights_entry in rights_fields: rights = rights_entry["label"] simplified_rights = simplify_legal_tool(rights) From d597e6ae7d915beed38e6d09561e2fcc1bd2103e Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Tue, 28 Oct 2025 19:35:28 +0300 Subject: [PATCH 17/30] Reduced time to sleep --- scripts/1-fetch/europeana_fetch.py | 25 ++++++++++++++----------- 1 file changed, 14 insertions(+), 11 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index e1877c91..80e28b4e 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -318,7 +318,7 @@ def fetch_europeana_data_without_themes(session, limit=None): LOGGER.warning( f"Failed for provider={provider}, rights={rights}: {e}" ) - time.sleep(0.2) + time.sleep(0.01) LOGGER.info(f"Aggregated {len(output)} records (without themes).") return output @@ -393,7 +393,7 @@ def fetch_europeana_data_with_themes(session, themes, limit=None): f"theme={theme}: " f"{e}" ) - time.sleep(0.5) + time.sleep(0.01) LOGGER.info(f"Aggregated {len(output)} records (with themes).") return output @@ -443,6 +443,11 @@ def main(): ) session = get_requests_session() + + providers_full = get_facet_list(session, "DATA_PROVIDER") + rights_full = get_facet_list(session, "RIGHTS") + LOGGER.info(f"Facet providers loaded: {len(providers_full)}") + LOGGER.info(f"Facet rights loaded: {len(rights_full)}") data_no_theme = fetch_europeana_data_without_themes( session, limit=args.limit ) @@ -451,15 +456,13 @@ def main(): ) args = write_data(args, data_no_theme, data_with_theme) - - if args.enable_git: - args = shared.git_add_and_commit( - args, - PATHS["repo"], - PATHS["data_quarter"], - f"Add Europeana files (with and without themes) for {QUARTER}", - ) - shared.git_push_changes(args, PATHS["repo"]) + args = shared.git_add_and_commit( + args, + PATHS["repo"], + PATHS["data_quarter"], + f"Add Europeana files (with and without themes) for {QUARTER}", + ) + shared.git_push_changes(args, PATHS["repo"]) LOGGER.info("Europeana fetch completed successfully.") From fc29c492a01fa368109cbabfc5fa96ec9f08e4c3 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Wed, 29 Oct 2025 13:01:01 +0300 Subject: [PATCH 18/30] Added a timeout constant --- scripts/1-fetch/europeana_fetch.py | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index 80e28b4e..d56f096a 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -45,6 +45,7 @@ HEADER_WITH_THEMES = ["DATA_PROVIDER", "LEGAL_TOOL", "THEME", "COUNT"] HEADER_WITHOUT_THEMES = ["DATA_PROVIDER", "LEGAL_TOOL", "COUNT"] QUARTER = os.path.basename(PATHS["data_quarter"]) +TIMEOUT = 25 # Define themes here (alphabetically for consistency) # Themes are listed at # https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation#Request @@ -223,7 +224,7 @@ def get_facet_list(session, facet_field): } try: - resp = session.get(BASE_URL, params=params, timeout=30) + resp = session.get(BASE_URL, params=params, timeout=TIMEOUT) resp.raise_for_status() data = resp.json() except requests.RequestException as e: @@ -271,7 +272,7 @@ def fetch_europeana_data_without_themes(session, limit=None): } try: - resp = session.get(BASE_URL, params=params, timeout=30) + resp = session.get(BASE_URL, params=params, timeout=TIMEOUT) resp.raise_for_status() data = resp.json() except requests.RequestException as e: @@ -301,7 +302,7 @@ def fetch_europeana_data_without_themes(session, limit=None): } try: resp_detail = session.get( - BASE_URL, params=params_detail, timeout=60 + BASE_URL, params=params_detail, timeout=TIMEOUT ) resp_detail.raise_for_status() count = resp_detail.json().get("totalResults", 0) @@ -338,7 +339,7 @@ def fetch_europeana_data_with_themes(session, themes, limit=None): } try: - resp = session.get(BASE_URL, params=params, timeout=60) + resp = session.get(BASE_URL, params=params, timeout=TIMEOUT) resp.raise_for_status() data = resp.json() except requests.RequestException as e: @@ -372,7 +373,7 @@ def fetch_europeana_data_with_themes(session, themes, limit=None): } try: resp_detail = session.get( - BASE_URL, params=params_detail, timeout=30 + BASE_URL, params=params_detail, timeout=TIMEOUT ) resp_detail.raise_for_status() count = resp_detail.json().get("totalResults", 0) From 2c9e71fa1459bc83dfbd4744e2566c2a5faec19f Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Thu, 30 Oct 2025 12:32:00 +0300 Subject: [PATCH 19/30] Updated sources.md --- sources.md | 21 ++++++++++++--------- 1 file changed, 12 insertions(+), 9 deletions(-) diff --git a/sources.md b/sources.md index 79352fb3..76fbf26b 100644 --- a/sources.md +++ b/sources.md @@ -171,15 +171,18 @@ adjust video parameters, and obtain search results. ## EUROPEANA DATA API -**Description:** The Europeana Search API provides access to digital cultural heritage records from museums, libraries, and archives across Europe +**Description:** +The **Europeana Search API** provides access to digital cultural heritage metadata records aggregated from museums, libraries, and archives across Europe. +This project uses the API to fetch aggregated counts of cultural heritage records by data provider, rights statement, and theme. -**API Documentation link:** -- [Search API](https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation) +**Official API Documentation:** +- [Search API Documentation](https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation) **API information:** - - API key required - - Query parameters allow: - - Full-text searching (query) - - Retrieving only metadata facets (e.g. via profile=facets) - - Data returned in JSON format - - Uses offset based pagination +- API key required +- Query parameters allow: + - Full-text searching (`query`) + - Retrieving metadata facets (`profile=facets`) + - Filtering by data provider, rights statement, and theme +- Data available through JSON format +- Offset-based pagination From 87bfd9d270caffafa11fec50474e740c842a8913 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Thu, 30 Oct 2025 12:55:48 +0300 Subject: [PATCH 20/30] Resolve merge conflict in sources.md and keep Europeana Data API section --- sources.md | 1 - 1 file changed, 1 deletion(-) diff --git a/sources.md b/sources.md index 76fbf26b..f0794f90 100644 --- a/sources.md +++ b/sources.md @@ -168,7 +168,6 @@ adjust video parameters, and obtain search results. - Query limit: depends on the type and number of requests - Data available through JSON format - ## EUROPEANA DATA API **Description:** From 1d4ae389bc0e30a516921bb7d979b198f9b927fa Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Thu, 30 Oct 2025 13:01:17 +0300 Subject: [PATCH 21/30] Add Europeana Data API section only --- sources.md | 68 +++++++++++++++++++++++++++++------------------------- 1 file changed, 37 insertions(+), 31 deletions(-) diff --git a/sources.md b/sources.md index f0794f90..3fce919d 100644 --- a/sources.md +++ b/sources.md @@ -23,6 +23,36 @@ tool paths. [prioritized-tool-urls]: data/prioritized-tool-urls.txt +## Flickr + +**Description:** _With over 5 billion photos (many with valuable metadata such +as tags, geolocation, and Exif data), the Flickr community creates wonderfully +rich data. The Flickr API is how you can access that data. In fact, almost all +the functionality that runs flickr.com is available through the API._ ([Flickr: +The Flickr Developer Guide](https://www.flickr.com/services/developer/)) + +**API documentation link:** +- [API documentation - Flickr Services](https://www.flickr.com/services/api/) + +**API information:** +- API key required +- Query limit: 3600 requests per hour +- Data available through CSV format + +## GitHub + +**Description:** A development platform for hosting and managing code. + +**API documentation link:** +- [GitHub REST API v3](https://docs.github.com/en/rest) + +**API information:** +- API key not required but recommended by GitHub +- Query limit: 60 requests per hour if unauthenticated, + 5000 requests per hour if authenticated +- Data available through JSON format + + ## GCS (Google Custom Search) JSON API **Description:** The Custom Search JSON API allows user-defined detailed query @@ -68,40 +98,17 @@ and access towards related query data using a programmable search engine. [reference-appendix]: https://developers.google.com/custom-search/docs/xml_results_appendices -## GitHub - -**Description:** A development platform for hosting and managing code. - -**API documentation link:** -- [GitHub REST API v3](https://docs.github.com/en/rest) - -**API information:** -- API key not required but recommended by GitHub -- Query limit: 60 requests per hour if unauthenticated, - 5000 requests per hour if authenticated -- Data available through JSON format - -## Wikipedia +## Internet Archive Python Interface -**Description:** The Wikipedia API allows users to query statistics of pages, -categories, revisions from a public API endpoint. We have included two urls in -the project: The `WIKIPEDIA_BASE_URL` AND `WIKIPEDIA_MATRIX_URL`. The -`WIKIPEDIA_BASE_URL` provides access to articles, categories, and metadata from -the English version of Wikipedia. It runs on the MediaWiki Action API, but this -instance only provides English Wikipedia data. Then the `WIKIPEDIA_MATRIX_URL` -provides access to information of all wikimedia projects including the different -language edition of wikipedia. It runs on the Meta-Wiki API. +**Description:** A python interface to archive.org to achieve API requests +towards internet archive. **API documentation link:** -[WIKIPEDIA_BASE_URL documentation](https://en.wikipedia.org/w/api.php) -[WIKIPEDIA_BASE_URL reference page](https://www.mediawiki.org/wiki/API:Main_page) -[WIKIPEDIA_MATRIX_URL documentation](https://meta.wikimedia.org/w/api.php) -[WIKIPEDIA_MATRIX_URL reference page](https://www.mediawiki.org/wiki/API:Sitematrix) +- [internetarchive.Search - Internetarchive: A Python Interface to + archive.org][ia-search] **API information:** - No API key required -- Query limit: It is rate-limited only to prevent abuse -- Data available through XML or JSON format - No query limits [ia-search]: https://internetarchive.readthedocs.io/en/stable/internetarchive.html#internetarchive.Search @@ -168,11 +175,10 @@ adjust video parameters, and obtain search results. - Query limit: depends on the type and number of requests - Data available through JSON format -## EUROPEANA DATA API + ## EUROPEANA DATA API **Description:** -The **Europeana Search API** provides access to digital cultural heritage metadata records aggregated from museums, libraries, and archives across Europe. -This project uses the API to fetch aggregated counts of cultural heritage records by data provider, rights statement, and theme. +The **Europeana Search API** provides access to digital cultural heritage metadata records aggregated from museums, libraries, and archives across Europe. This project uses the API to fetch aggregated counts of cultural heritage records by data provider, rights statement, and theme. **Official API Documentation:** - [Search API Documentation](https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation) From ed87d34958500f470fb42ece22a4dd8cc119200a Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Thu, 30 Oct 2025 13:10:48 +0300 Subject: [PATCH 22/30] Add Europeana Data API section --- sources.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sources.md b/sources.md index 3fce919d..7a5dd27c 100644 --- a/sources.md +++ b/sources.md @@ -175,7 +175,7 @@ adjust video parameters, and obtain search results. - Query limit: depends on the type and number of requests - Data available through JSON format - ## EUROPEANA DATA API +## EUROPEANA DATA API **Description:** The **Europeana Search API** provides access to digital cultural heritage metadata records aggregated from museums, libraries, and archives across Europe. This project uses the API to fetch aggregated counts of cultural heritage records by data provider, rights statement, and theme. From 6e096d179231e7d9e6cf47784a557b8efd7947be Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Thu, 30 Oct 2025 14:26:39 +0300 Subject: [PATCH 23/30] chore: update sources.md to match latest main (remove outdated sources) --- sources.md | 135 +++++++++-------------------------------------------- 1 file changed, 23 insertions(+), 112 deletions(-) diff --git a/sources.md b/sources.md index 7a5dd27c..a4119ee5 100644 --- a/sources.md +++ b/sources.md @@ -23,36 +23,6 @@ tool paths. [prioritized-tool-urls]: data/prioritized-tool-urls.txt -## Flickr - -**Description:** _With over 5 billion photos (many with valuable metadata such -as tags, geolocation, and Exif data), the Flickr community creates wonderfully -rich data. The Flickr API is how you can access that data. In fact, almost all -the functionality that runs flickr.com is available through the API._ ([Flickr: -The Flickr Developer Guide](https://www.flickr.com/services/developer/)) - -**API documentation link:** -- [API documentation - Flickr Services](https://www.flickr.com/services/api/) - -**API information:** -- API key required -- Query limit: 3600 requests per hour -- Data available through CSV format - -## GitHub - -**Description:** A development platform for hosting and managing code. - -**API documentation link:** -- [GitHub REST API v3](https://docs.github.com/en/rest) - -**API information:** -- API key not required but recommended by GitHub -- Query limit: 60 requests per hour if unauthenticated, - 5000 requests per hour if authenticated -- Data available through JSON format - - ## GCS (Google Custom Search) JSON API **Description:** The Custom Search JSON API allows user-defined detailed query @@ -98,96 +68,37 @@ and access towards related query data using a programmable search engine. [reference-appendix]: https://developers.google.com/custom-search/docs/xml_results_appendices -## Internet Archive Python Interface - -**Description:** A python interface to archive.org to achieve API requests -towards internet archive. - -**API documentation link:** -- [internetarchive.Search - Internetarchive: A Python Interface to - archive.org][ia-search] - -**API information:** -- No API key required -- No query limits - -[ia-search]: https://internetarchive.readthedocs.io/en/stable/internetarchive.html#internetarchive.Search - - -## MediaWiki Action API - -**Description:** _The MediaWiki Action API is a web service that allows access -to some wiki features like authentication, page operations, and search. It can -provide meta information about the wiki and the logged-in user._ ([API:Main -page - MediaWiki](https://www.mediawiki.org/wiki/API:Main_page)) - -**API documentation link:** -- [MediaWiki Action API](https://www.mediawiki.org/wiki/API:Main_page) - -**API information:** - - No API key required - - Query limit: depends on user status and request type - - Data available through XML or JSON format - - -## The Metropolitan Museum of Art Collection API - -**Description:** _The Met’s Open Access datasets are available through our API. -The API (RESTful web service in JSON format) gives access to all of The Met’s -Open Access data and to corresponding high resolution images (JPEG format) that -are in the public domain._ ([The Metropolitan Museum of Art Collection -API](https://metmuseum.github.io/)) - -**API documentation link:** -- [Latest Updates | The Metropolitan Museum of Art Collection - API](https://metmuseum.github.io/) - -**API information:** - - No API key required - - 80 queries per second - - -## Vimeo API +## GitHub -**Description:** The Vimeo API allows users to perform filtered, advanced -search on Vimeo videos. +**Description:** A development platform for hosting and managing code. **API documentation link:** -- [Getting Started with the Vimeo API](https://developer.vimeo.com/api/start) +- [GitHub REST API v3](https://docs.github.com/en/rest) **API information:** - - API key required - - Query limit: 5000 authenticated requests per day - - Data available through JSON format - +- API key not required but recommended by GitHub +- Query limit: 60 requests per hour if unauthenticated, + 5000 requests per hour if authenticated +- Data available through JSON format -## YouTube Data API +## Wikipedia -**Description:** An API from YouTube for platform users to upload videos, -adjust video parameters, and obtain search results. +**Description:** The Wikipedia API allows users to query statistics of pages, +categories, revisions from a public API endpoint. We have included two urls in +the project: The `WIKIPEDIA_BASE_URL` AND `WIKIPEDIA_MATRIX_URL`. The +`WIKIPEDIA_BASE_URL` provides access to articles, categories, and metadata from +the English version of Wikipedia. It runs on the MediaWiki Action API, but this +instance only provides English Wikipedia data. Then the `WIKIPEDIA_MATRIX_URL` +provides access to information of all wikimedia projects including the different +language edition of wikipedia. It runs on the Meta-Wiki API. **API documentation link:** -- [Search: list | YouTube Data API | Google - Developers](https://developers.google.com/youtube/v3/docs/search/list) +[WIKIPEDIA_BASE_URL documentation](https://en.wikipedia.org/w/api.php) +[WIKIPEDIA_BASE_URL reference page](https://www.mediawiki.org/wiki/API:Main_page) +[WIKIPEDIA_MATRIX_URL documentation](https://meta.wikimedia.org/w/api.php) +[WIKIPEDIA_MATRIX_URL reference page](https://www.mediawiki.org/wiki/API:Sitematrix) **API information:** - - API key required - - Query limit: depends on the type and number of requests - - Data available through JSON format - -## EUROPEANA DATA API - -**Description:** -The **Europeana Search API** provides access to digital cultural heritage metadata records aggregated from museums, libraries, and archives across Europe. This project uses the API to fetch aggregated counts of cultural heritage records by data provider, rights statement, and theme. - -**Official API Documentation:** -- [Search API Documentation](https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation) - -**API information:** -- API key required -- Query parameters allow: - - Full-text searching (`query`) - - Retrieving metadata facets (`profile=facets`) - - Filtering by data provider, rights statement, and theme -- Data available through JSON format -- Offset-based pagination +- No API key required +- Query limit: It is rate-limited only to prevent abuse +- Data available through XML or JSON format From de913b55922e63b7c135b5f00c23ee5a7d788ca2 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Thu, 30 Oct 2025 14:35:11 +0300 Subject: [PATCH 24/30] Add europeana sources --- sources.md | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/sources.md b/sources.md index a4119ee5..edca3aa2 100644 --- a/sources.md +++ b/sources.md @@ -102,3 +102,20 @@ language edition of wikipedia. It runs on the Meta-Wiki API. - No API key required - Query limit: It is rate-limited only to prevent abuse - Data available through XML or JSON format + +## EUROPEANA DATA API + +**Description:** +The **Europeana Search API** provides access to digital cultural heritage metadata records aggregated from museums, libraries, and archives across Europe. This project uses the API to fetch aggregated counts of cultural heritage records by data provider, rights statement, and theme. + +**Official API Documentation:** +- [Search API Documentation](https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation) + +**API information:** +- API key required +- Query parameters allow: + - Full-text searching (`query`) + - Retrieving metadata facets (`profile=facets`) + - Filtering by data provider, rights statement, and theme +- Data available through JSON format +- Offset-based pagination From ddecf4571de71d9f8ac524540cee3e890e63a436 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Thu, 30 Oct 2025 14:48:17 +0300 Subject: [PATCH 25/30] chore: update env.example to match latest main --- env.example | 7 ------- 1 file changed, 7 deletions(-) diff --git a/env.example b/env.example index 8c66ebd3..b53d4f03 100644 --- a/env.example +++ b/env.example @@ -29,10 +29,3 @@ # https://docs.github.com/en/rest/authentication/authenticating-to-the-rest-api # GH_TOKEN = -# "The flickr developer guide: https://www.flickr.com/services/developer/" - -# FLICKR_API_KEY = -# FLICKR_API_SECRET = - -# "Europeana Search API Documentation: https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation#Request -# EUROPEANA_API_KEY = From 6893d82d3f8aae453f9e809ff49753ca31bd4499 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Thu, 30 Oct 2025 14:52:19 +0300 Subject: [PATCH 26/30] Updated env.example --- env.example | 3 +++ 1 file changed, 3 insertions(+) diff --git a/env.example b/env.example index b53d4f03..547e82ba 100644 --- a/env.example +++ b/env.example @@ -29,3 +29,6 @@ # https://docs.github.com/en/rest/authentication/authenticating-to-the-rest-api # GH_TOKEN = + +# "Europeana Search API Documentation: https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation#Request +# EUROPEANA_API_KEY = From 4fd4920938ae7889185ed3fcb663461b1f8eb151 Mon Sep 17 00:00:00 2001 From: Joy Akinyi Date: Fri, 31 Oct 2025 13:06:03 +0300 Subject: [PATCH 27/30] Uses prefetched facets --- scripts/1-fetch/europeana_fetch.py | 195 ++++++++++++++--------------- 1 file changed, 95 insertions(+), 100 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index d56f096a..3c3e6f82 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -15,6 +15,7 @@ import textwrap import time import traceback +from operator import itemgetter # Third-party import requests @@ -42,7 +43,7 @@ FILE_WITHOUT_THEMES = shared.path_join( PATHS["data_phase"], "europeana_without_themes.csv" ) -HEADER_WITH_THEMES = ["DATA_PROVIDER", "LEGAL_TOOL", "THEME", "COUNT"] +HEADER_WITH_THEMES = ["DATA_PROVIDER", "THEME", "LEGAL_TOOL", "COUNT"] HEADER_WITHOUT_THEMES = ["DATA_PROVIDER", "LEGAL_TOOL", "COUNT"] QUARTER = os.path.basename(PATHS["data_quarter"]) TIMEOUT = 25 @@ -205,12 +206,16 @@ def simplify_legal_tool(legal_tool): def get_facet_list(session, facet_field): - """Fetch complete facet list from Europeana API for a given facet field.""" + """ + Fetch complete facet list from Europeana API for a given facet field, + returning both label and count, sorted by count descending. + Returns: list of dicts: [{'label': ..., 'count': ...}] + """ all_values = [] offset = 0 limit = 1000 - LOGGER.info(f"Fetching {facet_field} facet values.") + LOGGER.info(f"Fetching {facet_field} facet values with counts.") while True: params = { @@ -238,13 +243,13 @@ def get_facet_list(session, facet_field): break fields = facets[0]["fields"] - new_values = [f["label"] for f in fields if f.get("label")] - - for v in new_values: - if v not in all_values: - all_values.append(v) + for f in fields: + label = f.get("label") + count = f.get("count", 0) + if label and not any(d["label"] == label for d in all_values): + all_values.append({"label": label, "count": count}) - if len(new_values) < limit: + if len(fields) < limit: break offset += limit @@ -253,118 +258,98 @@ def get_facet_list(session, facet_field): LOGGER.info( f"Completed fetching {facet_field}. Total unique: {len(all_values)}" ) - all_values.sort() - return all_values + # Sort by count descending + all_values.sort(key=lambda x: x["count"], reverse=True) + return all_values -def fetch_europeana_data_without_themes(session, limit=None): - """Fetch counts by DATA_PROVIDER and RIGHTS using facets.""" - LOGGER.info("Fetching Europeana counts without themes.") - params = { - "wskey": EUROPEANA_API_KEY, - "query": "*", - "rows": 0, - "profile": "facets", - "facet": ["DATA_PROVIDER", "RIGHTS"], - "f.DATA_PROVIDER.facet.limit": 1000, - "f.RIGHTS.facet.limit": 100, - } +def fetch_europeana_data_without_themes( + session, providers_full, rights_full, limit=None +): + """ + Fetch counts per DATA_PROVIDER × RIGHTS using pre-fetched facets. + """ + output = [] - try: - resp = session.get(BASE_URL, params=params, timeout=TIMEOUT) - resp.raise_for_status() - data = resp.json() - except requests.RequestException as e: - LOGGER.error(f"Failed to fetch facets: {e}") - return [] - - facets = {f["name"]: f["fields"] for f in data.get("facets", [])} - provider_fields = facets.get("DATA_PROVIDER", []) - rights_fields = facets.get("RIGHTS", []) + # Filter non-zero providers + providers_nonzero = [p["label"] for p in providers_full if p["count"] > 0] if limit: - provider_fields = provider_fields[:limit] + providers_nonzero = providers_nonzero[:limit] - output = [] - for provider_entry in provider_fields: - provider = provider_entry["label"] - provider_count = provider_entry["count"] - if provider_count == 0: - continue - LOGGER.info(f"Fetching rights data for provider={provider}") - for rights_entry in rights_fields: - rights = rights_entry["label"] - query = f'DATA_PROVIDER:"{provider}" AND RIGHTS:"{rights}"' + # Filter non-zero rights + rights_nonzero = [r["label"] for r in rights_full if r["count"] > 0] + + for i, provider in enumerate(providers_nonzero, start=1): + LOGGER.info( + f"[{i}/{len(providers_nonzero)}] " + f"Fetching counts for provider={provider}" + ) + + for rights_url in rights_nonzero: + simplified_rights = simplify_legal_tool(rights_url) + query = f'DATA_PROVIDER:"{provider}" AND RIGHTS:"{rights_url}"' params_detail = { "wskey": EUROPEANA_API_KEY, "rows": 0, "query": query, } try: - resp_detail = session.get( + resp = session.get( BASE_URL, params=params_detail, timeout=TIMEOUT ) - resp_detail.raise_for_status() - count = resp_detail.json().get("totalResults", 0) + resp.raise_for_status() + count = resp.json().get("totalResults", 0) if count > 0: output.append( { "DATA_PROVIDER": provider, - "LEGAL_TOOL": simplify_legal_tool(rights), + "LEGAL_TOOL": simplified_rights, "COUNT": count, } ) - except requests.RequestException as e: LOGGER.warning( - f"Failed for provider={provider}, rights={rights}: {e}" + f"Failed for provider={provider}, rights={rights_url}: {e}" ) time.sleep(0.01) - LOGGER.info(f"Aggregated {len(output)} records (without themes).") - return output + # Sort by DATA_PROVIDER, LEGAL_TOOL + output = sorted(output, key=itemgetter("DATA_PROVIDER", "LEGAL_TOOL")) + + LOGGER.info( + f"Aggregated {len(output)} records for provider-rights counts." + ) + return output -def fetch_europeana_data_with_themes(session, themes, limit=None): - """Fetch counts by DATA_PROVIDER, RIGHTS, and THEME using facets.""" - LOGGER.info("Fetching Europeana counts with themes") - params = { - "wskey": EUROPEANA_API_KEY, - "query": "*", - "rows": 0, - "profile": "facets", - "facet": ["DATA_PROVIDER", "RIGHTS"], - "f.DATA_PROVIDER.facet.limit": 1000, - "f.RIGHTS.facet.limit": 100, - } +def fetch_europeana_data_with_themes( + session, providers_full, rights_full, themes, limit=None +): + """ + Fetch counts per DATA_PROVIDER × RIGHTS × THEME + Uses pre-fetched providers_full and rights_full lists. + """ + output = [] - try: - resp = session.get(BASE_URL, params=params, timeout=TIMEOUT) - resp.raise_for_status() - data = resp.json() - except requests.RequestException as e: - LOGGER.error(f"Failed to fetch facets: {e}") - return [] - - facets = {f["name"]: f["fields"] for f in data.get("facets", [])} - provider_fields = facets.get("DATA_PROVIDER", []) - rights_fields = facets.get("RIGHTS", []) + # Filter non-zero providers + providers_nonzero = [p["label"] for p in providers_full if p["count"] > 0] if limit: - provider_fields = provider_fields[:limit] + providers_nonzero = providers_nonzero[:limit] - output = [] - for provider_entry in provider_fields: - provider = provider_entry["label"] - provider_count = provider_entry["count"] - if provider_count == 0: - continue - LOGGER.info(f"Fetching theme+rights data for provider={provider}") - for rights_entry in rights_fields: - rights = rights_entry["label"] - simplified_rights = simplify_legal_tool(rights) + # Filter non-zero rights + rights_nonzero = [r["label"] for r in rights_full if r["count"] > 0] + for i, provider in enumerate(providers_nonzero, start=1): + LOGGER.info( + f"[{i}/{len(providers_nonzero)}]" + f"Fetching rights+theme counts for provider={provider}" + ) + + for rights_url in rights_nonzero: + simplified_rights = simplify_legal_tool(rights_url) for theme in themes: - query = f'DATA_PROVIDER:"{provider}" AND RIGHTS:"{rights}"' + query = f'DATA_PROVIDER:"{provider}" AND RIGHTS:"{rights_url}"' params_detail = { "wskey": EUROPEANA_API_KEY, "rows": 0, @@ -372,30 +357,35 @@ def fetch_europeana_data_with_themes(session, themes, limit=None): "theme": theme, } try: - resp_detail = session.get( + resp = session.get( BASE_URL, params=params_detail, timeout=TIMEOUT ) - resp_detail.raise_for_status() - count = resp_detail.json().get("totalResults", 0) + resp.raise_for_status() + count = resp.json().get("totalResults", 0) if count > 0: output.append( { "DATA_PROVIDER": provider, - "LEGAL_TOOL": simplified_rights, "THEME": theme, + "LEGAL_TOOL": simplified_rights, "COUNT": count, } ) - except requests.RequestException as e: LOGGER.warning( - f"Failed for provider={provider}, " - f"rights={rights}, " - f"theme={theme}: " - f"{e}" + f"Failed for provider={provider}," + f"rights={rights_url}, theme={theme}: {e}" ) time.sleep(0.01) - LOGGER.info(f"Aggregated {len(output)} records (with themes).") + + # Sort by DATA_PROVIDER, THEME, LEGAL_TOOL + output = sorted( + output, key=itemgetter("DATA_PROVIDER", "THEME", "LEGAL_TOOL") + ) + + LOGGER.info( + f"Aggregated {len(output)} records for provider-rights-theme counts." + ) return output @@ -445,17 +435,22 @@ def main(): session = get_requests_session() + # Fetch facet lists once, including counts providers_full = get_facet_list(session, "DATA_PROVIDER") rights_full = get_facet_list(session, "RIGHTS") + LOGGER.info(f"Facet providers loaded: {len(providers_full)}") LOGGER.info(f"Facet rights loaded: {len(rights_full)}") + + # Pass facets to fetch functions data_no_theme = fetch_europeana_data_without_themes( - session, limit=args.limit + session, providers_full, rights_full, limit=args.limit ) data_with_theme = fetch_europeana_data_with_themes( - session, THEMES, limit=args.limit + session, providers_full, rights_full, THEMES, limit=args.limit ) + # Write to CSV and optionally push to git args = write_data(args, data_no_theme, data_with_theme) args = shared.git_add_and_commit( args, From 49577464a2d97c0cb1931203ba690e977a775b8e Mon Sep 17 00:00:00 2001 From: Timid Robot Zehta Date: Sat, 1 Nov 2025 17:23:49 +0100 Subject: [PATCH 28/30] order sources and make formatting more consistent --- env.example | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/env.example b/env.example index 547e82ba..d30d3762 100644 --- a/env.example +++ b/env.example @@ -1,7 +1,15 @@ # This file must be copied to .env and the appropriate variables populated. -## GCS (Google Custom Search) +## Europeana + +# Europeana Search API Documentation: +# https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation#Request + +# EUROPEANA_API_KEY = + + +# GCS (Google Custom Search) # https://developers.google.com/custom-search/v1/introduction # "Custom Search JSON API requires the use of an API key. An API key is a way @@ -29,6 +37,3 @@ # https://docs.github.com/en/rest/authentication/authenticating-to-the-rest-api # GH_TOKEN = - -# "Europeana Search API Documentation: https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation#Request -# EUROPEANA_API_KEY = From 5b1aa20cee91b2b2fc03a6dc6e4c780ef8be37c6 Mon Sep 17 00:00:00 2001 From: Timid Robot Zehta Date: Sat, 1 Nov 2025 17:27:35 +0100 Subject: [PATCH 29/30] order sources and add a couple of Europeana notes --- sources.md | 37 ++++++++++++++++++++----------------- 1 file changed, 20 insertions(+), 17 deletions(-) diff --git a/sources.md b/sources.md index edca3aa2..2b3e1a79 100644 --- a/sources.md +++ b/sources.md @@ -23,6 +23,26 @@ tool paths. [prioritized-tool-urls]: data/prioritized-tool-urls.txt +## EUROPEANA DATA API + +**Description:** +The **Europeana Search API** provides access to digital cultural heritage metadata records aggregated from museums, libraries, and archives across Europe. This project uses the API to fetch aggregated counts of cultural heritage records by data provider, rights statement, and theme. + +**Official API Documentation:** +- [Search API Documentation](https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation) + - Themes are listed in the Search API Request Parameter accordion + +**API information:** +- API key required +- Minimum 0.003 seconds between queries +- Query parameters allow: + - Full-text searching (`query`) + - Retrieving metadata facets (`profile=facets`) + - Filtering by data provider, rights statement, and theme +- Data available through JSON format +- Offset-based pagination + + ## GCS (Google Custom Search) JSON API **Description:** The Custom Search JSON API allows user-defined detailed query @@ -102,20 +122,3 @@ language edition of wikipedia. It runs on the Meta-Wiki API. - No API key required - Query limit: It is rate-limited only to prevent abuse - Data available through XML or JSON format - -## EUROPEANA DATA API - -**Description:** -The **Europeana Search API** provides access to digital cultural heritage metadata records aggregated from museums, libraries, and archives across Europe. This project uses the API to fetch aggregated counts of cultural heritage records by data provider, rights statement, and theme. - -**Official API Documentation:** -- [Search API Documentation](https://europeana.atlassian.net/wiki/spaces/EF/pages/2385739812/Search+API+Documentation) - -**API information:** -- API key required -- Query parameters allow: - - Full-text searching (`query`) - - Retrieving metadata facets (`profile=facets`) - - Filtering by data provider, rights statement, and theme -- Data available through JSON format -- Offset-based pagination From 4c8588b1afc76c5483081dc0ef5f620a4103353f Mon Sep 17 00:00:00 2001 From: Timid Robot Zehta Date: Sat, 1 Nov 2025 17:39:49 +0100 Subject: [PATCH 30/30] minor updates to europeana fetch - update --limit help text - use standard (for this repository) backup_factor=10 - use itemgettr for all sorts - remove redundant facet logging --- scripts/1-fetch/europeana_fetch.py | 9 +++------ 1 file changed, 3 insertions(+), 6 deletions(-) diff --git a/scripts/1-fetch/europeana_fetch.py b/scripts/1-fetch/europeana_fetch.py index 3c3e6f82..cd5a9d41 100755 --- a/scripts/1-fetch/europeana_fetch.py +++ b/scripts/1-fetch/europeana_fetch.py @@ -95,7 +95,7 @@ def parse_arguments(): "--limit", type=int, default=5, - help="Limit number of providers for testing.", + help="Limit number of data providers", ) args = parser.parse_args() if not args.enable_save and args.enable_git: @@ -106,7 +106,7 @@ def parse_arguments(): def get_requests_session(): """Create a requests session with retry.""" max_retries = Retry( - total=5, backoff_factor=5, status_forcelist=shared.STATUS_FORCELIST + total=5, backoff_factor=10, status_forcelist=shared.STATUS_FORCELIST ) session = requests.Session() session.mount("https://", HTTPAdapter(max_retries=max_retries)) @@ -260,7 +260,7 @@ def get_facet_list(session, facet_field): ) # Sort by count descending - all_values.sort(key=lambda x: x["count"], reverse=True) + all_values.sort(key=itemgetter("count"), reverse=True) return all_values @@ -439,9 +439,6 @@ def main(): providers_full = get_facet_list(session, "DATA_PROVIDER") rights_full = get_facet_list(session, "RIGHTS") - LOGGER.info(f"Facet providers loaded: {len(providers_full)}") - LOGGER.info(f"Facet rights loaded: {len(rights_full)}") - # Pass facets to fetch functions data_no_theme = fetch_europeana_data_without_themes( session, providers_full, rights_full, limit=args.limit