diff --git a/scripts/2-process/gcs_process.py b/scripts/2-process/gcs_process.py index a9a60c9a..0726df92 100755 --- a/scripts/2-process/gcs_process.py +++ b/scripts/2-process/gcs_process.py @@ -94,27 +94,29 @@ def process_product_totals(args, count_data): "Licenses version 1.0": 0, "CC0 1.0": 0, "Public Domain Mark 1.0": 0, - "Certification 1.0 US": 0, + "CERTIFICATION 1.0 US": 0, } for row in count_data.itertuples(index=False): tool = row[0] count = row[1] - if tool.startswith("PDM"): + if tool == "PDM 1.0": key = "Public Domain Mark 1.0" - elif "CC0" in tool: - key = "CC0 1.0" - elif "PUBLICDOMAIN" in tool: - key = "Certification 1.0 US" - elif "4.0" in tool: + elif tool == "CC0 1.0": + key = tool + elif tool == "CERTIFICATION 1.0 US": + key = tool + elif tool.startswith("CC ") and tool.endswith("4.0"): key = "Licenses version 4.0" - elif "3.0" in tool: + elif tool.startswith("CC ") and "3.0" in tool: key = "Licenses version 3.0" - elif "2." in tool: + elif tool.startswith("CC ") and "2." in tool: key = "Licenses version 2.x" - elif "1.0" in tool: + elif tool.startswith("CC ") and "1.0" in tool: key = "Licenses version 1.0" else: - raise shared.QuantifyingException("Invalid TOOL_IDENTIFIER") + raise shared.QuantifyingException( + f"Invalid TOOL_IDENTIFIER: {tool}" + ) data[key] += count data = pd.DataFrame( @@ -129,7 +131,11 @@ def process_latest_prior_retired_totals(args, count_data): Process count data: totals by unit in three categories: latest, prior, and retired """ - LOGGER.info(process_latest_prior_retired_totals.__doc__.strip()) + LOGGER.info( + process_latest_prior_retired_totals.__doc__.strip().replace( + "\n ", "" + ) + ) # https://creativecommons.org/retiredlicenses/ retired = [ # DevNations, @@ -144,14 +150,14 @@ def process_latest_prior_retired_totals(args, count_data): "CC NC-SAMPLING+", # NonCommercial-ShareAlike "CC NC-SA ", - # Public Domain Dedication and Certification - "CC PUBLICDOMAIN", # Sampling "CC SAMPLING ", # Sampling+ "CC SAMPLING+ ", # ShareAlike "CC SA ", + # Certification + "CERTIFICATION ", ] data = {"latest": {}, "prior": {}, "retired": {}} status = {"Latest": 0, "Prior": 0, "Retired": 0} @@ -218,7 +224,7 @@ def process_totals_by_free_cultural(args, count_data): for row in count_data.itertuples(index=False): tool = row[0] count = row[1] - if tool.startswith("PDM") or "CC0" in tool or "PUBLICDOMAIN" in tool: + if tool in ("CERTIFICATION 1.0 US", "CC0 1.0", "PDM 1.0"): key = "Approved for Free Cultural Works" else: parts = tool.split() @@ -252,7 +258,7 @@ def process_totals_by_restrictions(args, count_data): for row in count_data.itertuples(index=False): tool = row[0] count = row[1] - if tool.startswith("PDM") or "CC0" in tool or "PUBLICDOMAIN" in tool: + if tool.startswith("PDM") or "CC0" in tool or "CERTIFICATION" in tool: key = "level 0 - unrestricted" else: parts = tool.split() diff --git a/scripts/3-report/gcs_report.py b/scripts/3-report/gcs_report.py index f0d831c1..59f707f7 100755 --- a/scripts/3-report/gcs_report.py +++ b/scripts/3-report/gcs_report.py @@ -122,7 +122,7 @@ def plot_products(args): LOGGER.info(f"data file: {file_path.replace(PATHS['repo'], '.')}") name_label = "CC legal tool product" data = shared.open_data_file(LOGGER, file_path, index_col=name_label) - + data.index = data.index.map(lambda x: x.replace("CC0 1.0", "CC0 1.0")) data = data[::-1] # reverse order title = "Products totals and percentages" @@ -213,6 +213,7 @@ def plot_latest_tools(args): LOGGER.info(f"data file: {file_path.replace(PATHS['repo'], '.')}") name_label = "CC legal tool" data = shared.open_data_file(LOGGER, file_path, index_col=name_label) + data.index = data.index.map(lambda x: x.replace(" ", " ")) data.sort_values(name_label, ascending=False, inplace=True) title = "Latest CC legal tools" @@ -242,6 +243,9 @@ def plot_latest_tools(args): image_path, "The plot shows the latest Creative Commons (CC) legal tool totals and" " percentages.", + "The latest tools include Licenses version 4.0 (CC BY 4.0, CC BY-NC" + " 4.0, CC BY-NC-ND 4.0, CC BY-NC-SA 4.0, CC-BY-ND 4.0, CC BY-SA 4.0)," + " CC0 1.0, and the Public Domain Mark (PDM 1.0).", ) @@ -256,6 +260,7 @@ def plot_prior_tools(args): LOGGER.info(f"data file: {file_path.replace(PATHS['repo'], '.')}") name_label = "CC legal tool" data = shared.open_data_file(LOGGER, file_path, index_col=name_label) + data.index = data.index.map(lambda x: x.replace(" ", " ")) data.sort_values(name_label, ascending=False, inplace=True) title = "Prior CC legal tools" @@ -284,7 +289,8 @@ def plot_prior_tools(args): title, image_path, "The plot shows prior Creative Commons (CC) legal tool totals and" - " percentages. The unit names have been normalized (~~`CC BY-ND-NC`~~" + " percentages. Prior CC licenses include versions 1.0, 2.0, 2.1, 2.5," + " and 3.0. The unit names have been normalized (~~`CC BY-ND-NC`~~" " => `CC BY-NC-ND`).", ) @@ -301,6 +307,7 @@ def plot_retired_tools(args): LOGGER.info(f"data file: {file_path.replace(PATHS['repo'], '.')}") name_label = "CC legal tool" data = shared.open_data_file(LOGGER, file_path, index_col=name_label) + data.index = data.index.map(lambda x: x.replace(" ", " ")) data.sort_values(name_label, ascending=False, inplace=True) title = "Retired CC legal tools" @@ -379,14 +386,15 @@ def plot_countries_highest_usage(args): SECTION_TITLE, title, image_path, - "The plot shows countries with the highest useage of the latest" + "The plot shows countries with the highest usage of the latest" " Creative Commons (CC) legal tools.", "The latest tools include Licenses version 4.0 (CC BY 4.0, CC BY-NC" " 4.0, CC BY-NC-ND 4.0, CC BY-NC-SA 4.0, CC-BY-ND 4.0, CC BY-SA 4.0)," " CC0 1.0, and the Public Domain Mark (PDM 1.0).\n" "\n" - f"The complete data set indicates there are a total of {total_count}" - " online works using one of the latest CC legal tools.", + "The complete countries data set indicates there are a total of" + f" {total_count} online works using one of the latest CC legal tools." + " This conflicts with the languages data set, below.", ) @@ -433,14 +441,15 @@ def plot_languages_highest_usage(args): SECTION_TITLE, title, image_path, - "The plot shows the languages with the highest useage of the latest" + "The plot shows the languages with the highest usage of the latest" " Creative Commons (CC) legal tools.", "The latest tools include Licenses version 4.0 (CC BY 4.0, CC BY-NC" " 4.0, CC BY-NC-ND 4.0, CC BY-NC-SA 4.0, CC-BY-ND 4.0, CC BY-SA 4.0)," " CC0 1.0, and the Public Domain Mark (PDM 1.0).\n" "\n" - f"The complete data set indicates there are a total of {total_count}" - " online works using one of the latest CC legal tools.", + "The complete languages data set indicates there are a total of" + f" {total_count} online works using one of the latest CC legal tools." + " This conflicts with the countries data set, above.", )