Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
38 changes: 22 additions & 16 deletions scripts/2-process/gcs_process.py
Original file line number Diff line number Diff line change
Expand Up @@ -94,27 +94,29 @@ def process_product_totals(args, count_data):
"Licenses version 1.0": 0,
"CC0 1.0": 0,
"Public Domain Mark 1.0": 0,
"Certification 1.0 US": 0,
"CERTIFICATION 1.0 US": 0,
}
for row in count_data.itertuples(index=False):
tool = row[0]
count = row[1]
if tool.startswith("PDM"):
if tool == "PDM 1.0":
key = "Public Domain Mark 1.0"
elif "CC0" in tool:
key = "CC0 1.0"
elif "PUBLICDOMAIN" in tool:
key = "Certification 1.0 US"
elif "4.0" in tool:
elif tool == "CC0 1.0":
key = tool
elif tool == "CERTIFICATION 1.0 US":
key = tool
elif tool.startswith("CC ") and tool.endswith("4.0"):
key = "Licenses version 4.0"
elif "3.0" in tool:
elif tool.startswith("CC ") and "3.0" in tool:
key = "Licenses version 3.0"
elif "2." in tool:
elif tool.startswith("CC ") and "2." in tool:
key = "Licenses version 2.x"
elif "1.0" in tool:
elif tool.startswith("CC ") and "1.0" in tool:
key = "Licenses version 1.0"
else:
raise shared.QuantifyingException("Invalid TOOL_IDENTIFIER")
raise shared.QuantifyingException(
f"Invalid TOOL_IDENTIFIER: {tool}"
)
data[key] += count

data = pd.DataFrame(
Expand All @@ -129,7 +131,11 @@ def process_latest_prior_retired_totals(args, count_data):
Process count data: totals by unit in three categories: latest, prior,
and retired
"""
LOGGER.info(process_latest_prior_retired_totals.__doc__.strip())
LOGGER.info(
process_latest_prior_retired_totals.__doc__.strip().replace(
"\n ", ""
)
)
# https://creativecommons.org/retiredlicenses/
retired = [
# DevNations,
Expand All @@ -144,14 +150,14 @@ def process_latest_prior_retired_totals(args, count_data):
"CC NC-SAMPLING+",
# NonCommercial-ShareAlike
"CC NC-SA ",
# Public Domain Dedication and Certification
"CC PUBLICDOMAIN",
# Sampling
"CC SAMPLING ",
# Sampling+
"CC SAMPLING+ ",
# ShareAlike
"CC SA ",
# Certification
"CERTIFICATION ",
]
data = {"latest": {}, "prior": {}, "retired": {}}
status = {"Latest": 0, "Prior": 0, "Retired": 0}
Expand Down Expand Up @@ -218,7 +224,7 @@ def process_totals_by_free_cultural(args, count_data):
for row in count_data.itertuples(index=False):
tool = row[0]
count = row[1]
if tool.startswith("PDM") or "CC0" in tool or "PUBLICDOMAIN" in tool:
if tool in ("CERTIFICATION 1.0 US", "CC0 1.0", "PDM 1.0"):
key = "Approved for Free Cultural Works"
else:
parts = tool.split()
Expand Down Expand Up @@ -252,7 +258,7 @@ def process_totals_by_restrictions(args, count_data):
for row in count_data.itertuples(index=False):
tool = row[0]
count = row[1]
if tool.startswith("PDM") or "CC0" in tool or "PUBLICDOMAIN" in tool:
if tool.startswith("PDM") or "CC0" in tool or "CERTIFICATION" in tool:
key = "level 0 - unrestricted"
else:
parts = tool.split()
Expand Down
25 changes: 17 additions & 8 deletions scripts/3-report/gcs_report.py
Original file line number Diff line number Diff line change
Expand Up @@ -122,7 +122,7 @@ def plot_products(args):
LOGGER.info(f"data file: {file_path.replace(PATHS['repo'], '.')}")
name_label = "CC legal tool product"
data = shared.open_data_file(LOGGER, file_path, index_col=name_label)

data.index = data.index.map(lambda x: x.replace("CC0 1.0", "CC0 1.0"))
data = data[::-1] # reverse order

title = "Products totals and percentages"
Expand Down Expand Up @@ -213,6 +213,7 @@ def plot_latest_tools(args):
LOGGER.info(f"data file: {file_path.replace(PATHS['repo'], '.')}")
name_label = "CC legal tool"
data = shared.open_data_file(LOGGER, file_path, index_col=name_label)
data.index = data.index.map(lambda x: x.replace(" ", " "))
data.sort_values(name_label, ascending=False, inplace=True)

title = "Latest CC legal tools"
Expand Down Expand Up @@ -242,6 +243,9 @@ def plot_latest_tools(args):
image_path,
"The plot shows the latest Creative Commons (CC) legal tool totals and"
" percentages.",
"The latest tools include Licenses version 4.0 (CC BY 4.0, CC BY-NC"
" 4.0, CC BY-NC-ND 4.0, CC BY-NC-SA 4.0, CC-BY-ND 4.0, CC BY-SA 4.0),"
" CC0 1.0, and the Public Domain Mark (PDM 1.0).",
)


Expand All @@ -256,6 +260,7 @@ def plot_prior_tools(args):
LOGGER.info(f"data file: {file_path.replace(PATHS['repo'], '.')}")
name_label = "CC legal tool"
data = shared.open_data_file(LOGGER, file_path, index_col=name_label)
data.index = data.index.map(lambda x: x.replace(" ", " "))
data.sort_values(name_label, ascending=False, inplace=True)

title = "Prior CC legal tools"
Expand Down Expand Up @@ -284,7 +289,8 @@ def plot_prior_tools(args):
title,
image_path,
"The plot shows prior Creative Commons (CC) legal tool totals and"
" percentages. The unit names have been normalized (~~`CC BY-ND-NC`~~"
" percentages. Prior CC licenses include versions 1.0, 2.0, 2.1, 2.5,"
" and 3.0. The unit names have been normalized (~~`CC BY-ND-NC`~~"
" => `CC BY-NC-ND`).",
)

Expand All @@ -301,6 +307,7 @@ def plot_retired_tools(args):
LOGGER.info(f"data file: {file_path.replace(PATHS['repo'], '.')}")
name_label = "CC legal tool"
data = shared.open_data_file(LOGGER, file_path, index_col=name_label)
data.index = data.index.map(lambda x: x.replace(" ", " "))
data.sort_values(name_label, ascending=False, inplace=True)

title = "Retired CC legal tools"
Expand Down Expand Up @@ -379,14 +386,15 @@ def plot_countries_highest_usage(args):
SECTION_TITLE,
title,
image_path,
"The plot shows countries with the highest useage of the latest"
"The plot shows countries with the highest usage of the latest"
" Creative Commons (CC) legal tools.",
"The latest tools include Licenses version 4.0 (CC BY 4.0, CC BY-NC"
" 4.0, CC BY-NC-ND 4.0, CC BY-NC-SA 4.0, CC-BY-ND 4.0, CC BY-SA 4.0),"
" CC0 1.0, and the Public Domain Mark (PDM 1.0).\n"
"\n"
f"The complete data set indicates there are a total of {total_count}"
" online works using one of the latest CC legal tools.",
"The complete countries data set indicates there are a total of"
f" {total_count} online works using one of the latest CC legal tools."
" This conflicts with the languages data set, below.",
)


Expand Down Expand Up @@ -433,14 +441,15 @@ def plot_languages_highest_usage(args):
SECTION_TITLE,
title,
image_path,
"The plot shows the languages with the highest useage of the latest"
"The plot shows the languages with the highest usage of the latest"
" Creative Commons (CC) legal tools.",
"The latest tools include Licenses version 4.0 (CC BY 4.0, CC BY-NC"
" 4.0, CC BY-NC-ND 4.0, CC BY-NC-SA 4.0, CC-BY-ND 4.0, CC BY-SA 4.0),"
" CC0 1.0, and the Public Domain Mark (PDM 1.0).\n"
"\n"
f"The complete data set indicates there are a total of {total_count}"
" online works using one of the latest CC legal tools.",
"The complete languages data set indicates there are a total of"
f" {total_count} online works using one of the latest CC legal tools."
" This conflicts with the countries data set, above.",
)


Expand Down
Loading