# Which workbook economies failed the name join although they have complete consumption and
# territorial emissions and the World Bank has complete GDP for them under another name?
import csv, json, sys, openpyxl
bundle = sys.argv[1]
wb = openpyxl.load_workbook(bundle + "/data/national-fossil-2025.xlsx", read_only=True, data_only=True)
def sheet(name, header_row):
    rows = list(wb[name].iter_rows(values_only=True)); names = rows[header_row]; data = {}
    for r in rows[header_row + 1:]:
        if isinstance(r[0], (int, float)):
            data[int(r[0])] = {names[j]: r[j] for j in range(1, len(r)) if names[j] and isinstance(r[j], (int, float))}
    return data
terr = sheet("Territorial Emissions", 11); cons = sheet("Consumption Emissions", 8)
unmatched = [r["gcb_name"] for r in csv.DictReader(open(bundle + "/results/exclusions.csv")) if not r["iso3"]]
complete = [n for n in unmatched if all(n in cons.get(y, {}) and n in terr.get(y, {}) for y in range(2005, 2024))]
print("unmatched names:", len(unmatched))
print("unmatched with complete emissions 2005-2023:", complete)
gdp = json.load(open(bundle + "/data/worldbank-gdp.json"))[1]
countries = json.load(open(bundle + "/data/worldbank-countries.json"))[1]
economies = {c["id"]: c["name"] for c in countries if c["region"]["value"] != "Aggregates"}
full = {}
for row in gdp:
    iso = row["countryiso3code"]
    if iso in economies and row["value"] is not None and 2005 <= int(row["date"]) <= 2023:
        full.setdefault(iso, set()).add(int(row["date"]))
complete_gdp = {iso: economies[iso] for iso, ys in full.items() if len(ys) == 19}
matched = {r["iso3"] for r in csv.DictReader(open(bundle + "/results/crosswalk.csv"))}
print("World Bank economies with complete GDP 2005-2023 but not in the crosswalk:")
for iso, name in sorted(complete_gdp.items(), key=lambda x: x[1]):
    if iso not in matched: print("  ", iso, name)
