This notebook is available to use, reproduce, and validate the author results for ClimateKG. Source notebook: authors.ipynb

It analyzes author contributions in the ClimateKG graph using the same workflow as the Report Structure notebook and the ER reference in The Rock.

ER-model baseline (The Rock)

Expected author-side model:

  • Authors are items with P1 -> Q3998 (Author class).
  • Authors should have a stable ClimateKG Author ID via P20.
  • Author contributions are modeled with P27 from Author to Chapter.
  • Chapters are items with P1 -> Q6 (Chapter class).

The queries below first check live graph conformance to that model, then summarize contributions using duplicate-safe counting by P20 (not by item URI).

Query UI: https://climatekg.tibwiki.io/query/

SPARQL endpoint: https://climatekg.tibwiki.io/query/proxy/sparql

Show code
import os
import pandas as pd
from SPARQLWrapper import JSON
from IPython.display import display, Markdown

from wikibase_auth import DEFAULT_SPARQL_ENDPOINT, DEFAULT_WIKIBASE_URL, build_sparql_client

os.environ.setdefault("CLIMATEKG_SPARQL_USERNAME", "ckg")
os.environ.setdefault("CLIMATEKG_SPARQL_PASSWORD", "fairdata")

SPARQL_ENDPOINT = os.getenv("CLIMATEKG_SPARQL_ENDPOINT", DEFAULT_SPARQL_ENDPOINT)
WIKIBASE_URL = os.getenv("CLIMATEKG_WIKIBASE_URL", DEFAULT_WIKIBASE_URL)

ENTITY_NS = f"{WIKIBASE_URL.rstrip('/')}/entity/"
PROPERTY_NS = f"{WIKIBASE_URL.rstrip('/')}/prop/direct/"

print(f"Endpoint: {SPARQL_ENDPOINT}")
print(f"Entity namespace: {ENTITY_NS}")
print(f"Property namespace: {PROPERTY_NS}")
print("Libraries imported successfully")
'ckg'
'fairdata'
Endpoint: https://climatekg.tibwiki.io/query/proxy/sparql
Entity namespace: https://climatekg.tibwiki.io/entity/
Property namespace: https://climatekg.tibwiki.io/prop/direct/
Libraries imported successfully
Show code
author_query = f'''
PREFIX wd: <{ENTITY_NS}>
PREFIX wdt: <{PROPERTY_NS}>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>

SELECT ?author ?authorLabel ?authorId ?chapter ?chapterLabel
WHERE {{
  ?author wdt:P1 wd:Q3998 ;
          wdt:P20 ?authorId ;
          wdt:P27 ?chapter .
  ?chapter wdt:P1 wd:Q6 .

  OPTIONAL {{ ?author rdfs:label ?authorLabel . FILTER(LANG(?authorLabel) = "en") }}
  OPTIONAL {{ ?chapter rdfs:label ?chapterLabel . FILTER(LANG(?chapterLabel) = "en") }}
}}
ORDER BY ?authorLabel ?chapterLabel
'''

sparql = build_sparql_client(SPARQL_ENDPOINT)
sparql.setQuery(author_query)
sparql.setReturnFormat(JSON)
author_results = sparql.query().convert()
author_rows = author_results.get("results", {}).get("bindings", [])

authors_df = pd.DataFrame([{k: v["value"] for k, v in row.items()} for row in author_rows])

if authors_df.empty:
    display(Markdown("No author-chapter rows returned from live graph."))
else:
    authors_df["authorId"] = authors_df["authorId"].astype(str)
    authors_df["authorQID"] = authors_df["author"].astype(str).str.rsplit("/", n=1).str[-1]
    dedup_authors = authors_df.drop_duplicates(subset=["authorId"])
    chapter_counts = (
        authors_df.groupby("authorId", as_index=False)
        .agg(
            authorQID=("authorQID", "first"),
            authorLabel=("authorLabel", "first"),
            chapter_count=("chapter", "nunique"),
        )
        .sort_values(["chapter_count", "authorLabel"], ascending=[False, True])
    )

    print(f"Author-chapter rows returned: {len(authors_df)}")
    print(f"Distinct authors by P20: {dedup_authors['authorId'].nunique()}")
    print(f"Distinct chapter URIs linked from P27: {authors_df['chapter'].nunique()}")

    display(Markdown("### Sample author-chapter links"))
    display(authors_df[["authorQID", "authorLabel", "authorId", "chapterLabel"]].head(20))

    display(Markdown("### Top authors by unique chapter contributions"))
    display(chapter_counts.head(20))

def run_count(query_text):
    sparql.setQuery(query_text)
    sparql.setReturnFormat(JSON)
    result = sparql.query().convert()
    row = result.get("results", {}).get("bindings", [{}])[0]
    if not row:
        return 0
    first_key = next(iter(row.keys()))
    return int(row[first_key]["value"])

def run_table(query_text):
    sparql.setQuery(query_text)
    sparql.setReturnFormat(JSON)
    result = sparql.query().convert()
    rows = result.get("results", {}).get("bindings", [])
    return pd.DataFrame([{k: v["value"] for k, v in row.items()} for row in rows])

conformance_metrics = [
    (
        "Author items (Q3998)",
        f'''
        PREFIX wd: <{ENTITY_NS}>
        PREFIX wdt: <{PROPERTY_NS}>
        SELECT (COUNT(DISTINCT ?item) AS ?count) WHERE {{
          ?item wdt:P1 wd:Q3998 .
        }}
        '''
    ),
    (
        "Distinct ClimateKG Author IDs (P20)",
        f'''
        PREFIX wd: <{ENTITY_NS}>
        PREFIX wdt: <{PROPERTY_NS}>
        SELECT (COUNT(DISTINCT ?authorId) AS ?count) WHERE {{
          ?item wdt:P1 wd:Q3998 ; wdt:P20 ?authorId .
        }}
        '''
    ),
    (
        "Chapter items (Q6)",
        f'''
        PREFIX wd: <{ENTITY_NS}>
        PREFIX wdt: <{PROPERTY_NS}>
        SELECT (COUNT(DISTINCT ?chapter) AS ?count) WHERE {{
          ?chapter wdt:P1 wd:Q6 .
        }}
        '''
    ),
    (
        "Distinct P20 IDs with P27->Q6 links",
        f'''
        PREFIX wd: <{ENTITY_NS}>
        PREFIX wdt: <{PROPERTY_NS}>
        SELECT (COUNT(DISTINCT ?authorId) AS ?count) WHERE {{
          ?author wdt:P1 wd:Q3998 ;
                  wdt:P20 ?authorId ;
                  wdt:P27 ?chapter .
          ?chapter wdt:P1 wd:Q6 .
        }}
        '''
    ),
]

metrics_rows = []
for metric_name, metric_query in conformance_metrics:
    metrics_rows.append({
        "metric": metric_name,
        "value": run_count(metric_query),
    })

conformance_df = pd.DataFrame(metrics_rows)
display(Markdown("### ER-model conformance snapshot"))
display(conformance_df)

validation_queries = {
    "Authors missing P20": f'''
        PREFIX wd: <{ENTITY_NS}>
        PREFIX wdt: <{PROPERTY_NS}>
        PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
        SELECT ?author ?authorLabel
        WHERE {{
          ?author wdt:P1 wd:Q3998 .
          FILTER NOT EXISTS {{ ?author wdt:P20 ?authorId }}
          OPTIONAL {{ ?author rdfs:label ?authorLabel . FILTER(LANG(?authorLabel) = "en") }}
        }}
        ORDER BY ?author
    ''',
    "Authors missing any P27 contribution": f'''
        PREFIX wd: <{ENTITY_NS}>
        PREFIX wdt: <{PROPERTY_NS}>
        PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
        SELECT ?author ?authorLabel ?authorId
        WHERE {{
          ?author wdt:P1 wd:Q3998 ; wdt:P20 ?authorId .
          FILTER NOT EXISTS {{ ?author wdt:P27 ?target }}
          OPTIONAL {{ ?author rdfs:label ?authorLabel . FILTER(LANG(?authorLabel) = "en") }}
        }}
        ORDER BY ?authorId
    ''',
    "Authors with P27 target not typed Q6": f'''
        PREFIX wd: <{ENTITY_NS}>
        PREFIX wdt: <{PROPERTY_NS}>
        PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
        SELECT ?author ?authorLabel ?authorId ?target ?targetLabel ?targetClass ?targetClassLabel
        WHERE {{
          ?author wdt:P1 wd:Q3998 ; wdt:P20 ?authorId ; wdt:P27 ?target .
          FILTER NOT EXISTS {{ ?target wdt:P1 wd:Q6 }}
          OPTIONAL {{ ?author rdfs:label ?authorLabel . FILTER(LANG(?authorLabel) = "en") }}
          OPTIONAL {{ ?target rdfs:label ?targetLabel . FILTER(LANG(?targetLabel) = "en") }}
          OPTIONAL {{ ?target wdt:P1 ?targetClass . OPTIONAL {{ ?targetClass rdfs:label ?targetClassLabel . FILTER(LANG(?targetClassLabel) = "en") }} }}
        }}
        ORDER BY ?authorId ?target
    ''',
    "Duplicate P20 IDs assigned to multiple author items": f'''
        PREFIX wd: <{ENTITY_NS}>
        PREFIX wdt: <{PROPERTY_NS}>
        SELECT ?authorId (COUNT(DISTINCT ?author) AS ?authorItems)
        WHERE {{
          ?author wdt:P1 wd:Q3998 ; wdt:P20 ?authorId .
        }}
        GROUP BY ?authorId
        HAVING (COUNT(DISTINCT ?author) > 1)
        ORDER BY DESC(?authorItems) ?authorId
    ''',
}

display(Markdown("### Validation checks"))
validation_results = {}
for check_name, check_query in validation_queries.items():
    check_df = run_table(check_query)
    validation_results[check_name] = check_df
    display(Markdown(f"#### {check_name}"))
    print(f"Rows: {len(check_df)}")
    if len(check_df) == 0:
        display(Markdown("No issues found."))
    else:
        for col in ["author", "target", "targetClass"]:
            if col in check_df.columns:
                qid_col = f"{col}QID"
                check_df[qid_col] = check_df[col].astype(str).str.rsplit("/", n=1).str[-1]
        display(check_df.head(30))

p27_class_profile = run_table(f'''
    PREFIX wd: <{ENTITY_NS}>
    PREFIX wdt: <{PROPERTY_NS}>
    PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
    SELECT ?targetClass ?targetClassLabel (COUNT(DISTINCT ?author) AS ?authors) (COUNT(DISTINCT ?target) AS ?targets)
    WHERE {{
      ?author wdt:P1 wd:Q3998 ; wdt:P27 ?target .
      OPTIONAL {{ ?target wdt:P1 ?targetClass . OPTIONAL {{ ?targetClass rdfs:label ?targetClassLabel . FILTER(LANG(?targetClassLabel) = "en") }} }}
    }}
    GROUP BY ?targetClass ?targetClassLabel
    ORDER BY DESC(?authors) DESC(?targets)
''')

display(Markdown("#### P27 target class profile"))
display(p27_class_profile)

sample_author_qid = "Q4750"
statement_anatomy_query = f'''
    PREFIX wd: <{ENTITY_NS}>
    PREFIX p: <{WIKIBASE_URL.rstrip('/')}/prop/>
    PREFIX ps: <{WIKIBASE_URL.rstrip('/')}/prop/statement/>
    PREFIX pq: <{WIKIBASE_URL.rstrip('/')}/prop/qualifier/>
    PREFIX pr: <{WIKIBASE_URL.rstrip('/')}/prop/reference/>
    PREFIX prov: <http://www.w3.org/ns/prov#>
    PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
    SELECT ?author ?authorLabel ?statement ?chapter ?chapterLabel ?role ?refSource ?refAccessDate
    WHERE {{
      BIND(wd:{sample_author_qid} AS ?author)
      ?author p:P27 ?statement .
      ?statement ps:P27 ?chapter .
      OPTIONAL {{ ?author rdfs:label ?authorLabel . FILTER(LANG(?authorLabel) = "en") }}
      OPTIONAL {{ ?chapter rdfs:label ?chapterLabel . FILTER(LANG(?chapterLabel) = "en") }}
      OPTIONAL {{ ?statement pq:P28 ?role . }}
      OPTIONAL {{
        ?statement prov:wasDerivedFrom ?refNode .
        OPTIONAL {{ ?refNode pr:P17 ?refSource . }}
        OPTIONAL {{ ?refNode pr:P18 ?refAccessDate . }}
      }}
    }}
    ORDER BY ?chapterLabel
'''

statement_df = run_table(statement_anatomy_query)
display(Markdown(f"#### Statement anatomy (sample Author {sample_author_qid})"))
if statement_df.empty:
    display(Markdown("No statement-level rows found for the sample author."))
else:
    statement_df["authorQID"] = statement_df["author"].astype(str).str.rsplit("/", n=1).str[-1]
    statement_df["chapterQID"] = statement_df["chapter"].astype(str).str.rsplit("/", n=1).str[-1]
    if "refAccessDate" in statement_df.columns:
        statement_df["refAccessDate"] = statement_df["refAccessDate"].astype(str).str.replace("T00:00:00Z", "", regex=False)
    cols = ["authorQID", "authorLabel", "chapterQID", "chapterLabel", "role", "refSource", "refAccessDate"]
    existing_cols = [c for c in cols if c in statement_df.columns]
    display(statement_df[existing_cols].drop_duplicates())

# Notebook note block: capture key validation findings in plain language.
missing_p20_count = len(validation_results.get("Authors missing P20", pd.DataFrame()))
missing_p27_count = len(validation_results.get("Authors missing any P27 contribution", pd.DataFrame()))
non_q6_df = validation_results.get("Authors with P27 target not typed Q6", pd.DataFrame())
duplicate_p20_count = len(validation_results.get("Duplicate P20 IDs assigned to multiple author items", pd.DataFrame()))

non_q6_count = len(non_q6_df)
if non_q6_count > 0 and "target" in non_q6_df.columns:
    non_q6_targets = sorted(non_q6_df["target"].dropna().astype(str).unique().tolist())
    non_q6_target_qids = sorted([uri.rsplit("/", 1)[-1] for uri in non_q6_targets])
else:
    non_q6_target_qids = []

notes_lines = [
    "### Validation notes",
    f"- Authors missing P20: {missing_p20_count}",
    f"- Authors missing P27 contributions: {missing_p27_count}",
    f"- Duplicate P20 IDs on multiple author items: {duplicate_p20_count}",
    f"- P27 links to non-Q6 targets: {non_q6_count}",
]

if non_q6_target_qids:
    notes_lines.append(f"- Non-Q6 target QIDs observed: {', '.join(non_q6_target_qids)}")

if not statement_df.empty:
    notes_lines.extend([
        "",
        f"Statement anatomy check ({sample_author_qid}) confirms claim-level metadata:",
        "- Main claim: P27 (contributed to chapter)",
        "- Author-side relationship role: P28 (role, e.g. Lead Author)",
        "- References: P17 (reference URL) and P18 (date accessed)",
    ])

notes_lines.extend([
    "",
    "Author property pattern observed from item-level inspection (Q4750):",
    "- Structural typing and relation properties: P1 (instance of), P27 (contributed to)",
    "- Identifier property: P20 (ClimateKG Author ID)",
    "- Biographic profile properties: P21 (last name), P22 (first name), P23 (gender), P24 (citizenship), P25 (country of residence), P26 (affiliation)",
])

display(Markdown("\n".join(notes_lines)))
Author-chapter rows returned: 1150
Distinct authors by P20: 921
Distinct chapter URIs linked from P27: 74
authorQID authorLabel authorId chapterLabel
0 Q4750 Adam Mohammed Sebbit AU0752 Mitigation Pathways Compatible with Long-term ...
1 Q4833 Adelle Thomas AU0835 Cities and Settlements by the Sea
2 Q4833 Adelle Thomas AU0835 Impacts of 1.5°C Global Warming on Natural and...
3 Q4833 Adelle Thomas AU0835 Key Risks across Sectors and Regions
4 Q4592 Aditi Mukherji AU0594 High Mountain Areas
5 Q4592 Aditi Mukherji AU0594 Longer Report
6 Q4592 Aditi Mukherji AU0594 Summary for Policymakers
7 Q4592 Aditi Mukherji AU0594 Water
8 Q4007 Adolf Acquaye AU0009 Industry
9 Q4484 Adrian Leip AU0486 Cross-sectoral Perspectives
10 Q4785 Adrian Spence AU0787 Framing and context
11 Q4287 Adugna Gemeda AU0289 Africa
12 Q4739 Agus Pratama Sari AU0741 International Cooperation
13 Q4253 Aidan Farrell AU0255 Food, Fibre and Other Ecosystem Products
14 Q4253 Aidan Farrell AU0255 Tropical Forests
15 Q4774 Aimée Slangen AU0776 Longer Report
16 Q4774 Aimée Slangen AU0776 Ocean, Cryosphere and Sea Level Change
17 Q4774 Aimée Slangen AU0776 Summary for Policymakers
18 Q4769 Ajay Kumar Singh AU0771 Energy Systems
19 Q4441 Akio Kitoh AU0443 Asia

Top authors by unique chapter contributions

authorId authorQID authorLabel chapter_count
282 AU0284 Q4282 Matthias Garschagen 5
588 AU0594 Q4592 Aditi Mukherji 4
3 AU0004 Q4002 Amjad Abdulla 4
696 AU0704 Q4702 Aromar Revi 4
341 AU0343 Q4341 Bronwyn Hayward 4
193 AU0195 Q4193 Fatima Denton 4
718 AU0727 Q4725 Joyashree Roy 4
108 AU0109 Q4107 Katherine Calvin 4
917 AU0929 Q4927 Zinta Zommers 4
825 AU0835 Q4833 Adelle Thomas 3
767 AU0776 Q4774 Aimée Slangen 3
719 AU0728 Q4726 Alexander Ruane 3
569 AU0575 Q4573 Alisher Mirzabaev 3
260 AU0262 Q4260 Andreas Fischlin 3
674 AU0682 Q4680 Anjal Prakash 3
809 AU0818 Q4816 Anna Amelia Sörensson 3
792 AU0801 Q4799 Avelino Gumersindo Suarez Rodriguez 3
210 AU0212 Q4210 Aïda Diongue-Niang 3
300 AU0302 Q4300 Bruce Glavovic 3
12 AU0013 Q4011 Carolina Adler 3

ER-model conformance snapshot

metric value
0 Author items (Q3998) 932
1 Distinct ClimateKG Author IDs (P20) 932
2 Chapter items (Q6) 87
3 Distinct P20 IDs with P27->Q6 links 921

Validation checks

Authors missing P20

Rows: 0

No issues found.

Authors missing any P27 contribution

Rows: 0

No issues found.

Authors with P27 target not typed Q6

Rows: 14
author authorLabel authorId target authorQID targetQID
0 https://climatekg.tibwiki.io/entity/Q4142 Francis Chiew AU0144 https://climatekg.tibwiki.io/entity/Q128 Q4142 Q128
1 https://climatekg.tibwiki.io/entity/Q4165 Mark Costello AU0167 https://climatekg.tibwiki.io/entity/Q128 Q4165 Q128
2 https://climatekg.tibwiki.io/entity/Q4345 Kevin Hennessy AU0347 https://climatekg.tibwiki.io/entity/Q128 Q4345 Q128
3 https://climatekg.tibwiki.io/entity/Q4360 Ove Hoegh-Guldberg AU0362 https://climatekg.tibwiki.io/entity/Q128 Q4360 Q128
4 https://climatekg.tibwiki.io/entity/Q4474 Nina Lansbury AU0476 https://climatekg.tibwiki.io/entity/Q128 Q4474 Q128
5 https://climatekg.tibwiki.io/entity/Q4476 Judy Lawrence AU0478 https://climatekg.tibwiki.io/entity/Q128 Q4476 Q128
6 https://climatekg.tibwiki.io/entity/Q4518 Brendan Mackey AU0520 https://climatekg.tibwiki.io/entity/Q128 Q4518 Q128
7 https://climatekg.tibwiki.io/entity/Q4620 Uday Bhaskar Nidumolu AU0622 https://climatekg.tibwiki.io/entity/Q128 Q4620 Q128
8 https://climatekg.tibwiki.io/entity/Q4658 Gretta Pecl AU0660 https://climatekg.tibwiki.io/entity/Q128 Q4658 Q128
9 https://climatekg.tibwiki.io/entity/Q4707 Lauren Rickards AU0709 https://climatekg.tibwiki.io/entity/Q128 Q4707 Q128
10 https://climatekg.tibwiki.io/entity/Q4825 Nigel Tapper AU0827 https://climatekg.tibwiki.io/entity/Q128 Q4825 Q128
11 https://climatekg.tibwiki.io/entity/Q4894 Alistair Woodward AU0896 https://climatekg.tibwiki.io/entity/Q128 Q4894 Q128
12 https://climatekg.tibwiki.io/entity/Q4896 David Wratt AU0898 https://climatekg.tibwiki.io/entity/Q128 Q4896 Q128
13 https://climatekg.tibwiki.io/entity/Q4897 Anita Wreford AU0899 https://climatekg.tibwiki.io/entity/Q128 Q4897 Q128

Duplicate P20 IDs assigned to multiple author items

Rows: 0

No issues found.

P27 target class profile

targetClass targetClassLabel authors targets
0 https://climatekg.tibwiki.io/entity/Q6 Chapter 921 74
1 https://climatekg.tibwiki.io/entity/Q5 Text Division 19 1
2 NaN NaN 14 1

Statement anatomy (sample Author Q4750)

authorQID authorLabel chapterQID chapterLabel role refSource refAccessDate
0 Q4750 Adam Mohammed Sebbit Q158 Mitigation Pathways Compatible with Long-term ... Lead Author https://apps.ipcc.ch/report/authors/ 2026-03-18

Validation notes

  • Authors missing P20: 0
  • Authors missing P27 contributions: 0
  • Duplicate P20 IDs on multiple author items: 0
  • P27 links to non-Q6 targets: 14
  • Non-Q6 target QIDs observed: Q128

Statement anatomy check (Q4750) confirms claim-level metadata: - Main claim: P27 (contributed to chapter) - Author-side relationship role: P28 (role, e.g. Lead Author) - References: P17 (reference URL) and P18 (date accessed)

Author property pattern observed from item-level inspection (Q4750): - Structural typing and relation properties: P1 (instance of), P27 (contributed to) - Identifier property: P20 (ClimateKG Author ID) - Biographic profile properties: P21 (last name), P22 (first name), P23 (gender), P24 (citizenship), P25 (country of residence), P26 (affiliation)

Interpretation

This notebook now follows the same structure used in Report Structure:

  1. Start from the ER baseline in The Rock (Q3998 Author, Q6 Chapter, P20 author ID, P27 contributed-to relation).
  2. Query the live graph for author-chapter links.
  3. Report duplicate-safe metrics by deduplicating authors on P20.
  4. Run validation checks to flag modeling exceptions.

The key lesson from the structure notebook also applies here: we should not infer completeness from raw item counts alone. Where duplicate imports exist, author-level counts must be grounded on distinct P20 values, while relationships (P27) are validated against chapters typed as Q6.

Statement anatomy

The validation cell now includes a statement-level anatomy view for sample author Q4750, showing how one P27 contribution is represented with:

  • Main claim value (P27 -> chapter item)
  • Author-side relationship role (P28 role, e.g. Lead Author)
  • References (P17 reference URL and P18 date accessed)

This helps verify that the graph carries provenance at claim level, not only at item level.

Author property usage learned from item inspection

From the concrete author item Q4750 and the broader validation checks, the current Author pattern uses the following fields.

Property Label / Purpose Role in Author model
P1 Instance of Class typing (Q3998 Author)
P20 ClimateKG Author ID Stable author identifier used for deduplication
P27 contributed to chapter Link from Author to contribution target (expected class Q6 Chapter)
P28 author role on chapter contribution Author-side contribution role qualifier used on the P27 statement
P21 last name Biographic/profile attribute
P22 first name Biographic/profile attribute
P23 gender Biographic/profile attribute
P24 citizenship Biographic/profile attribute
P25 country of residence Biographic/profile attribute
P26 affiliation Biographic/profile attribute

Bibliographic questions

From the GitHub issue: https://github.com/TIBHannover/ClimateKG-Data-Bench/issues/1

  1. How many distinct authors contributed to each of the 7 reports?
  2. Which countries of residence are most represented among AR6 authors, and how does this differ between WGI, WGII and WGIII?
  3. What is the share of female authors per working group report, and which chapters have the lowest female representation?
  4. Which authors contributed to chapters in more than one working group report?
  5. Which two authors have co-authored the most chapters together?
  6. Which institutions have the most authors across AR6?

Question 1

How many distinct authors contributed to each of the 7 reports? This cell uses the live SPARQL endpoint and counts distinct author IDs (P20) for each report.

Show code
query = f'''
PREFIX wd: <{ENTITY_NS}>
PREFIX wdt: <{PROPERTY_NS}>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?report ?reportLabel (COUNT(DISTINCT ?authorId) AS ?distinctAuthors)
WHERE {{
  ?author wdt:P1 wd:Q3998 ;
          wdt:P20 ?authorId ;
          wdt:P27 ?chapter .
  ?chapter wdt:P3+ ?report .
  ?report wdt:P1 wd:Q4 ;
          rdfs:label ?reportLabel .
  FILTER(LANG(?reportLabel) = "en")
}}
GROUP BY ?report ?reportLabel
ORDER BY ?reportLabel
'''

display(Markdown("**SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)**"))
display(Markdown("```sparql\n" + query + "\n```"))
report_df = run_table(query).sort_values("reportLabel").reset_index(drop=True)
report_df["reportQID"] = report_df["report"].astype(str).str.rsplit("/", n=1).str[-1]
display(report_df[["reportQID", "reportLabel", "distinctAuthors"]])

SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)


PREFIX wd: <https://climatekg.tibwiki.io/entity/>
PREFIX wdt: <https://climatekg.tibwiki.io/prop/direct/>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?report ?reportLabel (COUNT(DISTINCT ?authorId) AS ?distinctAuthors)
WHERE {
  ?author wdt:P1 wd:Q3998 ;
          wdt:P20 ?authorId ;
          wdt:P27 ?chapter .
  ?chapter wdt:P3+ ?report .
  ?report wdt:P1 wd:Q4 ;
          rdfs:label ?reportLabel .
  FILTER(LANG(?reportLabel) = "en")
}
GROUP BY ?report ?reportLabel
ORDER BY ?reportLabel
reportQID reportLabel distinctAuthors
0 Q189 Climate Change 2023: Synthesis Report. Contrib... 30
1 Q35 Special Report: Climate Change and Land 107
2 Q10 Special Report: Global Warming of 1.5°C 91
3 Q57 Special Report: The Ocean and Cryosphere in a ... 103
4 Q77 Working Group I: Climate Change 2021 – The Phy... 234
5 Q106 Working Group II: Climate Change 2022 – Impact... 257
6 Q150 Working Group III: Climate Change 2022 – Mitig... 239

Question 2

Which countries of residence are most represented among AR6 authors, and how does this differ between WGI, WGII and WGIII? This cell uses the live SPARQL endpoint and counts distinct authors (P20) by country (P25).

Show code
query = f'''
PREFIX wd: <{ENTITY_NS}>
PREFIX wdt: <{PROPERTY_NS}>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?workingGroupLabel ?country (COUNT(DISTINCT ?authorId) AS ?authors)
WHERE {{
  ?author wdt:P1 wd:Q3998 ;
          wdt:P20 ?authorId ;
          wdt:P25 ?country ;
          wdt:P27 ?chapter .
  ?chapter wdt:P3+ ?report .
  ?report wdt:P1 wd:Q4 ;
          rdfs:label ?reportLabel .
  FILTER(LANG(?reportLabel) = "en")
  BIND(IF(CONTAINS(?reportLabel, "Working Group I:"), "Working Group I",
       IF(CONTAINS(?reportLabel, "Working Group II:"), "Working Group II",
       IF(CONTAINS(?reportLabel, "Working Group III:"), "Working Group III", "Other"))) AS ?workingGroupLabel)
  FILTER(?workingGroupLabel != "Other")
}}
GROUP BY ?workingGroupLabel ?country
ORDER BY ?workingGroupLabel DESC(?authors) ?country
'''

display(Markdown("**SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)**"))
display(Markdown("```sparql\n" + query + "\n```"))
country_df = run_table(query)
country_top = country_df.sort_values(["workingGroupLabel", "authors", "country"], ascending=[True, False, True]).groupby("workingGroupLabel", as_index=False).head(10)
display(country_top[["workingGroupLabel", "country", "authors"]])

SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)


PREFIX wd: <https://climatekg.tibwiki.io/entity/>
PREFIX wdt: <https://climatekg.tibwiki.io/prop/direct/>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?workingGroupLabel ?country (COUNT(DISTINCT ?authorId) AS ?authors)
WHERE {
  ?author wdt:P1 wd:Q3998 ;
          wdt:P20 ?authorId ;
          wdt:P25 ?country ;
          wdt:P27 ?chapter .
  ?chapter wdt:P3+ ?report .
  ?report wdt:P1 wd:Q4 ;
          rdfs:label ?reportLabel .
  FILTER(LANG(?reportLabel) = "en")
  BIND(IF(CONTAINS(?reportLabel, "Working Group I:"), "Working Group I",
       IF(CONTAINS(?reportLabel, "Working Group II:"), "Working Group II",
       IF(CONTAINS(?reportLabel, "Working Group III:"), "Working Group III", "Other"))) AS ?workingGroupLabel)
  FILTER(?workingGroupLabel != "Other")
}
GROUP BY ?workingGroupLabel ?country
ORDER BY ?workingGroupLabel DESC(?authors) ?country
workingGroupLabel country authors
6 Working Group I Canada 8
7 Working Group I Germany 8
8 Working Group I Norway 7
9 Working Group I Brazil 6
10 Working Group I India 6
11 Working Group I Italy 6
12 Working Group I Argentina 5
13 Working Group I Netherlands 5
14 Working Group I Republic of Korea 5
15 Working Group I South Africa 5
71 Working Group II Canada 9
72 Working Group II Mexico 8
73 Working Group II South Africa 8
74 Working Group II France 7
75 Working Group II Norway 7
76 Working Group II Switzerland 7
77 Working Group II Netherlands 6
78 Working Group II Brazil 5
79 Working Group II Spain 5
80 Working Group II Tanzania 5
133 Working Group III Brazil 9
134 Working Group III Austria 7
135 Working Group III France 7
136 Working Group III Italy 7
137 Working Group III Canada 6
138 Working Group III Netherlands 6
139 Working Group III Norway 6
140 Working Group III Argentina 4
141 Working Group III Denmark 4
142 Working Group III Spain 4

Question 3

What is the share of female authors per working group report, and which chapters have the lowest female representation? This cell uses the live SPARQL endpoint and deduplicates authors by P20 before calculating shares.

Show code
query = f'''
PREFIX wd: <{ENTITY_NS}>
PREFIX wdt: <{PROPERTY_NS}>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?workingGroupLabel (COUNT(DISTINCT ?authorId) AS ?authors) (SUM(?isFemale) AS ?femaleAuthors) (SUM(?isMale) AS ?maleAuthors)
WHERE {{
  {{
    SELECT DISTINCT ?workingGroupLabel ?authorId ?gender
    WHERE {{
      ?author wdt:P1 wd:Q3998 ;
              wdt:P20 ?authorId ;
              wdt:P23 ?gender ;
              wdt:P27 ?chapter .
      ?chapter wdt:P3+ ?report .
      ?report wdt:P1 wd:Q4 ;
              rdfs:label ?reportLabel .
      FILTER(LANG(?reportLabel) = "en")
      BIND(IF(CONTAINS(?reportLabel, "Working Group I:"), "Working Group I",
           IF(CONTAINS(?reportLabel, "Working Group II:"), "Working Group II",
           IF(CONTAINS(?reportLabel, "Working Group III:"), "Working Group III", "Other"))) AS ?workingGroupLabel)
      FILTER(?workingGroupLabel != "Other")
    }}
  }}
  BIND(IF(LCASE(STR(?gender)) = "f", 1, 0) AS ?isFemale)
  BIND(IF(LCASE(STR(?gender)) = "m", 1, 0) AS ?isMale)
}}
GROUP BY ?workingGroupLabel
ORDER BY ?workingGroupLabel
'''

display(Markdown("**SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)**"))
display(Markdown("```sparql\n" + query + "\n```"))
female_report_df = run_table(query)
female_report_df["femaleSharePct"] = (female_report_df["femaleAuthors"].astype(float) / female_report_df["authors"].astype(float) * 100).round(1)
female_report_df["maleSharePct"] = (female_report_df["maleAuthors"].astype(float) / female_report_df["authors"].astype(float) * 100).round(1)
display(female_report_df[["workingGroupLabel", "authors", "femaleAuthors", "maleAuthors", "femaleSharePct", "maleSharePct"]])

chapter_query = f'''
PREFIX wd: <{ENTITY_NS}>
PREFIX wdt: <{PROPERTY_NS}>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?workingGroupLabel ?chapter ?chapterLabel (COUNT(DISTINCT ?authorId) AS ?authors) (SUM(?isFemale) AS ?femaleAuthors) (SUM(?isMale) AS ?maleAuthors)
WHERE {{
  {{
    SELECT DISTINCT ?workingGroupLabel ?chapter ?chapterLabel ?authorId ?gender
    WHERE {{
      ?author wdt:P1 wd:Q3998 ;
              wdt:P20 ?authorId ;
              wdt:P23 ?gender ;
              wdt:P27 ?chapter .
      ?chapter wdt:P1 wd:Q6 ;
               rdfs:label ?chapterLabel .
      FILTER(LANG(?chapterLabel) = "en")
      ?chapter wdt:P3+ ?report .
      ?report wdt:P1 wd:Q4 ;
              rdfs:label ?reportLabel .
      FILTER(LANG(?reportLabel) = "en")
      BIND(IF(CONTAINS(?reportLabel, "Working Group I:"), "Working Group I",
           IF(CONTAINS(?reportLabel, "Working Group II:"), "Working Group II",
           IF(CONTAINS(?reportLabel, "Working Group III:"), "Working Group III", "Other"))) AS ?workingGroupLabel)
      FILTER(?workingGroupLabel != "Other")
    }}
  }}
  BIND(IF(LCASE(STR(?gender)) = "f", 1, 0) AS ?isFemale)
  BIND(IF(LCASE(STR(?gender)) = "m", 1, 0) AS ?isMale)
}}
GROUP BY ?workingGroupLabel ?chapter ?chapterLabel
ORDER BY ASC((?femaleAuthors / ?authors)) ?workingGroupLabel ?chapterLabel
'''

display(Markdown("**SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)**"))
display(Markdown("```sparql\n" + chapter_query + "\n```"))
female_chapter_df = run_table(chapter_query)
female_chapter_df["femaleSharePct"] = (female_chapter_df["femaleAuthors"].astype(float) / female_chapter_df["authors"].astype(float) * 100).round(1)
female_chapter_df["maleSharePct"] = (female_chapter_df["maleAuthors"].astype(float) / female_chapter_df["authors"].astype(float) * 100).round(1)
lowest_female = female_chapter_df.sort_values(["femaleSharePct", "workingGroupLabel", "chapterLabel"], ascending=[True, True, True]).head(10)
display(lowest_female[["workingGroupLabel", "chapterLabel", "authors", "femaleAuthors", "maleAuthors", "femaleSharePct", "maleSharePct"]])

SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)


PREFIX wd: <https://climatekg.tibwiki.io/entity/>
PREFIX wdt: <https://climatekg.tibwiki.io/prop/direct/>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?workingGroupLabel (COUNT(DISTINCT ?authorId) AS ?authors) (SUM(?isFemale) AS ?femaleAuthors) (SUM(?isMale) AS ?maleAuthors)
WHERE {
  {
    SELECT DISTINCT ?workingGroupLabel ?authorId ?gender
    WHERE {
      ?author wdt:P1 wd:Q3998 ;
              wdt:P20 ?authorId ;
              wdt:P23 ?gender ;
              wdt:P27 ?chapter .
      ?chapter wdt:P3+ ?report .
      ?report wdt:P1 wd:Q4 ;
              rdfs:label ?reportLabel .
      FILTER(LANG(?reportLabel) = "en")
      BIND(IF(CONTAINS(?reportLabel, "Working Group I:"), "Working Group I",
           IF(CONTAINS(?reportLabel, "Working Group II:"), "Working Group II",
           IF(CONTAINS(?reportLabel, "Working Group III:"), "Working Group III", "Other"))) AS ?workingGroupLabel)
      FILTER(?workingGroupLabel != "Other")
    }
  }
  BIND(IF(LCASE(STR(?gender)) = "f", 1, 0) AS ?isFemale)
  BIND(IF(LCASE(STR(?gender)) = "m", 1, 0) AS ?isMale)
}
GROUP BY ?workingGroupLabel
ORDER BY ?workingGroupLabel
workingGroupLabel authors femaleAuthors maleAuthors femaleSharePct maleSharePct
0 Working Group I 234 66 168 28.2 71.8
1 Working Group II 257 106 151 41.2 58.8
2 Working Group III 239 77 162 32.2 67.8

SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)


PREFIX wd: <https://climatekg.tibwiki.io/entity/>
PREFIX wdt: <https://climatekg.tibwiki.io/prop/direct/>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?workingGroupLabel ?chapter ?chapterLabel (COUNT(DISTINCT ?authorId) AS ?authors) (SUM(?isFemale) AS ?femaleAuthors) (SUM(?isMale) AS ?maleAuthors)
WHERE {
  {
    SELECT DISTINCT ?workingGroupLabel ?chapter ?chapterLabel ?authorId ?gender
    WHERE {
      ?author wdt:P1 wd:Q3998 ;
              wdt:P20 ?authorId ;
              wdt:P23 ?gender ;
              wdt:P27 ?chapter .
      ?chapter wdt:P1 wd:Q6 ;
               rdfs:label ?chapterLabel .
      FILTER(LANG(?chapterLabel) = "en")
      ?chapter wdt:P3+ ?report .
      ?report wdt:P1 wd:Q4 ;
              rdfs:label ?reportLabel .
      FILTER(LANG(?reportLabel) = "en")
      BIND(IF(CONTAINS(?reportLabel, "Working Group I:"), "Working Group I",
           IF(CONTAINS(?reportLabel, "Working Group II:"), "Working Group II",
           IF(CONTAINS(?reportLabel, "Working Group III:"), "Working Group III", "Other"))) AS ?workingGroupLabel)
      FILTER(?workingGroupLabel != "Other")
    }
  }
  BIND(IF(LCASE(STR(?gender)) = "f", 1, 0) AS ?isFemale)
  BIND(IF(LCASE(STR(?gender)) = "m", 1, 0) AS ?isMale)
}
GROUP BY ?workingGroupLabel ?chapter ?chapterLabel
ORDER BY ASC((?femaleAuthors / ?authors)) ?workingGroupLabel ?chapterLabel
workingGroupLabel chapterLabel authors femaleAuthors maleAuthors femaleSharePct maleSharePct
0 Working Group II Tropical Forests 8 1 7 12.5 87.5
1 Working Group I The Earth’s Energy Budget, Climate Feedbacks a... 15 2 13 13.3 86.7
2 Working Group III Cross-sectoral Perspectives 13 2 11 15.4 84.6
3 Working Group III Industry 11 2 9 18.2 81.8
4 Working Group I Global Carbon and Other Biogeochemical Cycles ... 19 4 15 21.1 78.9
5 Working Group I Linking Global to Regional Climate Change 19 4 15 21.1 78.9
6 Working Group II Point of Departure and Key Concepts 14 3 11 21.4 78.6
7 Working Group III Energy Systems 14 3 11 21.4 78.6
8 Working Group I Climate Change Information for Regional Impact... 18 4 14 22.2 77.8
9 Working Group I Future Global Climate: Scenario-based Projecti... 18 4 14 22.2 77.8

Question 4

Which authors contributed to chapters in more than one working group report? This cell uses the live SPARQL endpoint and deduplicates authors by P20 before counting working groups.

Show code
query = f'''
PREFIX wd: <{ENTITY_NS}>
PREFIX wdt: <{PROPERTY_NS}>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?authorId ?authorLabel (COUNT(DISTINCT ?workingGroup) AS ?workingGroups) (GROUP_CONCAT(DISTINCT ?workingGroupLabel; separator=", ") AS ?groupLabels)
WHERE {{
  ?author wdt:P1 wd:Q3998 ;
          wdt:P20 ?authorId ;
          wdt:P27 ?chapter .
  OPTIONAL {{ ?author rdfs:label ?authorLabel . FILTER(LANG(?authorLabel) = "en") }}
  ?chapter wdt:P3+ ?report .
  ?report wdt:P1 wd:Q4 ;
          rdfs:label ?reportLabel .
  FILTER(LANG(?reportLabel) = "en")
  BIND(IF(CONTAINS(?reportLabel, "Working Group I:"), "WGI",
       IF(CONTAINS(?reportLabel, "Working Group II:"), "WGII",
       IF(CONTAINS(?reportLabel, "Working Group III:"), "WGIII", "Other"))) AS ?workingGroup)
  FILTER(?workingGroup != "Other")
  BIND(IF(?workingGroup = "WGI", "Working Group I", IF(?workingGroup = "WGII", "Working Group II", "Working Group III")) AS ?workingGroupLabel)
}}
GROUP BY ?authorId ?authorLabel
HAVING (COUNT(DISTINCT ?workingGroup) > 1)
ORDER BY DESC(?workingGroups) ?authorLabel
'''

display(Markdown("**SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)**"))
display(Markdown("```sparql\n" + query + "\n```"))
multi_wg_df = run_table(query)
display(multi_wg_df[["authorLabel", "authorId", "workingGroups", "groupLabels"]])

SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)


PREFIX wd: <https://climatekg.tibwiki.io/entity/>
PREFIX wdt: <https://climatekg.tibwiki.io/prop/direct/>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?authorId ?authorLabel (COUNT(DISTINCT ?workingGroup) AS ?workingGroups) (GROUP_CONCAT(DISTINCT ?workingGroupLabel; separator=", ") AS ?groupLabels)
WHERE {
  ?author wdt:P1 wd:Q3998 ;
          wdt:P20 ?authorId ;
          wdt:P27 ?chapter .
  OPTIONAL { ?author rdfs:label ?authorLabel . FILTER(LANG(?authorLabel) = "en") }
  ?chapter wdt:P3+ ?report .
  ?report wdt:P1 wd:Q4 ;
          rdfs:label ?reportLabel .
  FILTER(LANG(?reportLabel) = "en")
  BIND(IF(CONTAINS(?reportLabel, "Working Group I:"), "WGI",
       IF(CONTAINS(?reportLabel, "Working Group II:"), "WGII",
       IF(CONTAINS(?reportLabel, "Working Group III:"), "WGIII", "Other"))) AS ?workingGroup)
  FILTER(?workingGroup != "Other")
  BIND(IF(?workingGroup = "WGI", "Working Group I", IF(?workingGroup = "WGII", "Working Group II", "Working Group III")) AS ?workingGroupLabel)
}
GROUP BY ?authorId ?authorLabel
HAVING (COUNT(DISTINCT ?workingGroup) > 1)
ORDER BY DESC(?workingGroups) ?authorLabel
authorLabel authorId workingGroups groupLabels
0 Linda Mearns AU0554 2 Working Group II, Working Group I
1 Patricia Romero Lankao AU0721 2 Working Group III, Working Group II

Question 5

Which two authors have co-authored the most chapters together? This cell uses the live SPARQL endpoint and counts shared chapters per author pair.

Show code
query = f'''
PREFIX wd: <{ENTITY_NS}>
PREFIX wdt: <{PROPERTY_NS}>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?author1Id ?author1Label ?author2Id ?author2Label (COUNT(DISTINCT ?chapter) AS ?sharedChapters)
WHERE {{
  {{
    SELECT DISTINCT ?author1 ?author1Id ?author1Label ?author2 ?author2Id ?author2Label ?chapter
    WHERE {{
      ?author1 wdt:P1 wd:Q3998 ; wdt:P20 ?author1Id ; wdt:P27 ?chapter .
      ?author2 wdt:P1 wd:Q3998 ; wdt:P20 ?author2Id ; wdt:P27 ?chapter .
      FILTER(?author1Id < ?author2Id)
      OPTIONAL {{ ?author1 rdfs:label ?author1Label . FILTER(LANG(?author1Label) = "en") }}
      OPTIONAL {{ ?author2 rdfs:label ?author2Label . FILTER(LANG(?author2Label) = "en") }}
    }}
  }}
}}
GROUP BY ?author1Id ?author1Label ?author2Id ?author2Label
ORDER BY DESC(?sharedChapters) ?author1Label ?author2Label
LIMIT 20
'''

display(Markdown("**SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)**"))
display(Markdown("```sparql\n" + query + "\n```"))
pairs_df = run_table(query)
if not pairs_df.empty:
    pairs_df = pairs_df.sort_values(["sharedChapters", "author1Label", "author2Label"], ascending=[False, True, True]).reset_index(drop=True)
    top_pairs = pairs_df[pairs_df["sharedChapters"] == pairs_df["sharedChapters"].max()]
    display(top_pairs[["author1Label", "author2Label", "sharedChapters"]])
else:
    display(pairs_df)

SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)


PREFIX wd: <https://climatekg.tibwiki.io/entity/>
PREFIX wdt: <https://climatekg.tibwiki.io/prop/direct/>
PREFIX rdfs: <http://www.w3.org/2000/01/rdf-schema#>
SELECT ?author1Id ?author1Label ?author2Id ?author2Label (COUNT(DISTINCT ?chapter) AS ?sharedChapters)
WHERE {
  {
    SELECT DISTINCT ?author1 ?author1Id ?author1Label ?author2 ?author2Id ?author2Label ?chapter
    WHERE {
      ?author1 wdt:P1 wd:Q3998 ; wdt:P20 ?author1Id ; wdt:P27 ?chapter .
      ?author2 wdt:P1 wd:Q3998 ; wdt:P20 ?author2Id ; wdt:P27 ?chapter .
      FILTER(?author1Id < ?author2Id)
      OPTIONAL { ?author1 rdfs:label ?author1Label . FILTER(LANG(?author1Label) = "en") }
      OPTIONAL { ?author2 rdfs:label ?author2Label . FILTER(LANG(?author2Label) = "en") }
    }
  }
}
GROUP BY ?author1Id ?author1Label ?author2Id ?author2Label
ORDER BY DESC(?sharedChapters) ?author1Label ?author2Label
LIMIT 20
author1Label author2Label sharedChapters
0 Aïda Diongue-Niang Malte Meinshausen 3
1 Bronwyn Hayward Joyashree Roy 3
2 Christopher Jones June-yi Lee 3
3 David Dodman Bronwyn Hayward 3
4 Gerhard Krinner Aimée Slangen 3
5 Katherine Calvin Detlef Van Vuuren 3

Question 6

Which institutions have the most authors across AR6? This cell uses the live SPARQL endpoint and counts distinct author IDs (P20) per affiliation (P26).

Show code
query = f'''
PREFIX wd: <{ENTITY_NS}>
PREFIX wdt: <{PROPERTY_NS}>
SELECT ?institution (COUNT(DISTINCT ?authorId) AS ?authors)
WHERE {{
  ?author wdt:P1 wd:Q3998 ;
          wdt:P20 ?authorId ;
          wdt:P26 ?institution .
}}
GROUP BY ?institution
ORDER BY DESC(?authors) ?institution
LIMIT 20
'''

display(Markdown("**SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)**"))
display(Markdown("```sparql\n" + query + "\n```"))
institution_df = run_table(query).rename(columns={"institution": "Institution", "authors": "Authors"})
display(institution_df)

SPARQL query (copy/paste into ClimateKG Query UI: https://climatekg.tibwiki.io/query/)


PREFIX wd: <https://climatekg.tibwiki.io/entity/>
PREFIX wdt: <https://climatekg.tibwiki.io/prop/direct/>
SELECT ?institution (COUNT(DISTINCT ?authorId) AS ?authors)
WHERE {
  ?author wdt:P1 wd:Q3998 ;
          wdt:P20 ?authorId ;
          wdt:P26 ?institution .
}
GROUP BY ?institution
ORDER BY DESC(?authors) ?institution
LIMIT 20
Institution Authors
0 University of Cape Town 6
1 University of Leeds 6
2 Institute of Atmospheric Physics, Chinese Acad... 5
3 International Institute for Applied Systems An... 5
4 Met Office Hadley Centre 5
5 Tsinghua University 5
6 University of Bristol 5
7 University of Exeter 5
8 Australian National University 4
9 ETH Zurich 4
10 Federal University of Rio de Janeiro 4
11 Potsdam Institute for Climate Impact Research 4
12 University of Reading 4
13 Beijing Normal University 3
14 Environment and Climate Change Canada 3
15 Imperial College London 3
16 Institute of Marine Research 3
17 Lund University 3
18 Max Planck Institute for Meteorology 3
19 National Institute for Environmental Studies 3