diff --git a/.zenodo.json b/.zenodo.json index 06c8e5c..2b7480b 100644 --- a/.zenodo.json +++ b/.zenodo.json @@ -35,5 +35,5 @@ "orcid": "http://orcid.org/0000-0001-7956-4498" }, "access_right": "open", - "description": "
The Department of Prime Minister and Cabinet’s PM Transcripts site provides transcripts of more than 20,000 speeches, media releases, and interviews by Australian Prime Ministers. These transcripts can be searched online, and the underlying XML files can be downloaded using a simple API. This repository includes Jupyter notebooks for harvesting, indexing, analysing, and aggregating the transcripts.
For more information and documentation see the PM Transcripts - GLAM Workbench section of the GLAM Workbench.
Created by Tim Sherratt for the GLAM Workbench
" + "description": "CURRENT VERSION: v1.0.0
The Department of Prime Minister and Cabinet’s PM Transcripts site provides transcripts of more than 20,000 speeches, media releases, and interviews by Australian Prime Ministers. These transcripts can be searched online, and the underlying XML files can be downloaded using a simple API. This repository includes transcripts harvested from the PM Transcripts site, together with a CSV index, and aggregations of transcripts by Prime Minister.
For more information and documentation see the PM Transcripts - GLAM Workbench section of the GLAM Workbench.
Created by Tim Sherratt for the GLAM Workbench
" } diff --git a/README.md b/README.md index 04fd194..014e7e2 100644 --- a/README.md +++ b/README.md @@ -1,13 +1,15 @@ # pm-transcripts -The Department of Prime Minister and Cabinet's [PM Transcripts site](https://pmtranscripts.pmc.gov.au) provides transcripts of more than 20,000 speeches, media releases, and interviews by Australian Prime Ministers. These transcripts can be searched online, and the underlying XML files can be downloaded using a simple API. This repository includes Jupyter notebooks for harvesting, indexing, analysing, and aggregating the transcripts. +CURRENT VERSION: v1.0.0 + +The Department of Prime Minister and Cabinet's [PM Transcripts site](https://pmtranscripts.pmc.gov.au) provides transcripts of more than 20,000 speeches, media releases, and interviews by Australian Prime Ministers. These transcripts can be searched online, and the underlying XML files can be downloaded using a simple API. This repository includes transcripts harvested from the PM Transcripts site, together with a CSV index, and aggregations of transcripts by Prime Minister. For more information and documentation see the [PM Transcripts - GLAM Workbench](https://www.glam-workbench.net/pm-transcripts) section of the [GLAM Workbench](https://glam-workbench.net). ## Notebooks -- [Aggregate transcripts by PM](https://github.com/GLAM-Workbench/pm-transcripts/blob/master/aggregate_transcripts.ipynb) -- [Harvest transcripts](https://github.com/GLAM-Workbench/pm-transcripts/blob/master/harvest_transcripts.ipynb) -- [Create an index to the harvested files](https://github.com/GLAM-Workbench/pm-transcripts/blob/master/index_and_analyse_transcript_metadata.ipynb) +- [Aggregate transcripts by PM](https://github.com/GLAM-Workbench/pm-transcripts/blob/main/aggregate_transcripts.ipynb) +- [Harvest transcripts](https://github.com/GLAM-Workbench/pm-transcripts/blob/main/harvest_transcripts.ipynb) +- [Create an index to the harvested files](https://github.com/GLAM-Workbench/pm-transcripts/blob/main/index_and_analyse_transcript_metadata.ipynb) ## Associated datasets diff --git a/aggregate_transcripts.ipynb b/aggregate_transcripts.ipynb index dcb1845..957936f 100644 --- a/aggregate_transcripts.ipynb +++ b/aggregate_transcripts.ipynb @@ -254,7 +254,7 @@ "localPath": "../pm-transcripts-data/pms/combined", "mainEntityOfPage": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/", "name": "pms/combined", - "url": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/combined" + "url": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/combined" }, { "copyrightHolder": "Commonwealth of Australia", @@ -264,7 +264,7 @@ "localPath": "../pm-transcripts-data/pms/zips", "mainEntityOfPage": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/", "name": "pms/zips", - "url": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/zips" + "url": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/zips" }, { "copyrightHolder": "Commonwealth of Australia", @@ -274,7 +274,7 @@ "localPath": "../pm-transcripts-data/pms/speech", "mainEntityOfPage": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/", "name": "pms/speech", - "url": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/speech" + "url": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/speech" } ] } diff --git a/harvest_transcripts.ipynb b/harvest_transcripts.ipynb index f480d87..a507e65 100644 --- a/harvest_transcripts.ipynb +++ b/harvest_transcripts.ipynb @@ -177,7 +177,7 @@ "license": "cc-by", "localPath": "../pm-transcripts-data/transcripts", "mainEntityOfPage": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/", - "url": "https://github.com/wragge/pm-transcripts-data/tree/master/transcripts" + "url": "https://github.com/wragge/pm-transcripts-data/tree/main/transcripts" } ] } diff --git a/index_and_analyse_transcript_metadata.ipynb b/index_and_analyse_transcript_metadata.ipynb index ba195ab..099c668 100644 --- a/index_and_analyse_transcript_metadata.ipynb +++ b/index_and_analyse_transcript_metadata.ipynb @@ -1385,7 +1385,7 @@ "license": "cc0", "localPath": "../pm-transcripts-data/index.csv", "mainEntityOfPage": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/", - "url": "https://github.com/wragge/pm-transcripts-data/blob/master/index.csv" + "url": "https://github.com/wragge/pm-transcripts-data/blob/main/index.csv" } ] } diff --git a/ro-crate-metadata.json b/ro-crate-metadata.json index 596ed81..4c80a32 100644 --- a/ro-crate-metadata.json +++ b/ro-crate-metadata.json @@ -1,49 +1,55 @@ { "@context": "https://w3id.org/ro/crate/1.2/context", - "@graph": [ + "@graph": + [ { "@id": "./", "@type": "Dataset", - "author": [ + "author": + [ { "@id": "https://orcid.org/0000-0001-7956-4498" } ], - "datePublished": "2026-09-14T02:05:22+00:00", - "description": "The Department of Prime Minister and Cabinet's [PM Transcripts site](https://pmtranscripts.pmc.gov.au) provides transcripts of more than 20,000 speeches, media releases, and interviews by Australian Prime Ministers. These transcripts can be searched online, and the underlying XML files can be downloaded using a simple API. This repository includes Jupyter notebooks for harvesting, indexing, analysing, and aggregating the transcripts.", - "hasPart": [ + "datePublished": "2026-09-14T03:36:42+00:00", + "description": "The Department of Prime Minister and Cabinet's [PM Transcripts site](https://pmtranscripts.pmc.gov.au) provides transcripts of more than 20,000 speeches, media releases, and interviews by Australian Prime Ministers. These transcripts can be searched online, and the underlying XML files can be downloaded using a simple API. This repository includes transcripts harvested from the PM Transcripts site, together with a CSV index, and aggregations of transcripts by Prime Minister.", + "hasPart": + [ { "@id": "aggregate_transcripts.ipynb" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/combined" + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/combined" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/zips" + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/zips" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/speech" + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/speech" }, { "@id": "harvest_transcripts.ipynb" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/transcripts" + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/transcripts" }, { "@id": "index_and_analyse_transcript_metadata.ipynb" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/blob/master/index.csv" + "@id": "https://github.com/wragge/pm-transcripts-data/blob/main/index.csv" } ], - "license": { + "license": + { "@id": "https://creativecommons.org/publicdomain/zero/1.0/" }, - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://www.glam-workbench.net/pm-transcripts" }, - "mentions": [ + "mentions": + [ { "@id": "#aggregate_transcripts_run_0" }, @@ -60,13 +66,25 @@ { "@id": "ro-crate-metadata.json", "@type": "CreativeWork", - "about": { + "about": + { "@id": "./" }, - "conformsTo": { + "conformsTo": + { "@id": "https://w3id.org/ro/crate/1.2" } }, + { + "@id": "create_version_v1_0_0", + "@type": "UpdateAction", + "actionStatus": + { + "@id": "http://schema.org/CompletedActionStatus" + }, + "endDate": "2026-09-14", + "name": "Create version v1.0.0" + }, { "@id": "https://creativecommons.org/publicdomain/zero/1.0/", "@type": "CreativeWork", @@ -79,18 +97,10 @@ "name": "PM Transcripts - GLAM Workbench", "url": "https://www.glam-workbench.net/pm-transcripts" }, - { - "@id": "create_version_v1_0_0", - "@type": "UpdateAction", - "actionStatus": { - "@id": "http://schema.org/CompletedActionStatus" - }, - "endDate": "2026-09-14", - "name": "Create version v1.0.0" - }, { "@id": "https://www.python.org/downloads/release/python-31211/", - "@type": [ + "@type": + [ "ComputerLanguage", "SoftwareApplication" ], @@ -100,51 +110,61 @@ }, { "@id": "aggregate_transcripts.ipynb", - "@type": [ + "@type": + [ "File", "SoftwareSourceCode" ], - "author": { + "author": + { "@id": "https://orcid.org/0000-0001-7956-4498" }, - "conformsTo": { + "conformsTo": + { "@id": "https://purl.archive.org/textcommons/profile#Notebook" }, "description": "Depending on how you want to analyse them, it can be useful to group the transcripts by prime minister. This notebook aggregates the transcripts in two ways: by extracting the text content of each XML file and combining them into one big text file, and by zipping up the original XML files.", "encodingFormat": "application/x-ipynb+json", - "isPartOf": { + "isPartOf": + { "@id": "https://github.com/GLAM-Workbench/pm-transcripts/" }, - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://www.glam-workbench.net/pm-transcripts/aggregate_transcripts/" }, "name": "Aggregate transcripts by PM", - "programmingLanguage": { + "programmingLanguage": + { "@id": "https://www.python.org/downloads/release/python-31211/" }, - "url": "https://github.com/GLAM-Workbench/pm-transcripts/blob/master/aggregate_transcripts.ipynb" + "url": "https://github.com/GLAM-Workbench/pm-transcripts/blob/main/aggregate_transcripts.ipynb" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/combined", - "@type": [ + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/combined", + "@type": + [ "File", "Dataset" ], "copyrightHolder": "Commonwealth of Australia", "dateModified": "2026-09-12T02:39:44.738760+00:00", "description": "One text file for each Prime Minister containing the aggregated contents of all the XML transcript files.", - "isPartOf": { + "isPartOf": + { "@id": "https://github.com/wragge/pm-transcripts-data" }, - "license": { + "license": + { "@id": "https://creativecommons.org/licenses/by/4.0/" }, - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/" }, "name": "pms/combined", "size": 18, - "url": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/combined" + "url": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/combined" }, { "@id": "https://github.com/wragge/pm-transcripts-data", @@ -165,76 +185,88 @@ "url": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/zips", - "@type": [ + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/zips", + "@type": + [ "File", "Dataset" ], "copyrightHolder": "Commonwealth of Australia", "dateModified": "2026-09-12T02:38:55.038536+00:00", "description": "One zip file for each Prime Minister containing the aggregated XML transcript files.", - "isPartOf": { + "isPartOf": + { "@id": "https://github.com/wragge/pm-transcripts-data" }, - "license": { + "license": + { "@id": "https://creativecommons.org/licenses/by/4.0/" }, - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/" }, "name": "pms/zips", "size": 18, - "url": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/zips" + "url": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/zips" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/speech", - "@type": [ + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/speech", + "@type": + [ "File", "Dataset" ], "copyrightHolder": "Commonwealth of Australia", "dateModified": "2026-09-10T12:31:07.246810+00:00", "description": "One text file for each Prime Minister containing the aggregated contents of all the XML transcript files identified as speeches in the file metadata.", - "isPartOf": { + "isPartOf": + { "@id": "https://github.com/wragge/pm-transcripts-data" }, - "license": { + "license": + { "@id": "https://creativecommons.org/licenses/by/4.0/" }, - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/" }, "name": "pms/speech", "size": 18, - "url": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/speech" + "url": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/speech" }, { "@id": "#aggregate_transcripts_run_0", "@type": "CreateAction", - "actionStatus": { + "actionStatus": + { "@id": "http://schema.org/CompletedActionStatus" }, "endDate": "2026-09-12T02:39:44.738760+00:00", - "instrument": { + "instrument": + { "@id": "aggregate_transcripts.ipynb" }, "name": "Run of notebook: aggregate_transcripts.ipynb", - "result": [ + "result": + [ { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/combined" + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/combined" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/zips" + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/zips" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/pms/speech" + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/pms/speech" } ] }, { "@id": "https://orcid.org/0000-0001-7956-4498", "@type": "Person", - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://timsherratt.au" }, "name": "Sherratt, Tim" @@ -259,68 +291,82 @@ }, { "@id": "harvest_transcripts.ipynb", - "@type": [ + "@type": + [ "File", "SoftwareSourceCode" ], - "author": { + "author": + { "@id": "https://orcid.org/0000-0001-7956-4498" }, - "conformsTo": { + "conformsTo": + { "@id": "https://purl.archive.org/textcommons/profile#Notebook" }, "description": "Harvest all the XML transcripts from the PMs Transcripts site.", "encodingFormat": "application/x-ipynb+json", - "isPartOf": { + "isPartOf": + { "@id": "https://github.com/GLAM-Workbench/pm-transcripts/" }, - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://www.glam-workbench.net/pm-transcripts/harvest_transcripts/" }, "name": "Harvest transcripts", - "programmingLanguage": { + "programmingLanguage": + { "@id": "https://www.python.org/downloads/release/python-31211/" }, - "url": "https://github.com/GLAM-Workbench/pm-transcripts/blob/master/harvest_transcripts.ipynb" + "url": "https://github.com/GLAM-Workbench/pm-transcripts/blob/main/harvest_transcripts.ipynb" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/transcripts", - "@type": [ + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/transcripts", + "@type": + [ "File", "Dataset" ], "copyrightHolder": "Commonwealth of Australia", "dateModified": "2026-09-12T02:40:14.366555+00:00", "description": "The complete collection of XML-formatted transcript files downloaded from PM Transcripts.", - "isPartOf": { + "isPartOf": + { "@id": "https://github.com/wragge/pm-transcripts-data" }, - "license": { + "license": + { "@id": "https://creativecommons.org/licenses/by/4.0/" }, - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/" }, "name": "transcripts", "size": 28560, - "url": "https://github.com/wragge/pm-transcripts-data/tree/master/transcripts" + "url": "https://github.com/wragge/pm-transcripts-data/tree/main/transcripts" }, { "@id": "#harvest_transcripts_run_0", "@type": "CreateAction", - "actionStatus": { + "actionStatus": + { "@id": "http://schema.org/CompletedActionStatus" }, "endDate": "2026-09-12T02:40:14.366555+00:00", - "instrument": { + "instrument": + { "@id": "harvest_transcripts.ipynb" }, - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/" }, "name": "Run of notebook: harvest_transcripts.ipynb", - "result": { - "@id": "https://github.com/wragge/pm-transcripts-data/tree/master/transcripts" + "result": + { + "@id": "https://github.com/wragge/pm-transcripts-data/tree/main/transcripts" } }, { @@ -331,33 +377,40 @@ }, { "@id": "index_and_analyse_transcript_metadata.ipynb", - "@type": [ + "@type": + [ "File", "SoftwareSourceCode" ], - "author": { + "author": + { "@id": "https://orcid.org/0000-0001-7956-4498" }, - "conformsTo": { + "conformsTo": + { "@id": "https://purl.archive.org/textcommons/profile#Notebook" }, "description": "The transcript XML files contain embedded metadata that includes the name of the prime minister, and the title and date of the transcript. This notebook extracts that metadata from the harvested files and creates a CSV formatted spreadsheet for easy analysis. It also demonstrates some ways of summarising and visualising the metadata.", "encodingFormat": "application/x-ipynb+json", - "isPartOf": { + "isPartOf": + { "@id": "https://github.com/GLAM-Workbench/pm-transcripts/" }, - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://www.glam-workbench.net/pm-transcripts/index_and_analyse_transcript_metadata/" }, "name": "Create an index to the harvested files", - "programmingLanguage": { + "programmingLanguage": + { "@id": "https://www.python.org/downloads/release/python-31211/" }, - "url": "https://github.com/GLAM-Workbench/pm-transcripts/blob/master/index_and_analyse_transcript_metadata.ipynb" + "url": "https://github.com/GLAM-Workbench/pm-transcripts/blob/main/index_and_analyse_transcript_metadata.ipynb" }, { - "@id": "https://github.com/wragge/pm-transcripts-data/blob/master/index.csv", - "@type": [ + "@id": "https://github.com/wragge/pm-transcripts-data/blob/main/index.csv", + "@type": + [ "File", "Dataset" ], @@ -365,36 +418,43 @@ "dateModified": "2026-09-12T02:28:33.066151+00:00", "description": "A CSV file containing metadata extracted from the transcript XML files.", "encodingFormat": "text/csv", - "isPartOf": { + "isPartOf": + { "@id": "https://github.com/wragge/pm-transcripts-data" }, - "license": { + "license": + { "@id": "https://creativecommons.org/publicdomain/zero/1.0/" }, - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/" }, "name": "index.csv", - "sdDatePublished": "2026-09-14T02:05:29.843600+00:00", + "sdDatePublished": "2026-09-14T03:36:48.531780+00:00", "size": 28561, - "url": "https://github.com/wragge/pm-transcripts-data/blob/master/index.csv" + "url": "https://github.com/wragge/pm-transcripts-data/blob/main/index.csv" }, { "@id": "#index_and_analyse_transcript_metadata_run_0", "@type": "CreateAction", - "actionStatus": { + "actionStatus": + { "@id": "http://schema.org/CompletedActionStatus" }, "endDate": "2026-09-12T02:28:33.066151+00:00", - "instrument": { + "instrument": + { "@id": "index_and_analyse_transcript_metadata.ipynb" }, - "mainEntityOfPage": { + "mainEntityOfPage": + { "@id": "https://www.glam-workbench.net/pm-transcripts/pm-transcripts-data/" }, "name": "Run of notebook: index_and_analyse_transcript_metadata.ipynb", - "result": { - "@id": "https://github.com/wragge/pm-transcripts-data/blob/master/index.csv" + "result": + { + "@id": "https://github.com/wragge/pm-transcripts-data/blob/main/index.csv" } }, {