From 67b1e2217ce22d3158c3612b789d6f4b5235196a Mon Sep 17 00:00:00 2001 From: Hannes Schnaitter Date: Tue, 14 Apr 2026 14:54:43 +0200 Subject: [PATCH 1/4] fix: remove old metadata_filled.yml --- metadata_filled.yml | 254 -------------------------------------------- 1 file changed, 254 deletions(-) delete mode 100644 metadata_filled.yml diff --git a/metadata_filled.yml b/metadata_filled.yml deleted file mode 100644 index 3dc6b8d..0000000 --- a/metadata_filled.yml +++ /dev/null @@ -1,254 +0,0 @@ -# yaml-language-server: $schema=https://quadriga-dk.github.io/quadriga-schema/v1.0.0-alpha/schema.json -schema-version: 1.0.0-alpha -book-version: 1.0.0-beta.3 -title: Waves of the Spanish Flu – Case Study -identifier: https://doi.org/10.5281/zenodo.14970672 -authors: -- given-names: Daniil - family-names: Skorinkin - orcid: https://orcid.org/0000-0002-1845-9974 - affiliation: Universität Potsdam -- given-names: Henny - family-names: Sluyter-Gäthje - orcid: https://orcid.org/0000-0003-2969-3237 - affiliation: Universität Potsdam -- given-names: Peer - family-names: Trilcke - orcid: https://orcid.org/0000-0002-1421-4320 - affiliation: Universität Potsdam -contributors: -- given-names: Hannes - family-names: Schnaitter - orcid: https://orcid.org/0000-0002-1602-6032 - affiliation: Humboldt-Universität zu Berlin, Institut für Bibliotheks- und Informationswissenschaft -- given-names: Evgenia - family-names: Samoilova - orcid: https://orcid.org/0000-0003-3858-901X - affiliation: Universität Potsdam -- given-names: Lamia - family-names: Islam - affiliation: Universität Potsdam -table-of-contents: '- Präambel - - - Fragestellung und Operationalisierung. Einführung in die Fallstudie - - - Korpusaufbau. Auswählen, sammeln, dokumentieren - - - OCR. Von Bild zu Text - - - "OCR-Nachbearbeitung: manuell, automatisch, LLMs" - - - Korpusverarbeitung. Von Strings zu Token - - - Korpusanalyse. Von Häufigkeiten zu Diagrammen - - - Reflexion - - - Epilog' -description: Diese OER führt in die Erstellung von QUADRIGA-OERs ein, bietet Inhalte - für Nutzer:innen der OERs und dient gleichzeitig als Template für die Erstellung - eigener OERs auf Basis der QUADRIGA-Empfehlungen. -discipline: -- übergreifend -research-object-type: -- übergreifend -chapters: -- title: Präambel - url: https://quadriga-dk.github.io/Text-Fallstudie-1/präambel/einführung.html - description: Beschreibung der Lernziele und technischer Voraussetzungen der Fallstudie. - learning-goal: Die Teilnehmenden verstehen die Lernziele und technischen Voraussetzungen - der Fallstudie. - duration: 5min - educational-level: Basis - learning-objectives: - - learning-objective: Die Teilnehmenden kennen die Lernziele und technischen Voraussetzungen - der Fallstudie. - competency: nicht anwendbar - data-flow: nicht anwendbar - blooms-category: 2 Verstehen -- title: Fragestellung und Operationalisierung. Einführung in die Fallstudie - url: https://quadriga-dk.github.io/Text-Fallstudie-1/research_question/research-question_intro.html - description: Dieses Kapitel bildet den Auftakt der Fallstudie und dient der Klärung - des Erkenntnisinteresses, das die dann folgende Vorbereitung und Aufbereitung - des Forschungsgegenstands (Korpus) und schließlich die Analyse leitet. - learning-goal: Grundlagen korpusbasierter geisteswissenschaftlicher Forschung - duration: 45min - educational-level: Basis - learning-objectives: - - learning-objective: Die Entwicklung einer Digital Humanities-Fragestellung kann - am Beispiel der Medienwellen-Forschung zur Spanischen Grippe nachvollzogen und - erläutert werden. - competency: 1 Konzeptualisierung - data-flow: 1 Plannung - blooms-category: 2 Verstehen - - learning-objective: Der Operationalisierungsprozess kann am Beispiel der Spanischen - Grippe nachvollzogen und auf andere Forschungsfragen übertragen werden. - competency: 2 Formalisierung - data-flow: 2 Erhebung - blooms-category: 3 Anwenden -- title: Korpusaufbau. Auswählen, sammeln, dokumentieren - url: https://quadriga-dk.github.io/Text-Fallstudie-1/corpus_collection/corpus-collection_intro.html - description: Der Forschungsgegenstand wird in Form eines Korpus aufbereitet. - learning-goal: Ansätze des Korpusaufbaus und Erstellung basaler Metadaten - duration: 90min - educational-level: Basis - learning-objectives: - - learning-objective: Korpora können als geisteswissenschaftliche Forschungsobjekte - definiert und deren wesentliche Merkmale beschrieben werden. - competency: 1 Konzeptualisierung - data-flow: 2 Erhebung - blooms-category: 2 Verstehen - - learning-objective: Die vier Hauptformate digitaler Texte (Bilddigitalisate, Plain - Text, XML/TEI, CSV) können anhand ihrer charakteristischen Eigenschaften unterschieden - und deren Vor- und Nachteile für spezifische Anwendungsfälle analysiert werden. - competency: 3 Datenmodellierung - data-flow: 3 Anreicherung - blooms-category: 4 Analysieren - - learning-objective: Die grundlegenden Metadatenschemata (Dublin Core, TEI, MODS, - METS) und deren charakteristische Elemente für Korpora und Einzeldokumente können - beschrieben werden. - competency: 3 Datenmodellierung - data-flow: 2 Erhebung - blooms-category: 2 Verstehen - - learning-objective: Der schrittweise Prozess des praktischen Korpusaufbaus (Konzeptentwicklung, - Metadatenerstellung und Datensammlung) kann am Beispiel eines Zeitungskorpus - beschrieben werden. - competency: nicht anwendbar - data-flow: nicht anwendbar - blooms-category: 2 Verstehen -- title: OCR. Von Bild zu Text - url: https://quadriga-dk.github.io/Text-Fallstudie-1/ocr/ocr_intro.html - description: In diesem Kapitel lernen wir, wie man mit OCR Bilder in Text umwandelt. - learning-goal: OCR-basierte Korpuserstellung und Qualitätsbewertung - duration: 120min - educational-level: Fortgeschritten - learning-objectives: - - learning-objective: Der Prozess der Optical Character Recognition (OCR) für die - Korpuserstellung kann beschrieben und Tools zur Durchführung der OCR aufgezählt - werden. - competency: 4 Automatisierung - data-flow: 3 Anreicherung - blooms-category: 2 Verstehen - - learning-objective: Die notwendigen Schritte zur Verarbeitung ein- und mehrseitiger - PDFs zu Text können aufgezählt und die Unterschiede zwischen Ursprungs- und - Zielformat erklärt werden. - competency: 4 Automatisierung - data-flow: 3 Anreicherung - blooms-category: 2 Verstehen - - learning-objective: Die grundlegenden Metriken zur OCR-Qualitätsevaluation (Präzision, - Recall, F1-Score) können erläutert und deren Bedeutung für die Bewertung von - OCR-Systemen beschrieben werden. - competency: nicht anwendbar - data-flow: nicht anwendbar - blooms-category: 2 Verstehen - - learning-objective: Die Schritte zur Qualitätsmessung eines OCR-Outputs können - aufgezählt und die Qualitätsmaße interpretiert werden. - competency: nicht anwendbar - data-flow: 3 Anreicherung - blooms-category: 3 Anwenden -- title: "OCR-Nachbearbeitung: manuell, automatisch, LLMs" - url: https://quadriga-dk.github.io/Text-Fallstudie-1/ocr_post_correction/post-correcting_intro.html - description: In diesem Kapitel werden wir die Ergebnisse der OCR nachbearbeiten. - learning-goal: OCR-Nachbearbeitung und Qualitätsverbesserung - duration: 90min - educational-level: Fortgeschritten - learning-objectives: - - learning-objective: Verschiedene Verfahren der OCR-Nachbearbeitung können beschrieben - und deren Einsatzzwecke unterschieden werden. - competency: 4 Automatisierung - data-flow: 3 Anreicherung - blooms-category: 2 Verstehen - - learning-objective: Regelbasierte Ansätze zur OCR-Nachkorrektur können beschrieben - und deren Auswirkungen auf die OCR-Qualität anhand von Metriken erläutert werden. - competency: 4 Automatisierung - data-flow: 3 Anreicherung - blooms-category: 4 Analysieren - - learning-objective: Die grundlegenden Herausforderungen beim Einsatz von Large - Language Models für die OCR-Nachbearbeitung können beschrieben werden. - competency: 4 Automatisierung - data-flow: 3 Anreicherung - blooms-category: 2 Verstehen -- title: Korpusverarbeitung. Von Strings zu Token - url: https://quadriga-dk.github.io/Text-Fallstudie-1/corpus_processing/corpus-processing_intro.html - description: Die im Korpus enthaltenen Textdateien werden mit linguistischen Informationen - angereichert. - learning-goal: Korpusverarbeitung mit Natural Language Processing - duration: 60min - educational-level: Fortgeschritten - learning-objectives: - - learning-objective: Die Grundkonzepte des Natural Language Processing können erklärt - und die Funktionen von Tokenisierung und Lemmatisierung für die Textanalyse - beschrieben werden. - competency: 4 Automatisierung - data-flow: 3 Anreicherung - blooms-category: 2 Verstehen - - learning-objective: Die notwendigen Schritte zur automatischen Annotation eines - Texts können aufgezählt und Vorteile der Tokenisierung gegenüber einfacheren - Methoden der Worttrennung genannt werden. - competency: 4 Automatisierung - data-flow: 3 Anreicherung - blooms-category: 3 Anwenden -- title: Korpusanalyse. Von Häufigkeiten zu Diagrammen - url: https://quadriga-dk.github.io/Text-Fallstudie-1/corpus_analysis/corpus-analysis_intro.html - description: Nachdem die Korpuserstellung und -anreicherung abgeschlossen ist, wird - in diesem Kapitel zur Forschungsfrage zurückgekehrt. Es soll die öffentliche Aufmerksamkeit - für die spanische Grippe im Zeitaum von 1918-1920 an Hand von Worthäufigkeiten - derjenigen Wörter gemessen werden, die direkt oder indirekt auf die spanische - Grippe verweisen. - learning-goal: Frequenzanalysen semantischer Felder - duration: 90min - educational-level: Fortgeschritten - learning-objectives: - - learning-objective: Das Konzept des semantischen Feldes kann erklärt, der Unterschied - zwischen absoluten und relativen Häufigkeiten beschrieben und die Darstellungsmethoden - des Liniendiagramms und der Key Word in Context (KWIC)-Anzeige interpretiert - werden. - competency: 10 Fachspezifische datenanalyse - data-flow: 4 Analyse - blooms-category: 2 Verstehen - - learning-objective: Die notwendigen Schritte zur Frequenzanalyse eines semantischen - Felds können aufgezählt, Unterschiede in der Berechnung der Häufigkeiten benannt - und die Ergebnisse reflektiert werden. - competency: 10 Fachspezifische datenanalyse - data-flow: 4 Analyse - blooms-category: 5 Bewerten - - learning-objective: Die Darstellungsmethode Keywords in Context kann beschrieben, - Wörter zur Anzeige ausgewählt und diese angezeigt werden. - competency: 11 Visualisierung - data-flow: 5 Visualisierung - blooms-category: 3 Anwenden -- title: Reflexion - url: https://quadriga-dk.github.io/Text-Fallstudie-1/reflection/reflection_reflection.html - description: Auch wenn es uns gelungen ist, das Ziel der Fallstudie zu erreichen - und unsere eingangs formulierte Forschungsfrage nach den "Medienwellen" der Spanischen - Grippe insofern exemplarisch zu beantworten, als wir tatsächliche eine wellenartige - Entwicklung der quantitative verstandenen Aufmerksamkeit in ausgewählten Berliner - Tageszeitungen nachweisen konnten, gibt es mehrere Punkte, die abschließend kritisch - zu reflektieren sind. - learning-goal: Kritische Bewertung der Reichweite und Limitationen - duration: 30min - educational-level: Fortgeschritten - learning-objectives: - - learning-objective: Die methodischen Limitationen einer Digital Humanities-Fallstudie - können benannt werden. - competency: 7 Interpretation - data-flow: nicht anwendbar - blooms-category: 5 Bewerten -target-group: -- Promovierende -- Forschende (PostDoc) -- Hochschullehrende -duration: 12h -date-published: '2024-06-17' -date-modified: '2025-08-21' -context-of-creation: 'Die vorliegenden Open Educational Resources wurden durch das - Datenkompetenzzentrum QUADRIGA erstellt. - - Förderkennzeichen: 16DKZ2034' -url: https://quadriga-dk.github.io/Text-Fallstudie-1/ -git: https://github.com/quadriga-dk/Text-Fallstudie-1 -license: - code: https://opensource.org/licenses/AGPL-3.0 - content: - url: https://creativecommons.org/licenses/by-sa/4.0/ - name: CC BY-SA 4.0 \ No newline at end of file From 01658a13e0c00954e5cf6a61e50c452b1830df05 Mon Sep 17 00:00:00 2001 From: Hannes Schnaitter Date: Tue, 30 Jun 2026 08:53:18 +0200 Subject: [PATCH 2/4] feat: upgrade files and scripts to current version --- _config.yml | 8 +- _static/assessment.js | 436 +++++++++--------- _static/carousel.js | 140 +++--- _static/custom-table.css | 93 ++++ _static/drag_drop.css | 2 +- _static/quadriga.css | 34 ++ "einstieg/einf\303\274hrung.md" | 6 +- metadata.jsonld | 4 +- metadata.rdf | 4 +- metadata.yml | 4 +- quadriga/__init__.py | 2 +- quadriga/assessment.py | 51 +- quadriga/colors.py | 2 +- quadriga/metadata/__init__.py | 4 +- quadriga/metadata/create_bibtex.py | 318 +++---------- quadriga/metadata/create_citation_cff.py | 204 ++++++++ quadriga/metadata/create_jsonld.py | 75 +-- quadriga/metadata/create_rdfxml.py | 111 ++--- quadriga/metadata/create_zenodo_json.py | 183 ++------ quadriga/metadata/extract_from_book_config.py | 15 +- quadriga/metadata/extract_from_lernziele.py | 113 +++-- quadriga/metadata/inject_all_metadata.py | 105 ++--- quadriga/metadata/run_all.py | 32 +- quadriga/metadata/update_citation_cff.py | 323 ------------- quadriga/metadata/utils.py | 235 +++++++++- quadriga/metadata/validate_schema.py | 13 +- 26 files changed, 1162 insertions(+), 1355 deletions(-) create mode 100644 _static/custom-table.css create mode 100644 quadriga/metadata/create_citation_cff.py delete mode 100644 quadriga/metadata/update_citation_cff.py diff --git a/_config.yml b/_config.yml index c1c2e3a..08ff051 100644 --- a/_config.yml +++ b/_config.yml @@ -26,6 +26,12 @@ sphinx: config: bibtex_reference_style: author_year html_show_copyright: false + html_static_path: ["_static"] + html_css_files: + - quadriga.css + - carousel.css + html_js_files: + - carousel.js # Information about where the book exists on the web #repository: @@ -45,7 +51,7 @@ html: use_repository_button: true extra_footer:

Dieses Werk ist lizenziert unter der Lizenz CC BY-SA 4.0. Detailliertere Informationen finden Sie unter LICENSE.md

-

Diese OER wurde im Rahmen des Projekts QUADRIGA erstellt. Das Projekt wurde gefördert durch das Bundesministerium für Forschung, Technologie und Raumfahrt (BMFTR). Eine Kurzvorstellung des Projekts finden Sie im Abschluss.

+

Diese OER wurde im Rahmen des Projekts QUADRIGA erstellt. Das Projekt wurde gefördert durch das Bundesministerium für Forschung, Technologie und Raumfahrt (BMFTR). Mehr Informationen zum Projekt finden Sie auch im abschluss.

launch_buttons: notebook_interface: jupyterlab diff --git a/_static/assessment.js b/_static/assessment.js index 9e93fa5..505f418 100644 --- a/_static/assessment.js +++ b/_static/assessment.js @@ -1,227 +1,237 @@ // Function to lock an answer after submission function lockAnswer(id) { - const textarea = document.getElementById(id); - const button = document.getElementById('btn-' + id); - - if (textarea.value.trim() === '') { - alert('Please enter your answer before submitting.'); - return; - } - - textarea.readOnly = true; - textarea.style.fontWeight = 'bold'; - textarea.style.color = '#666'; - textarea.style.backgroundColor = '#f8f9fa'; - button.style.display = 'none'; + const textarea = document.getElementById(id); + const button = document.getElementById("btn-" + id); + + if (textarea.value.trim() === "") { + alert("Please enter your answer before submitting."); + return; } + textarea.readOnly = true; + textarea.style.fontWeight = "bold"; + textarea.style.color = "#666"; + textarea.style.backgroundColor = "#f8f9fa"; + button.style.display = "none"; +} class DragDropQuizManager { - constructor() { - this.quizzes = new Map(); - } - - initializeQuiz(quizId, correctPairs, showFeedback, originalOptions, customFeedback) { - const quizData = { - quizId, - correctPairs, - showFeedback, - originalOptions, - customFeedback, - draggedElement: null - }; - - this.quizzes.set(quizId, quizData); - this.initializeOptions(quizId); - this.setupEventListeners(quizId); - this.createGlobalFunctions(quizId); + constructor() { + this.quizzes = new Map(); + } + + initializeQuiz( + quizId, + correctPairs, + showFeedback, + originalOptions, + customFeedback, + ) { + const quizData = { + quizId, + correctPairs, + showFeedback, + originalOptions, + customFeedback, + draggedElement: null, + }; + + this.quizzes.set(quizId, quizData); + this.initializeOptions(quizId); + this.setupEventListeners(quizId); + this.createGlobalFunctions(quizId); + } + + // Shuffle array function + shuffleArray(array) { + const shuffled = [...array]; + for (let i = shuffled.length - 1; i > 0; i--) { + const j = Math.floor(Math.random() * (i + 1)); + [shuffled[i], shuffled[j]] = [shuffled[j], shuffled[i]]; } - - // Shuffle array function - shuffleArray(array) { - const shuffled = [...array]; - for (let i = shuffled.length - 1; i > 0; i--) { - const j = Math.floor(Math.random() * (i + 1)); - [shuffled[i], shuffled[j]] = [shuffled[j], shuffled[i]]; + return shuffled; + } + + // Initialize options with shuffling + initializeOptions(quizId) { + const quiz = this.quizzes.get(quizId); + const optionsList = document.getElementById(`${quizId}_options`); + const shuffledOptions = this.shuffleArray( + quiz.originalOptions.map((option, index) => ({ option, index })), + ); + + optionsList.innerHTML = ""; + shuffledOptions.forEach((item) => { + const draggableOption = document.createElement("div"); + draggableOption.className = "draggable-option"; + draggableOption.draggable = true; + draggableOption.dataset.itemId = item.index; + draggableOption.id = `${quizId}_option_${item.index}`; + draggableOption.textContent = item.option; + optionsList.appendChild(draggableOption); + }); + + this.initializeDragEvents(quizId); + } + + // Add event listeners to draggable options + initializeDragEvents(quizId) { + const quiz = this.quizzes.get(quizId); + const draggableOptions = document.querySelectorAll( + `#${quizId} .draggable-option`, + ); + + draggableOptions.forEach((option) => { + option.addEventListener("dragstart", (e) => { + quiz.draggedElement = option; + option.classList.add("dragging"); + e.dataTransfer.effectAllowed = "move"; + e.dataTransfer.setData("text/html", option.outerHTML); + }); + + option.addEventListener("dragend", (e) => { + option.classList.remove("dragging"); + }); + }); + } + + setupEventListeners(quizId) { + const quiz = this.quizzes.get(quizId); + const dropAreas = document.querySelectorAll(`#${quizId} .drop-area`); + + dropAreas.forEach((area) => { + area.addEventListener("dragover", (e) => { + e.preventDefault(); + area.classList.add("drag-over"); + }); + + area.addEventListener("dragleave", (e) => { + area.classList.remove("drag-over"); + }); + + area.addEventListener("drop", (e) => { + e.preventDefault(); + area.classList.remove("drag-over"); + + if (quiz.draggedElement) { + // Clear the drop area + area.innerHTML = ""; + + // Create a new element for the dropped item + const droppedItem = document.createElement("div"); + droppedItem.className = "dropped-item"; + droppedItem.textContent = quiz.draggedElement.textContent; + droppedItem.dataset.itemId = quiz.draggedElement.dataset.itemId; + + // Add click listener to return item to original position + droppedItem.addEventListener("click", () => { + this.returnItemToOriginal(quizId, droppedItem); + }); + + area.appendChild(droppedItem); + + // Remove the original draggable option + quiz.draggedElement.remove(); + quiz.draggedElement = null; } - return shuffled; - } - - // Initialize options with shuffling - initializeOptions(quizId) { - const quiz = this.quizzes.get(quizId); - const optionsList = document.getElementById(`${quizId}_options`); - const shuffledOptions = this.shuffleArray(quiz.originalOptions.map((option, index) => ({ option, index }))); - - optionsList.innerHTML = ''; - shuffledOptions.forEach(item => { - const draggableOption = document.createElement('div'); - draggableOption.className = 'draggable-option'; - draggableOption.draggable = true; - draggableOption.dataset.itemId = item.index; - draggableOption.id = `${quizId}_option_${item.index}`; - draggableOption.textContent = item.option; - optionsList.appendChild(draggableOption); - }); - - this.initializeDragEvents(quizId); - } - - // Add event listeners to draggable options - initializeDragEvents(quizId) { - const quiz = this.quizzes.get(quizId); - const draggableOptions = document.querySelectorAll(`#${quizId} .draggable-option`); - - draggableOptions.forEach(option => { - option.addEventListener('dragstart', (e) => { - quiz.draggedElement = option; - option.classList.add('dragging'); - e.dataTransfer.effectAllowed = 'move'; - e.dataTransfer.setData('text/html', option.outerHTML); - }); - - option.addEventListener('dragend', (e) => { - option.classList.remove('dragging'); - }); - }); - } - - setupEventListeners(quizId) { - const quiz = this.quizzes.get(quizId); - const dropAreas = document.querySelectorAll(`#${quizId} .drop-area`); - - dropAreas.forEach(area => { - area.addEventListener('dragover', (e) => { - e.preventDefault(); - area.classList.add('drag-over'); - }); - - area.addEventListener('dragleave', (e) => { - area.classList.remove('drag-over'); - }); - - area.addEventListener('drop', (e) => { - e.preventDefault(); - area.classList.remove('drag-over'); - - if (quiz.draggedElement) { - // Clear the drop area - area.innerHTML = ''; - - // Create a new element for the dropped item - const droppedItem = document.createElement('div'); - droppedItem.className = 'dropped-item'; - droppedItem.textContent = quiz.draggedElement.textContent; - droppedItem.dataset.itemId = quiz.draggedElement.dataset.itemId; - - // Add click listener to return item to original position - droppedItem.addEventListener('click', () => { - this.returnItemToOriginal(quizId, droppedItem); - }); - - area.appendChild(droppedItem); - - // Remove the original draggable option - quiz.draggedElement.remove(); - quiz.draggedElement = null; - } - }); - }); - } - - // Function to return item to original position - returnItemToOriginal(quizId, droppedItem) { - const optionsList = document.getElementById(`${quizId}_options`); - - // Create new draggable option - const newDraggableOption = document.createElement('div'); - newDraggableOption.className = 'draggable-option'; - newDraggableOption.draggable = true; - newDraggableOption.dataset.itemId = droppedItem.dataset.itemId; - newDraggableOption.id = `${quizId}_option_${droppedItem.dataset.itemId}`; - newDraggableOption.textContent = droppedItem.textContent; - - optionsList.appendChild(newDraggableOption); - - // Remove from drop zone and restore placeholder - const dropArea = droppedItem.parentElement; - dropArea.innerHTML = 'Hier ablegen'; - - // Re-initialize drag events for the new option - this.initializeDragEvents(quizId); - } - - // Check answer function - checkAnswer(quizId) { - const quiz = this.quizzes.get(quizId); - if (!quiz.showFeedback) return; - - const feedback = document.getElementById(`${quizId}_feedback`); - let correctCount = 0; - let totalPairs = quiz.correctPairs.length; - - // Check each correct pair and apply visual feedback - quiz.correctPairs.forEach(([descId, optionId]) => { - const dropArea = document.getElementById(`${quizId}_drop_${descId}`); - const droppedItem = dropArea.querySelector('.dropped-item'); - - // Remove previous feedback classes - dropArea.classList.remove('correct-answer', 'incorrect-answer'); - - if (droppedItem) { - if (parseInt(droppedItem.dataset.itemId) === optionId) { - correctCount++; - dropArea.classList.add('correct-answer'); - } else { - dropArea.classList.add('incorrect-answer'); - } - } else { - // Empty drop area is also incorrect - dropArea.classList.add('incorrect-answer'); - } - }); - - let message, className; - if (correctCount === totalPairs) { - message = quiz.customFeedback.correct.replace('{total}', totalPairs); - className = 'correct'; - } else if (correctCount === 0) { - message = quiz.customFeedback.incorrect; - className = 'incorrect'; + }); + }); + } + + // Function to return item to original position + returnItemToOriginal(quizId, droppedItem) { + const optionsList = document.getElementById(`${quizId}_options`); + + // Create new draggable option + const newDraggableOption = document.createElement("div"); + newDraggableOption.className = "draggable-option"; + newDraggableOption.draggable = true; + newDraggableOption.dataset.itemId = droppedItem.dataset.itemId; + newDraggableOption.id = `${quizId}_option_${droppedItem.dataset.itemId}`; + newDraggableOption.textContent = droppedItem.textContent; + + optionsList.appendChild(newDraggableOption); + + // Remove from drop zone and restore placeholder + const dropArea = droppedItem.parentElement; + dropArea.innerHTML = 'Hier ablegen'; + + // Re-initialize drag events for the new option + this.initializeDragEvents(quizId); + } + + // Check answer function + checkAnswer(quizId) { + const quiz = this.quizzes.get(quizId); + if (!quiz.showFeedback) return; + + const feedback = document.getElementById(`${quizId}_feedback`); + let correctCount = 0; + let totalPairs = quiz.correctPairs.length; + + // Check each correct pair and apply visual feedback + quiz.correctPairs.forEach(([descId, optionId]) => { + const dropArea = document.getElementById(`${quizId}_drop_${descId}`); + const droppedItem = dropArea.querySelector(".dropped-item"); + + // Remove previous feedback classes + dropArea.classList.remove("correct-answer", "incorrect-answer"); + + if (droppedItem) { + if (parseInt(droppedItem.dataset.itemId) === optionId) { + correctCount++; + dropArea.classList.add("correct-answer"); } else { - message = quiz.customFeedback.partial - .replace('{correct}', correctCount) - .replace('{total}', totalPairs); - className = 'partial'; + dropArea.classList.add("incorrect-answer"); } - - feedback.innerHTML = ``; - } - - // Reset quiz function - resetQuiz(quizId) { - const dropAreas = document.querySelectorAll(`#${quizId} .drop-area`); - const feedback = document.getElementById(`${quizId}_feedback`); - - // Clear all drop areas and remove feedback classes - dropAreas.forEach(area => { - area.innerHTML = 'Hier ablegen'; - area.classList.remove('correct-answer', 'incorrect-answer'); - }); - - // Clear feedback - feedback.innerHTML = ''; - - // Re-initialize options with new shuffle - this.initializeOptions(quizId); - } - - // Create global functions for each quiz - createGlobalFunctions(quizId) { - window[`checkAnswer_${quizId}`] = () => this.checkAnswer(quizId); - window[`resetQuiz_${quizId}`] = () => this.resetQuiz(quizId); + } else { + // Empty drop area is also incorrect + dropArea.classList.add("incorrect-answer"); + } + }); + + let message, className; + if (correctCount === totalPairs) { + message = quiz.customFeedback.correct.replace("{total}", totalPairs); + className = "correct"; + } else if (correctCount === 0) { + message = quiz.customFeedback.incorrect; + className = "incorrect"; + } else { + message = quiz.customFeedback.partial + .replace("{correct}", correctCount) + .replace("{total}", totalPairs); + className = "partial"; } + + feedback.innerHTML = ``; + } + + // Reset quiz function + resetQuiz(quizId) { + const dropAreas = document.querySelectorAll(`#${quizId} .drop-area`); + const feedback = document.getElementById(`${quizId}_feedback`); + + // Clear all drop areas and remove feedback classes + dropAreas.forEach((area) => { + area.innerHTML = 'Hier ablegen'; + area.classList.remove("correct-answer", "incorrect-answer"); + }); + + // Clear feedback + feedback.innerHTML = ""; + + // Re-initialize options with new shuffle + this.initializeOptions(quizId); + } + + // Create global functions for each quiz + createGlobalFunctions(quizId) { + window[`checkAnswer_${quizId}`] = () => this.checkAnswer(quizId); + window[`resetQuiz_${quizId}`] = () => this.resetQuiz(quizId); + } } // Global instance -window.dragDropQuizManager = new DragDropQuizManager(); \ No newline at end of file +window.dragDropQuizManager = new DragDropQuizManager(); + diff --git a/_static/carousel.js b/_static/carousel.js index 6ed928c..288f139 100644 --- a/_static/carousel.js +++ b/_static/carousel.js @@ -1,72 +1,78 @@ -document.addEventListener('DOMContentLoaded', function() { - // Wait a bit for sphinx-design to initialize - setTimeout(() => { - initializeCarousels(); - }, 500); +document.addEventListener("DOMContentLoaded", function () { + // Wait a bit for sphinx-design to initialize + setTimeout(() => { + initializeCarousels(); + }, 500); }); function initializeCarousels() { - const carousels = document.querySelectorAll('.sd-cards-carousel'); + const carousels = document.querySelectorAll(".sd-cards-carousel"); - carousels.forEach((carousel, index) => { - // Ensure carousel is wrapped - if (!carousel.parentElement.classList.contains('sd-cards-carousel-wrapper')) { - const wrapper = document.createElement('div'); - wrapper.className = 'sd-cards-carousel-wrapper'; - carousel.parentNode.insertBefore(wrapper, carousel); - wrapper.appendChild(carousel); - } - - // Create navigation buttons - const prevButton = document.createElement('button'); - prevButton.innerHTML = ''; - prevButton.className = 'carousel-nav-button carousel-prev'; - - const nextButton = document.createElement('button'); - nextButton.innerHTML = ''; - nextButton.className = 'carousel-nav-button carousel-next'; - - // Add buttons to carousel - carousel.parentElement.appendChild(prevButton); - carousel.parentElement.appendChild(nextButton); - - // Navigation logic - prevButton.addEventListener('click', () => { - const cardWidth = carousel.offsetWidth; - carousel.scrollBy({ - left: -cardWidth, - behavior: 'smooth' - }); - }); - - nextButton.addEventListener('click', () => { - const cardWidth = carousel.offsetWidth; - carousel.scrollBy({ - left: cardWidth, - behavior: 'smooth' - }); - }); - - // Update button states - const updateButtons = () => { - prevButton.classList.toggle('disabled', carousel.scrollLeft <= 0); - nextButton.classList.toggle('disabled', - carousel.scrollLeft >= carousel.scrollWidth - carousel.clientWidth - 5); // 5px buffer - }; - - // Add event listeners - carousel.addEventListener('scroll', updateButtons); - window.addEventListener('resize', () => { - updateButtons(); - // Ensure proper scroll position on resize - const currentIndex = Math.round(carousel.scrollLeft / carousel.offsetWidth); - carousel.scrollTo({ - left: currentIndex * carousel.offsetWidth, - behavior: 'auto' - }); - }); - - // Initialize button states - updateButtons(); + carousels.forEach((carousel, index) => { + // Ensure carousel is wrapped + if ( + !carousel.parentElement.classList.contains("sd-cards-carousel-wrapper") + ) { + const wrapper = document.createElement("div"); + wrapper.className = "sd-cards-carousel-wrapper"; + carousel.parentNode.insertBefore(wrapper, carousel); + wrapper.appendChild(carousel); + } + + // Create navigation buttons + const prevButton = document.createElement("button"); + prevButton.innerHTML = ''; + prevButton.className = "carousel-nav-button carousel-prev"; + + const nextButton = document.createElement("button"); + nextButton.innerHTML = ''; + nextButton.className = "carousel-nav-button carousel-next"; + + // Add buttons to carousel + carousel.parentElement.appendChild(prevButton); + carousel.parentElement.appendChild(nextButton); + + // Navigation logic + prevButton.addEventListener("click", () => { + const cardWidth = carousel.offsetWidth; + carousel.scrollBy({ + left: -cardWidth, + behavior: "smooth", + }); + }); + + nextButton.addEventListener("click", () => { + const cardWidth = carousel.offsetWidth; + carousel.scrollBy({ + left: cardWidth, + behavior: "smooth", + }); }); -} \ No newline at end of file + + // Update button states + const updateButtons = () => { + prevButton.classList.toggle("disabled", carousel.scrollLeft <= 0); + nextButton.classList.toggle( + "disabled", + carousel.scrollLeft >= carousel.scrollWidth - carousel.clientWidth - 5, + ); // 5px buffer + }; + + // Add event listeners + carousel.addEventListener("scroll", updateButtons); + window.addEventListener("resize", () => { + updateButtons(); + // Ensure proper scroll position on resize + const currentIndex = Math.round( + carousel.scrollLeft / carousel.offsetWidth, + ); + carousel.scrollTo({ + left: currentIndex * carousel.offsetWidth, + behavior: "auto", + }); + }); + + // Initialize button states + updateButtons(); + }); +} diff --git a/_static/custom-table.css b/_static/custom-table.css new file mode 100644 index 0000000..b1ca453 --- /dev/null +++ b/_static/custom-table.css @@ -0,0 +1,93 @@ +/* sparql output table styling*/ + +div.cell_output table { + width: 100% !important; + table-layout: fixed; + border-collapse: collapse; +} + + +div.cell_output table tbody tr, +div.cell_output table tbody tr th, +div.cell_output table tbody tr td { + text-align: left !important; + position: relative; +} + +div.cell_output table tbody tr td { + overflow: hidden; + white-space: nowrap; + text-overflow: ellipsis; + cursor: pointer; + transition: all 0.3s ease; + padding: 8px; + word-wrap: break-word; +} + +/* Dynamic width based on number of columns */ +div.cell_output table td:first-child:nth-last-child(2), +div.cell_output table td:first-child:nth-last-child(2) ~ td { + width: 50%; + max-width: 300px; +} + + +/* Adjust max-width if there are more than 4 columns */ +div.cell_output table td:first-child:nth-last-child(n+3), +div.cell_output table td:first-child:nth-last-child(n+3) ~ td { + max-width: 120px !important; +} + +/* Row expansion on hover */ +div.cell_output table tbody tr:hover td { + white-space: normal; + background-color: rgba(255, 255, 255, 0.02); + padding: 12px 8px; + vertical-align: top; +} + + +/* Enhanced styling for expanded state */ +div.cell_output table tbody tr:hover { + box-shadow: 0 2px 8px rgba(0, 0, 0, 0.1); + z-index: 1; + position: relative; +} + +/* Dark mode */ +html[data-theme='dark'] div.cell_output table { + background-color: #1a1a1a; + color: #ffffff; +} + +html[data-theme=dark] .bd-content div.cell_output .text_html:not(:has(table.dataframe)) { + background-color: #1a1a1a; + color: #ffffff; +} + +html[data-theme=dark] div.cell_output th { + background-color: #444444; + color: #ffffff; +} + +html[data-theme=dark] div.cell_output table tbody tr:nth-child(odd) { + background-color: #444444; + color: #ffffff; +} + +div.cell_output tbody tr:hover{ + background: rgba(66, 165, 245, 0.2) !important; +} + +/* Responsive adjustments */ +@media (max-width: 768px) { + div.cell_output table td { + min-width: 100px; + font-size: 14px; + } + + div.cell_output table tbody tr:hover td { + font-size: 13px; + padding: 10px 6px; + } +} diff --git a/_static/drag_drop.css b/_static/drag_drop.css index 1c6ff85..bed38fc 100644 --- a/_static/drag_drop.css +++ b/_static/drag_drop.css @@ -154,7 +154,7 @@ html[data-theme='dark'] .reset-button { .reset-button:hover { background: #00305e; - color: white; + color: white !important; } .feedback { diff --git a/_static/quadriga.css b/_static/quadriga.css index d8b6cac..bb41aed 100644 --- a/_static/quadriga.css +++ b/_static/quadriga.css @@ -347,3 +347,37 @@ html[data-theme='dark'] div.story > .admonition-title::after { text-decoration: none; white-space: nowrap; } +<<<<<<< ours + +/* Inline run-mode icons in the "Hinweise zur Ausführung" box. + inline-run-links.js turns these spans into links to Colab / the .ipynb + source; the rule makes them read as interactive (and keeps the emoji from + getting an underline). */ +.launch-colab-inline, +.launch-ipynb-inline { + cursor: pointer; + text-decoration: none; + white-space: nowrap; +} +||||||| ancestor +======= + +/* Fix bibliography URL overflow — targets actual rendered HTML */ +div.citation { + display: flex; + flex-wrap: wrap; + gap: 0.25rem; + margin-bottom: 1rem; + overflow: hidden; +} + +div.citation p, +div.citation dd, +div.citation .docutils { + overflow-wrap: break-word; + word-break: break-all; + word-wrap: break-word; + min-width: 0; + flex: 1; +} +>>>>>>> theirs diff --git "a/einstieg/einf\303\274hrung.md" "b/einstieg/einf\303\274hrung.md" index 15b1524..6060f1f 100644 --- "a/einstieg/einf\303\274hrung.md" +++ "b/einstieg/einf\303\274hrung.md" @@ -5,6 +5,6 @@ lang: de-DE Die einzelnen Kapitel dieser Open Educational Resource sind als eigenständige Lerneinheiten anzusehen und können unabhängig voneinander absolviert werden, insbesondere wenn bereits Vorwissen zum jeweils behandelten Lerninhalt vorhanden ist. Im Regelfall wird jedoch empfohlen, die Kapitel in der hier vorliegenden Reihenfolge zu bearbeiten, da die in den Kapiteln vorgestellten Schritte der Abfolge eines tatsächlichen Forschungsprojekts entsprechen. -Im Folgenden können Sie sich einen Überblick über die Lernziele und die technischen Voraussetzungen der Fallstudie verschaffen: -- [Lernziele](../introduction/learning-outcomes) -- [Technische Voraussetzungen](../introduction/introduction_requirements) +Machen Sie sich zunächst mit den +- [Lernzielen](../introduction/learning-outcomes) sowie den +- [Technische Voraussetzungen](../introduction/introduction_requirements) vertraut. diff --git a/metadata.jsonld b/metadata.jsonld index c3c1e48..0ad946f 100644 --- a/metadata.jsonld +++ b/metadata.jsonld @@ -176,9 +176,9 @@ "hasPart": [ { "@type": "LearningResource", - "name": "Präambel", + "name": "einstieg", "description": "Beschreibung der Lernziele und technischer Voraussetzungen der Fallstudie.", - "url": "https://quadriga-dk.github.io/Text-Fallstudie-1/präambel/einführung.html", + "url": "https://quadriga-dk.github.io/Text-Fallstudie-1/einstieg/einführung.html", "timeRequired": "PT1M", "teaches": "Die Teilnehmenden verstehen die Lernziele und technischen Voraussetzungen der Fallstudie.", "educationalAlignment": [ diff --git a/metadata.rdf b/metadata.rdf index 553e0c6..beeaee7 100644 --- a/metadata.rdf +++ b/metadata.rdf @@ -211,10 +211,10 @@ Förderkennzeichen: 16DKZ2034 Beschreibung der Lernziele und technischer Voraussetzungen der Fallstudie. - Präambel + einstieg Die Teilnehmenden verstehen die Lernziele und technischen Voraussetzungen der Fallstudie. PT1M - + diff --git a/metadata.yml b/metadata.yml index 63e13dd..457849a 100644 --- a/metadata.yml +++ b/metadata.yml @@ -70,8 +70,8 @@ identifier: https://doi.org/10.5281/zenodo.14970672 git: https://github.com/quadriga-dk/Text-Fallstudie-1 url: https://quadriga-dk.github.io/Text-Fallstudie-1/ chapters: -- title: Präambel - url: https://quadriga-dk.github.io/Text-Fallstudie-1/präambel/einführung.html +- title: einstieg + url: https://quadriga-dk.github.io/Text-Fallstudie-1/einstieg/einführung.html description: Beschreibung der Lernziele und technischer Voraussetzungen der Fallstudie. learning-goal: Die Teilnehmenden verstehen die Lernziele und technischen Voraussetzungen der Fallstudie. diff --git a/quadriga/__init__.py b/quadriga/__init__.py index 697a083..ad2a26a 100644 --- a/quadriga/__init__.py +++ b/quadriga/__init__.py @@ -8,6 +8,6 @@ __version__ = "0.1.0" # Import important modules to make them available at the top level -from . import colors +from . import colors as colors # Metadata submodule is not imported directly as it's intended to be used as a standalone CLI tool diff --git a/quadriga/assessment.py b/quadriga/assessment.py index 181db64..fb53eb4 100644 --- a/quadriga/assessment.py +++ b/quadriga/assessment.py @@ -1,14 +1,15 @@ -from IPython.display import HTML import json import uuid +from IPython.display import HTML + def create_answer_box(question_id, rows=4): - """Create an answer box with a submit button.""" + """Create an answer box with an info popup button.""" return HTML(f"""
- +
""") @@ -16,7 +17,7 @@ def create_answer_box(question_id, rows=4): class DragDropQuiz: """ A simple drag-and-drop quiz generator for Jupyter Books. - + Usage: quiz = DragDropQuiz() quiz.create_matching_quiz( @@ -26,14 +27,16 @@ class DragDropQuiz: correct_mapping={"Description 1": "Option A", "Description 2": "Option B", "Description 3": "Option C"} ) """ - + def __init__(self): self.quiz_counter = 0 - - def create_matching_quiz(self, title, descriptions, options, correct_mapping, show_feedback=True, feedback_messages=None): + + def create_matching_quiz( + self, title, descriptions, options, correct_mapping, show_feedback=True, feedback_messages=None + ): """ Create a drag-and-drop matching quiz. - + Parameters: - title (str): The quiz title/question - descriptions (list): List of items to be matched (static labels) @@ -44,33 +47,33 @@ def create_matching_quiz(self, title, descriptions, options, correct_mapping, sh """ self.quiz_counter += 1 quiz_id = f"drag_drop_quiz_{self.quiz_counter}_{uuid.uuid4().hex[:8]}" - + # Set default feedback messages if none provided if feedback_messages is None: feedback_messages = { "correct": "Perfekt! Alle {total} Zuordnungen sind korrekt!", "incorrect": "Leider sind keine Zuordnungen korrekt. Versuchen Sie es noch einmal!", - "partial": "Teilweise richtig: {correct} von {total} Zuordnungen sind korrekt." + "partial": "Teilweise richtig: {correct} von {total} Zuordnungen sind korrekt.", } - + # Convert correct mapping to use indices for easier JavaScript handling desc_to_idx = {desc: i for i, desc in enumerate(descriptions)} opt_to_idx = {opt: i for i, opt in enumerate(options)} - + correct_pairs = [] for desc, opt in correct_mapping.items(): if desc in desc_to_idx and opt in opt_to_idx: correct_pairs.append([desc_to_idx[desc], opt_to_idx[opt]]) - + html_content = self._generate_html( quiz_id, title, descriptions, options, correct_pairs, show_feedback, feedback_messages ) - + return HTML(html_content) - + def _generate_html(self, quiz_id, title, descriptions, options, correct_pairs, show_feedback, feedback_messages): """Generate the complete HTML for the drag-and-drop quiz.""" - + # Generate static description labels with drop zones description_zones = "" for i, desc in enumerate(descriptions): @@ -82,7 +85,7 @@ def _generate_html(self, quiz_id, title, descriptions, options, correct_pairs, s ''' - + # Generate draggable options draggable_options = "" for i, option in enumerate(options): @@ -91,29 +94,29 @@ def _generate_html(self, quiz_id, title, descriptions, options, correct_pairs, s {option} ''' - + return f'''
{title}
- +
{description_zones}
- +
-
Ziehen Sie die Elemente zu den passenden Beschreibungen.
+
Ziehen Sie diese zu den passenden Beschreibungen
{draggable_options}
- +
- +
''' - - \ No newline at end of file diff --git a/quadriga/colors.py b/quadriga/colors.py index 9d01053..5713035 100644 --- a/quadriga/colors.py +++ b/quadriga/colors.py @@ -1,4 +1,4 @@ -"""This submodule contains all QUADRIGA color presets for the various libraries used.""" +"""QUADRIGA color presets for various libraries used.""" jupyterquiz = { "--jq-multiple-choice-bg": "#00305e", diff --git a/quadriga/metadata/__init__.py b/quadriga/metadata/__init__.py index fc1ab8d..3a5ea8f 100644 --- a/quadriga/metadata/__init__.py +++ b/quadriga/metadata/__init__.py @@ -7,8 +7,8 @@ __all__ = [ "create_bibtex", + "create_citation_cff", "extract_from_book_config", - "update_citation_cff", "update_version_from_tag", "utils", ] @@ -16,8 +16,8 @@ # Import the modules to make their functions available from . import ( create_bibtex, + create_citation_cff, extract_from_book_config, - update_citation_cff, update_version_from_tag, utils, ) diff --git a/quadriga/metadata/create_bibtex.py b/quadriga/metadata/create_bibtex.py index aa960ac..eab11e2 100644 --- a/quadriga/metadata/create_bibtex.py +++ b/quadriga/metadata/create_bibtex.py @@ -1,83 +1,44 @@ +""" +Create the CITATION.bib file from metadata.yml. + +metadata.yml is the single source of truth for all metadata. This script +rebuilds CITATION.bib from it completely on every run. +""" + from __future__ import annotations import logging import sys +from datetime import datetime, timezone from .utils import ( extract_keywords, format_authors_for_bibtex, generate_citation_key, + get_content_license, + get_doi, get_file_path, + get_languages, + get_publication_year, load_yaml_file, ) logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") logger = logging.getLogger(__name__) -# Map CFF types to BibTeX entry types -CFF_TO_BIBTEX_TYPES = { - # Academic publications - "article": "article", # Journal article - "magazine-article": "article", # Magazine article - "newspaper-article": "article", # Newspaper article - "book": "book", # Complete book - "edited-work": "incollection", # Edited work should be incollection (contribution in book) - "conference-paper": "inproceedings", # Conference paper - "proceedings": "proceedings", # Conference proceedings - "conference": "proceedings", # Conference (same as proceedings) - "thesis": "thesis", # Generic thesis - will be refined in the code based on type field - "report": "techreport", # Technical report - "pamphlet": "booklet", # Short printed work - "unpublished": "unpublished", # Unpublished work - "manual": "manual", # Technical documentation/manual - # Digital resources - "software": "misc", # Software/code - use misc with howpublished field - "software-code": "misc", # Software source code - "software-container": "misc", # Software container - "software-executable": "misc", # Executable software - "software-virtual-machine": "misc", # Software VM - "data": "misc", # Dataset - "database": "misc", # Database - "website": "misc", # Website - use misc with howpublished=URL - "blog": "misc", # Blog post - use misc with howpublished=URL - # Media and audiovisual - "art": "misc", # Artwork - "audiovisual": "misc", # Audiovisual material - "film-broadcast": "misc", # Film or broadcast - "sound-recording": "misc", # Sound recording - "video": "misc", # Video - "multimedia": "misc", # Multimedia - "music": "misc", # Music - "slides": "misc", # Presentation slides - # Reference works - "catalogue": "book", # Catalogue - "dictionary": "book", # Dictionary - "encyclopedia": "book", # Encyclopedia - "map": "misc", # Map - # Legal and government - "bill": "misc", # Legislative bill - "legal-case": "misc", # Legal case - "legal-rule": "misc", # Legal rule or regulation - "government-document": "techreport", # Government document - "hearing": "misc", # Hearing - "statute": "misc", # Statute - "standard": "misc", # Standard (could be techreport in some contexts) - "patent": "patent", # Patent - supported in some BibTeX styles - # Other types - "generic": "misc", # Generic document - "grant": "misc", # Grant - "historical-work": "book", # Historical work - "personal-communication": "misc", # Personal communication - "serial": "periodical", # Serial publication -} +def _bibtex_field(name: str, value: str) -> str: + """Format a single BibTeX field line, sanitizing braces in the value.""" + clean_value = str(value).replace("{", "").replace("}", "") + return f" {name:<9} = {{{clean_value}}}," -def create_bibtex_from_cff() -> bool | None: + +def create_bibtex_from_metadata() -> bool: """ - Create a CITATION.bib file from CITATION.cff. + Create a CITATION.bib file from metadata.yml. - Reads citation data, prioritizing the 'preferred-citation' block if available, - formats authors, generates a citation key, and constructs a BibTeX entry. + QUADRIGA OERs are published as books (Jupyter Books), so the entry type + is '@book'. Returns ------- @@ -87,75 +48,41 @@ def create_bibtex_from_cff() -> bool | None: # Define file paths using utility functions try: repo_root = get_file_path("") # Get repo root - citation_cff_path = get_file_path("CITATION.cff", repo_root) + metadata_path = get_file_path("metadata.yml", repo_root) citation_bib_path = get_file_path("CITATION.bib", repo_root) except Exception: logger.exception("Failed to resolve file paths") return False - # Check if citation_cff_path exists - if not citation_cff_path.exists(): - logger.error("CITATION.cff file not found at %s", citation_cff_path) + if not metadata_path.exists(): + logger.error("metadata.yml file not found at %s", metadata_path) return False - # Read CITATION.cff using utility function - citation_data = load_yaml_file(citation_cff_path) + metadata = load_yaml_file(metadata_path) - if not citation_data or not isinstance(citation_data, dict): - logger.error("Could not load CITATION.cff or invalid format. Exiting.") + if not metadata or not isinstance(metadata, dict): + logger.error("Could not load metadata.yml or invalid format. Exiting.") return False - # Extract data from preferred-citation or root - if "preferred-citation" in citation_data: - logger.info("Using 'preferred-citation' section from CITATION.cff") - pref = citation_data.get("preferred-citation") - if not isinstance(pref, dict): - logger.error("preferred-citation is not a dictionary") - return False - else: - logger.info("No 'preferred-citation' section found, using root data") - pref = citation_data - - # Validate required fields - authors = pref.get("authors", []) - title = pref.get("title", "Untitled") - year = str(pref.get("year", "")) # Ensure year is a string for generate_citation_key + title = metadata.get("title", "") + if not title: + logger.warning("No title found in metadata.yml") + authors = metadata.get("authors", []) if not authors: - logger.warning("No authors found in CITATION.cff") - - if title == "Untitled": - logger.warning("No title found in CITATION.cff, using 'Untitled'") + logger.warning("No authors found in metadata.yml") + year = get_publication_year(metadata) if not year: - logger.warning("No year found in CITATION.cff") + year = str(datetime.now(tz=timezone.utc).year) + logger.info("No date-modified or date-issued found, using current year: %s", year) - # Use utility function to format authors try: author_str = format_authors_for_bibtex(authors) except Exception: logger.exception("Error formatting authors") author_str = "" - # Choose entry type based on type field - cff_type = pref.get("type", "software") # Default to software if not specified - - # Get the entry type from the mapping, default to 'misc' if not found - entry_type = CFF_TO_BIBTEX_TYPES.get(cff_type.lower(), "misc") - - # Special handling for thesis types - if entry_type == "thesis": - # Check for thesis type information - thesis_type = pref.get("thesis-type", "").lower() - if thesis_type in {"master", "masters", "master's"}: - entry_type = "mastersthesis" - else: - # Default to phdthesis if type is not specified or is something else - entry_type = "phdthesis" - - logger.info("Converting CFF type '%s' to BibTeX entry type: %s", cff_type, entry_type) - - # Use utility function to generate citation key try: citation_key = generate_citation_key(authors, title, year) except Exception: @@ -163,162 +90,37 @@ def create_bibtex_from_cff() -> bool | None: citation_key = "Unknown_Citation_Key" # Compile BibTeX entry - bibtex_lines = [f"@{entry_type}{{{citation_key},"] + bibtex_lines = [f"@book{{{citation_key},"] - # Add fields - if title != "Untitled": # Only add title if it's not the default placeholder - bibtex_lines.append(f" title = {{{title}}},") + if title: + bibtex_lines.append(_bibtex_field("title", title)) if author_str: - bibtex_lines.append(f" author = {{{author_str}}},") + bibtex_lines.append(_bibtex_field("author", author_str)) if year: - bibtex_lines.append(f" year = {{{year}}},") - if "version" in pref: - bibtex_lines.append(f" version = {{{pref['version']}}},") - - # Define common fields for all entry types - simple_fields = [ - "doi", - "url", - "copyright", - "publisher", - "address", - "edition", - "isbn", - ] - - # Add entry-specific fields based on the entry type - if entry_type == "article": - article_fields = ["journal", "volume", "number", "pages", "month", "issn"] - simple_fields.extend(article_fields) - - # Map CFF journal-specific fields to BibTeX fields - if "collection-title" in pref and "journal" not in pref: - bibtex_lines.append(f" journal = {{{pref['collection-title']}}},") - if "volume-title" in pref and "journal" not in pref and "collection-title" not in pref: - bibtex_lines.append(f" journal = {{{pref['volume-title']}}},") - - elif entry_type in ["inproceedings", "proceedings"]: - conf_fields = [ - "booktitle", - "series", - "volume", - "pages", - "month", - "organization", - ] - simple_fields.extend(conf_fields) + bibtex_lines.append(_bibtex_field("year", year)) + if "version" in metadata: + bibtex_lines.append(_bibtex_field("version", metadata["version"])) - # Map CFF conference-specific fields to BibTeX fields - if "conference" in pref and "booktitle" not in pref: - conf_name = ( - pref["conference"].get("name", "") - if isinstance(pref["conference"], dict) - else str(pref["conference"]) - ) - if conf_name: - bibtex_lines.append(f" booktitle = {{{conf_name}}},") - - if "collection-title" in pref and "booktitle" not in pref: - bibtex_lines.append(f" booktitle = {{{pref['collection-title']}}},") - - elif entry_type in ["phdthesis", "mastersthesis"]: - thesis_fields = ["school", "type", "month"] - simple_fields.extend(thesis_fields) - - # Add institution as school if present - if "institution" in pref and "school" not in pref: - institution = ( - pref["institution"].get("name", "") - if isinstance(pref["institution"], dict) - else str(pref["institution"]) - ) - if institution: - bibtex_lines.append(f" school = {{{institution}}},") - - elif entry_type == "techreport": - report_fields = ["institution", "number", "type"] - simple_fields.extend(report_fields) - - elif entry_type == "incollection": - # Add fields specific to book chapters - incollection_fields = [ - "booktitle", - "editor", - "chapter", - "pages", - "publisher", - "address", - ] - simple_fields.extend(incollection_fields) - - # Map collection-title to booktitle if present - if "collection-title" in pref and "booktitle" not in pref: - bibtex_lines.append(f" booktitle = {{{pref['collection-title']}}},") - - # Map collection editor if available - if "collection-editors" in pref and isinstance(pref["collection-editors"], list): - try: - editor_str = format_authors_for_bibtex(pref["collection-editors"]) - bibtex_lines.append(f" editor = {{{editor_str}}},") - except (KeyError, TypeError, AttributeError) as e: - logger.warning("Error formatting collection editors: %s", e) - - # Special handling for software, code, data entries - if cff_type.lower().startswith("software") or cff_type.lower() in [ - "data", - "database", - ]: - soft_fields = ["version", "note"] - simple_fields.extend(soft_fields) - - # Add repository info to note field if available - if "repository-code" in pref and "note" not in pref: - bibtex_lines.append(f" note = {{Repository: {pref['repository-code']}}},") - - # Note: version is already added in the common fields section above - - # Add software-specific details as howpublished if not present - if ("howpublished" not in pref) and ("repository-code" in pref or "url" in pref): - repo = pref.get("repository-code", pref.get("url", "")) - bibtex_lines.append(f" howpublished = {{Available from: {repo}}},") - - # Special handling for websites and blogs - elif cff_type.lower() in ["website", "blog"]: - # For websites and blogs, ensure the URL is included - if "url" in pref: - bibtex_lines.append(f" howpublished = {{\\url{{{pref['url']}}}}},") + doi = get_doi(metadata) + if doi: + bibtex_lines.append(_bibtex_field("doi", doi)) + else: + logger.warning("No DOI found in metadata.yml 'identifier' field") - # Include last accessed date if available - if "date-accessed" in pref: - bibtex_lines.append(f" note = {{Accessed: {pref['date-accessed']}}},") + if "url" in metadata: + bibtex_lines.append(_bibtex_field("url", metadata["url"])) - # Process all simple fields - for field in simple_fields: - if field in pref and field not in [ - "version", - "institution", - ]: # Skip already handled fields - # Sanitize field value to avoid BibTeX syntax errors - field_value = str(pref[field]).replace("{", "").replace("}", "") - bibtex_lines.append(f" {field:<9} = {{{field_value}}},") + license_id = get_content_license(metadata) + if license_id: + bibtex_lines.append(_bibtex_field("copyright", license_id)) - # Handle list fields like languages - if pref.get("languages"): - try: - languages_str = ", ".join(pref["languages"]) - bibtex_lines.append(f" language = {{{languages_str}}},") - except (TypeError, AttributeError) as e: - logger.warning("Error processing languages field: %s", e) + languages = get_languages(metadata) + if languages: + bibtex_lines.append(_bibtex_field("language", ", ".join(languages))) - # Handle keywords field - if pref.get("keywords"): - try: - keywords_list = extract_keywords(pref["keywords"]) - if keywords_list: - keywords_str = ", ".join(keywords_list) - bibtex_lines.append(f" keywords = {{{keywords_str}}},") - except (TypeError, AttributeError) as e: - logger.warning("Error processing keywords field: %s", e) + keywords = extract_keywords(metadata.get("keywords")) + if keywords: + bibtex_lines.append(_bibtex_field("keywords", ", ".join(keywords))) # Close the entry bibtex_lines.append("}") @@ -336,10 +138,10 @@ def create_bibtex_from_cff() -> bool | None: return True except Exception: - logger.exception("Unexpected error in create_bibtex_from_cff") + logger.exception("Unexpected error in create_bibtex_from_metadata") return False if __name__ == "__main__": - success = create_bibtex_from_cff() + success = create_bibtex_from_metadata() sys.exit(0 if success else 1) diff --git a/quadriga/metadata/create_citation_cff.py b/quadriga/metadata/create_citation_cff.py new file mode 100644 index 0000000..872e68a --- /dev/null +++ b/quadriga/metadata/create_citation_cff.py @@ -0,0 +1,204 @@ +""" +Generate the CITATION.cff file from metadata.yml. + +metadata.yml is the single source of truth for all metadata. This script +rebuilds CITATION.cff from it completely on every run — manual edits to +CITATION.cff will be overwritten. + +The generated file contains a root citation (CFF type 'software', as required +by the CFF 1.2.0 schema for repository citations) and a 'preferred-citation' +of type 'book' that carries the full citation for the OER itself, including +DOI, year, language and license. +""" + +from __future__ import annotations + +import logging +import sys +from pathlib import Path + +from .utils import ( + extract_keywords, + get_content_license, + get_doi, + get_file_path, + get_languages, + get_publication_year, + load_yaml_file, + save_yaml_file, +) + +logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") +logger = logging.getLogger(__name__) + + +def _cff_person(person: dict) -> dict: + """ + Convert a metadata.yml author/contributor entry to a CFF person. + + Copies only the fields CFF understands (drops e.g. 'credit'). + + Args: + person: Author or contributor dictionary from metadata.yml + + Returns + ------- + dict: CFF person entry + """ + cff_person = {} + for key in ("given-names", "family-names", "orcid", "affiliation"): + if key in person: + cff_person[key] = person[key] + return cff_person + + +def build_citation_data(metadata: dict) -> dict: + """ + Build the complete CITATION.cff data structure from metadata.yml. + + Args: + metadata: Parsed metadata.yml data + + Returns + ------- + dict: A valid CFF 1.2.0 data structure + """ + title = metadata.get("title", "Untitled") + if title == "Untitled": + logger.warning("No title found in metadata.yml, using 'Untitled'") + + authors = [person for person in (_cff_person(a) for a in metadata.get("authors", [])) if person] + if not authors: + logger.warning("No authors found in metadata.yml") + authors = [{"name": "Unknown"}] + + keywords = extract_keywords(metadata.get("keywords")) + doi = get_doi(metadata) + if not doi: + logger.warning("No DOI found in metadata.yml 'identifier' field") + license_id = get_content_license(metadata) + year = get_publication_year(metadata) + languages = get_languages(metadata) + + citation_data: dict = { + "cff-version": "1.2.0", + "title": title, + } + + if "description" in metadata: + citation_data["abstract"] = metadata["description"] + + citation_data["type"] = "software" + citation_data["message"] = "Please cite this software using the metadata from `preferred-citation` in `CITATION.cff`." + citation_data["authors"] = authors + + if doi: + citation_data["identifiers"] = [{"type": "doi", "value": doi, "description": "Zenodo"}] + + if "git" in metadata: + citation_data["repository-code"] = metadata["git"] + + if "url" in metadata: + citation_data["url"] = metadata["url"] + + if keywords: + citation_data["keywords"] = keywords + + if license_id: + citation_data["license"] = license_id + + # QUADRIGA OERs are built with Jupyter Book — cite it as a reference + if metadata.get("learning-resource-type") == "Jupyter Book": + citation_data["references"] = [ + { + "title": "Jupyter Book", + "type": "software", + "authors": [ + { + "name": "The Jupyter Book Community", + "website": "https://github.com/jupyter-book/jupyter-book/graphs/contributors", + } + ], + } + ] + + # The preferred citation for the OER itself (a book, not the repository). + # Authors and keywords reuse the same objects as the root citation, which + # PyYAML serializes as anchors/aliases. + preferred_citation: dict = {} + if year: + preferred_citation["year"] = year + preferred_citation["authors"] = authors + preferred_citation["title"] = title + preferred_citation["type"] = "book" + if doi: + preferred_citation["doi"] = doi + if "url" in metadata: + preferred_citation["url"] = metadata["url"] + if "git" in metadata: + preferred_citation["repository-code"] = metadata["git"] + if license_id: + preferred_citation["license"] = license_id + if languages: + preferred_citation["languages"] = languages + if license_id: + preferred_citation["copyright"] = license_id + if keywords: + preferred_citation["keywords"] = keywords + if "version" in metadata: + preferred_citation["version"] = metadata["version"] + + citation_data["preferred-citation"] = preferred_citation + + if "version" in metadata: + citation_data["version"] = metadata["version"] + + return citation_data + + +def create_citation_cff() -> bool: + """ + Generate the CITATION.cff file from metadata.yml. + + Returns + ------- + bool: True if successful, False otherwise. + """ + try: + # Define file paths + try: + repo_root = get_file_path("") # Get repo root by providing empty relative path + metadata_path = get_file_path("metadata.yml", repo_root) + citation_cff_path = get_file_path("CITATION.cff", repo_root) + except Exception: + logger.exception("Failed to resolve file paths") + return False + + # metadata.yml must exist + if not Path(metadata_path).exists(): + logger.error("Required file metadata.yml not found at %s", metadata_path) + return False + + # Load metadata.yml + metadata = load_yaml_file(metadata_path) + + if not metadata or not isinstance(metadata, dict): + logger.error("Could not load metadata.yml or invalid format. Exiting.") + return False + + citation_data = build_citation_data(metadata) + + return save_yaml_file( + citation_cff_path, + citation_data, + schema_comment="# yaml-language-server: $schema=https://citation-file-format.github.io/1.2.0/schema.json", + ) + + except Exception: + logger.exception("Unexpected error in create_citation_cff") + return False + + +if __name__ == "__main__": + success = create_citation_cff() + sys.exit(0 if success else 1) diff --git a/quadriga/metadata/create_jsonld.py b/quadriga/metadata/create_jsonld.py index e7b8017..9e09510 100644 --- a/quadriga/metadata/create_jsonld.py +++ b/quadriga/metadata/create_jsonld.py @@ -17,7 +17,7 @@ from pathlib import Path from typing import Any -from .utils import extract_keywords, get_file_path, get_repo_root, load_yaml_file +from .utils import clean_orcid, extract_keywords, get_doi, get_file_path, get_repo_root, load_yaml_file logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") logger = logging.getLogger(__name__) @@ -41,54 +41,6 @@ def build_jsonld_context() -> dict[str, str]: } -def clean_orcid(orcid_string: str) -> str | None: - """ - Extract ORCID identifier from an ORCID string or URL. - - Args: - orcid_string (str): ORCID string which may include URL prefix - - Returns - ------- - str: Clean ORCID identifier (e.g., "0000-0002-1602-6032") - """ - if not orcid_string: - return None - - orcid = str(orcid_string) - prefixes = ["https://orcid.org/", "http://orcid.org/", "orcid:"] - for prefix in prefixes: - if orcid.startswith(prefix): - orcid = orcid[len(prefix) :] - break - - return orcid.strip() - - -def clean_doi(doi_string: str) -> str | None: - """ - Extract DOI identifier from a DOI string or URL. - - Args: - doi_string (str): DOI string which may include URL prefix - - Returns - ------- - str: Clean DOI identifier (e.g., "10.5281/zenodo.14970672") - """ - if not doi_string: - return None - - doi = str(doi_string) - prefixes = ["https://doi.org/", "http://doi.org/", "doi:"] - for prefix in prefixes: - if doi.startswith(prefix): - doi = doi[len(prefix) :] - break - - return doi.strip() - - def transform_person(person_data: Any) -> dict[str, Any]: """ Transform author or contributor to Schema.org Person. @@ -390,16 +342,17 @@ def create_jsonld() -> bool | None: logger.info("Added description") # identifier (DOI) -> schema:identifier (exactMatch) - if "identifier" in metadata: - clean_doi_id = clean_doi(metadata["identifier"]) - if clean_doi_id: - jsonld["identifier"] = { - "@type": "PropertyValue", - "propertyID": "DOI", - "value": clean_doi_id, - "url": metadata["identifier"], - } - logger.info("Added DOI identifier: %s", clean_doi_id) + doi = get_doi(metadata) + if doi: + jsonld["identifier"] = { + "@type": "PropertyValue", + "propertyID": "DOI", + "value": doi, + "url": f"https://doi.org/{doi}", + } + logger.info("Added DOI identifier: %s", doi) + else: + logger.warning("No DOI found in metadata.yml 'identifier' field") # version -> schema:version (exactMatch) if "version" in metadata: @@ -516,9 +469,7 @@ def create_jsonld() -> bool | None: # target-group -> schema:audience (closeMatch) and lrmi:educationalAudience (closeMatch) if metadata.get("target-group"): - jsonld["audience"] = [ - {"@type": "Audience", "audienceType": group} for group in metadata["target-group"] - ] + jsonld["audience"] = [{"@type": "Audience", "audienceType": group} for group in metadata["target-group"]] logger.info("Added %d target groups", len(jsonld["audience"])) # time-required -> schema:timeRequired (exactMatch) diff --git a/quadriga/metadata/create_rdfxml.py b/quadriga/metadata/create_rdfxml.py index e075256..12d70a6 100644 --- a/quadriga/metadata/create_rdfxml.py +++ b/quadriga/metadata/create_rdfxml.py @@ -17,10 +17,16 @@ from pathlib import Path from typing import Any -from rdflib import RDF, Graph, Literal, Namespace, URIRef # type: ignore[import-not-found] +from rdflib import ( # type: ignore[import-not-found] + RDF, + Graph, + Literal, + Namespace, + URIRef, +) from rdflib.namespace import DCTERMS, SKOS, XSD # type: ignore[import-not-found] -from .utils import extract_keywords, get_file_path, get_repo_root, load_yaml_file +from .utils import clean_orcid, extract_keywords, get_doi, get_file_path, get_repo_root, load_yaml_file logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") logger = logging.getLogger(__name__) @@ -49,57 +55,7 @@ def _sort_xml_element(element: ET.Element) -> None: element[:] = children -def clean_orcid(orcid_string: str) -> str | None: - """ - Extract ORCID identifier from an ORCID string or URL. - - Args: - orcid_string (str): ORCID string which may include URL prefix - - Returns - ------- - str: Clean ORCID identifier (e.g., "0000-0002-1602-6032") - """ - if not orcid_string: - return None - - orcid = str(orcid_string) - prefixes = ["https://orcid.org/", "http://orcid.org/", "orcid:"] - for prefix in prefixes: - if orcid.startswith(prefix): - orcid = orcid[len(prefix) :] - break - - return orcid.strip() - - -def clean_doi(doi_string: str) -> str | None: - """ - Extract DOI identifier from a DOI string or URL. - - Args: - doi_string (str): DOI string which may include URL prefix - - Returns - ------- - str: Clean DOI identifier (e.g., "10.5281/zenodo.14970672") - """ - if not doi_string: - return None - - doi = str(doi_string) - prefixes = ["https://doi.org/", "http://doi.org/", "doi:"] - for prefix in prefixes: - if doi.startswith(prefix): - doi = doi[len(prefix) :] - break - - return doi.strip() - - -def add_person( - graph: Graph, person_data: Any, base_uri: str, person_type: str, index: int -) -> URIRef | None: +def add_person(graph: Graph, person_data: Any, base_uri: str, person_type: str, index: int) -> URIRef | None: """ Add a person (author or contributor) to the RDF graph. @@ -153,9 +109,7 @@ def add_person( graph.add((orcid_node, RDF.type, SCHEMA.PropertyValue)) graph.add((orcid_node, SCHEMA.propertyID, Literal("ORCID"))) graph.add((orcid_node, SCHEMA.value, Literal(clean_orcid_id))) - graph.add( - (orcid_node, SCHEMA.url, URIRef(f"https://orcid.org/{clean_orcid_id}")) - ) + graph.add((orcid_node, SCHEMA.url, URIRef(f"https://orcid.org/{clean_orcid_id}"))) graph.add((person_uri, SCHEMA.identifier, orcid_node)) # affiliation -> schema:affiliation (mapped in both author and contributor) @@ -236,9 +190,7 @@ def add_learning_objective( return obj_uri -def add_chapter( - graph: Graph, chapter_data: Any, base_uri: str, chapter_index: int -) -> URIRef | None: +def add_chapter(graph: Graph, chapter_data: Any, base_uri: str, chapter_index: int) -> URIRef | None: """ Add a chapter to the RDF graph as a LearningResource. @@ -290,9 +242,7 @@ def add_chapter( # learning-objectives -> educationalAlignment with AlignmentObject if chapter_data.get("learning-objectives"): for obj_index, obj_data in enumerate(chapter_data["learning-objectives"]): - obj_uri = add_learning_objective( - graph, obj_data, base_uri, chapter_index, obj_index - ) + obj_uri = add_learning_objective(graph, obj_data, base_uri, chapter_index, obj_index) if obj_uri: graph.add((chapter_uri, SCHEMA.educationalAlignment, obj_uri)) @@ -376,17 +326,18 @@ def create_rdfxml() -> bool | None: logger.info("Added description") # identifier (DOI) -> schema:identifier (exactMatch) - if "identifier" in metadata: - clean_doi_id = clean_doi(metadata["identifier"]) - if clean_doi_id: - # Create PropertyValue node for DOI - doi_node = URIRef(f"{base_uri}#doi") - graph.add((doi_node, RDF.type, SCHEMA.PropertyValue)) - graph.add((doi_node, SCHEMA.propertyID, Literal("DOI"))) - graph.add((doi_node, SCHEMA.value, Literal(clean_doi_id))) - graph.add((doi_node, SCHEMA.url, URIRef(metadata["identifier"]))) - graph.add((resource_uri, SCHEMA.identifier, doi_node)) - logger.info("Added DOI identifier: %s", clean_doi_id) + doi = get_doi(metadata) + if doi: + # Create PropertyValue node for DOI + doi_node = URIRef(f"{base_uri}#doi") + graph.add((doi_node, RDF.type, SCHEMA.PropertyValue)) + graph.add((doi_node, SCHEMA.propertyID, Literal("DOI"))) + graph.add((doi_node, SCHEMA.value, Literal(doi))) + graph.add((doi_node, SCHEMA.url, URIRef(f"https://doi.org/{doi}"))) + graph.add((resource_uri, SCHEMA.identifier, doi_node)) + logger.info("Added DOI identifier: %s", doi) + else: + logger.warning("No DOI found in metadata.yml 'identifier' field") # version -> schema:version (exactMatch) if "version" in metadata: @@ -395,9 +346,7 @@ def create_rdfxml() -> bool | None: # schema-version -> schema:schemaVersion if "schema-version" in metadata: - graph.add( - (resource_uri, SCHEMA.schemaVersion, Literal(str(metadata["schema-version"]))) - ) + graph.add((resource_uri, SCHEMA.schemaVersion, Literal(str(metadata["schema-version"])))) logger.info("Added schema version: %s", metadata["schema-version"]) # url -> schema:url (exactMatch) @@ -548,9 +497,7 @@ def create_rdfxml() -> bool | None: ) ) elif isinstance(content_license_data, str): - graph.add( - (content_license_node, SCHEMA.license, URIRef(content_license_data)) - ) + graph.add((content_license_node, SCHEMA.license, URIRef(content_license_data))) graph.add((resource_uri, SCHEMA.license, content_license_node)) logger.info("Added license information") @@ -566,9 +513,7 @@ def create_rdfxml() -> bool | None: # table-of-contents -> dcterms:tableOfContents (exactMatch) if "table-of-contents" in metadata: - graph.add( - (resource_uri, DCTERMS.tableOfContents, Literal(metadata["table-of-contents"])) - ) + graph.add((resource_uri, DCTERMS.tableOfContents, Literal(metadata["table-of-contents"]))) logger.info("Added table of contents") # ===== ADDITIONAL METADATA ===== @@ -616,7 +561,7 @@ def create_rdfxml() -> bool | None: ET.register_namespace(prefix, uri) # Parse, sort elements recursively, and re-serialize - root = ET.fromstring(xml_str) # noqa: S314 — parsing our own rdflib output + root = ET.fromstring(xml_str) _sort_xml_element(root) ET.indent(root, space=" ") diff --git a/quadriga/metadata/create_zenodo_json.py b/quadriga/metadata/create_zenodo_json.py index c11c97d..6cf96cb 100644 --- a/quadriga/metadata/create_zenodo_json.py +++ b/quadriga/metadata/create_zenodo_json.py @@ -1,9 +1,9 @@ """ -Creates a .zenodo.json file from CITATION.cff and metadata.yml. +Creates a .zenodo.json file from metadata.yml. -This script reads citation data from the 'preferred-citation' section of -CITATION.cff and additional metadata from metadata.yml to generate a Zenodo -deposit metadata file following the Zenodo JSON schema. +metadata.yml is the single source of truth for all metadata. This script +generates a Zenodo deposit metadata file from it following the Zenodo JSON +schema. The upload_type is set to "lesson" as specified for QUADRIGA OERs. """ @@ -15,61 +15,20 @@ import sys from typing import Any -from .utils import extract_keywords, get_file_path, get_repo_root, load_yaml_file +from .utils import ( + clean_orcid, + extract_keywords, + get_content_license, + get_file_path, + get_languages, + get_publication_date, + get_repo_root, + load_yaml_file, +) logger = logging.getLogger(__name__) -def clean_doi(doi_string: str) -> str | None: - """ - Extract DOI identifier from a DOI string or URL. - - Args: - doi_string (str): DOI string which may include URL prefix - - Returns - ------- - str: Clean DOI identifier (e.g., "10.5281/zenodo.14970672") - """ - if not doi_string: - return None - - # Remove common DOI URL prefixes - doi = str(doi_string) - prefixes = ["https://doi.org/", "http://doi.org/", "doi:"] - for prefix in prefixes: - if doi.startswith(prefix): - doi = doi[len(prefix) :] - break - - return doi.strip() - - -def clean_orcid(orcid_string: str) -> str | None: - """ - Extract ORCID identifier from an ORCID string or URL. - - Args: - orcid_string (str): ORCID string which may include URL prefix - - Returns - ------- - str: Clean ORCID identifier (e.g., "0000-0002-1602-6032") - """ - if not orcid_string: - return None - - # Remove common ORCID URL prefixes - orcid = str(orcid_string) - prefixes = ["https://orcid.org/", "http://orcid.org/", "orcid:"] - for prefix in prefixes: - if orcid.startswith(prefix): - orcid = orcid[len(prefix) :] - break - - return orcid.strip() - - def format_creators_for_zenodo(authors: list) -> list: """ Format authors list for Zenodo creators field. @@ -173,11 +132,9 @@ def format_contributors_for_zenodo(contributors: list | None) -> list: def create_zenodo_json() -> bool | None: """ - Create a .zenodo.json file from CITATION.cff and metadata.yml. + Create a .zenodo.json file from metadata.yml. - Reads the 'preferred-citation' section from CITATION.cff - and combines it with data from metadata.yml to create a Zenodo-compliant - metadata file. The upload_type is always set to "lesson" for QUADRIGA OERs. + The upload_type is always set to "lesson" for QUADRIGA OERs. Returns ------- @@ -187,58 +144,35 @@ def create_zenodo_json() -> bool | None: # Define file paths try: repo_root = get_repo_root() # Get repo root - citation_cff_path = get_file_path("CITATION.cff", repo_root) metadata_path = get_file_path("metadata.yml", repo_root) zenodo_json_path = get_file_path(".zenodo.json", repo_root) except Exception: logger.exception("Failed to resolve file paths") return False - # Check if required files exist - if not citation_cff_path.exists(): - logger.error("CITATION.cff file not found at %s", citation_cff_path) - return False - if not metadata_path.exists(): logger.error("metadata.yml file not found at %s", metadata_path) return False - # Load CITATION.cff - citation_data = load_yaml_file(citation_cff_path) - if not citation_data or not isinstance(citation_data, dict): - logger.error("Could not load CITATION.cff or invalid format. Exiting.") - return False - # Load metadata.yml metadata = load_yaml_file(metadata_path) if not metadata or not isinstance(metadata, dict): logger.error("Could not load metadata.yml or invalid format. Exiting.") return False - # Extract data from preferred-citation or root - if "preferred-citation" in citation_data: - logger.info("Using 'preferred-citation' section from CITATION.cff") - pref = citation_data.get("preferred-citation") - if not isinstance(pref, dict): - logger.error("preferred-citation is not a dictionary") - return False - else: - logger.info("No 'preferred-citation' section found, using root data") - pref = citation_data - zenodo_metadata: dict[str, Any] = {"upload_type": "lesson"} # title - if "title" in pref: - zenodo_metadata["title"] = pref["title"] - logger.info("Added title: %s", pref["title"]) + if "title" in metadata: + zenodo_metadata["title"] = metadata["title"] + logger.info("Added title: %s", metadata["title"]) else: - logger.error("No title found in CITATION.cff") + logger.error("No title found in metadata.yml") return False # creators - if pref.get("authors"): - creators = format_creators_for_zenodo(pref["authors"]) + if metadata.get("authors"): + creators = format_creators_for_zenodo(metadata["authors"]) if creators: zenodo_metadata["creators"] = creators logger.info("Added %d creators", len(creators)) @@ -246,10 +180,10 @@ def create_zenodo_json() -> bool | None: logger.error("Could not format any creators from authors") return False else: - logger.error("No authors found in preferred-citation") + logger.error("No authors found in metadata.yml") return False - # description + # description - must exist in metadata or schema checks block it description = "

" + metadata.get("description") + "

" description_base = f""" @@ -279,61 +213,38 @@ def create_zenodo_json() -> bool | None: logger.info("Added description") # publication date - publication_date = None - if "date-modified" in metadata: - # Use date-modified from metadata.yml and convert to string if needed - date_value = metadata["date-modified"] - # Handle both date objects and strings - if hasattr(date_value, "isoformat"): - # It's a date/datetime object, convert to ISO format string - publication_date = date_value.isoformat() - else: - # It's already a string - publication_date = str(date_value) - logger.info("Added publication_date from metadata.yml: %s", publication_date) - elif "year" in pref: - # Fall back to year from CITATION.cff - year = str(pref["year"]) - # Zenodo expects ISO 8601 date format (YYYY-MM-DD) - # We use January 1st as default when only year is provided - publication_date = f"{year}-01-01" - logger.info("Added publication_date from year (fallback): %s", publication_date) - else: - logger.warning("No publication date or year found") + publication_date = get_publication_date(metadata) if publication_date: zenodo_metadata["publication_date"] = publication_date + logger.info("Added publication_date: %s", publication_date) + else: + logger.warning("No publication date found") # Note: DOI field is intentionally NOT included in .zenodo.json # Zenodo assigns DOIs automatically upon upload/release. # Including a self-referencing DOI would be circular and incorrect. # keywords - if pref.get("keywords"): - keywords_list = extract_keywords(pref["keywords"]) + if metadata.get("keywords"): + keywords_list = extract_keywords(metadata["keywords"]) if keywords_list: zenodo_metadata["keywords"] = keywords_list logger.info("Added %d keywords", len(keywords_list)) # license - license_id = None - if "license" in pref: - license_id = pref["license"] - elif "copyright" in pref: - license_id = pref["copyright"] + license_id = get_content_license(metadata) if license_id: - # Zenodo expects license IDs like "CC-BY-SA-4.0" + # Zenodo expects license IDs like "CC-BY-4.0" # Clean up common variations - license_clean = str(license_id).upper().replace("_", "-") + license_clean = str(license_id).upper().replace("_", "-").replace(" ", "-") zenodo_metadata["license"] = license_clean logger.info("Added license: %s", license_clean) - # language - if pref.get("languages"): - lang = ( - pref["languages"][0] if isinstance(pref["languages"], list) else pref["languages"] - ) - zenodo_metadata["language"] = lang - logger.info("Added language: %s", lang) + # language (Zenodo expects ISO 639-2/3 codes, e.g. "deu") + languages = get_languages(metadata) + if languages: + zenodo_metadata["language"] = languages[0] + logger.info("Added language: %s", languages[0]) # contributors if metadata.get("contributors"): @@ -344,17 +255,13 @@ def create_zenodo_json() -> bool | None: # related_identifiers related_identifiers = [] - repo_url = pref.get("repository-code") + repo_url = metadata.get("git") if repo_url: - related_identifiers.append( - {"identifier": repo_url, "relation": "isSupplementedBy", "scheme": "url"} - ) + related_identifiers.append({"identifier": repo_url, "relation": "isSupplementedBy", "scheme": "url"}) logger.info("Added repository URL as related identifier") - url = pref.get("url") + url = metadata.get("url") if url and url != repo_url: - related_identifiers.append( - {"identifier": url, "relation": "isAlternateIdentifier", "scheme": "url"} - ) + related_identifiers.append({"identifier": url, "relation": "isAlternateIdentifier", "scheme": "url"}) logger.info("Added URL as related identifier") if related_identifiers: @@ -365,9 +272,9 @@ def create_zenodo_json() -> bool | None: logger.info("Added QUADRIGA community") # version - if "version" in pref: - zenodo_metadata["version"] = str(pref["version"]) - logger.info("Added version: %s", pref["version"]) + if "version" in metadata: + zenodo_metadata["version"] = str(metadata["version"]) + logger.info("Added version: %s", metadata["version"]) # write .zenodo.json try: diff --git a/quadriga/metadata/extract_from_book_config.py b/quadriga/metadata/extract_from_book_config.py index ea3946d..5cd36b2 100644 --- a/quadriga/metadata/extract_from_book_config.py +++ b/quadriga/metadata/extract_from_book_config.py @@ -16,6 +16,7 @@ get_file_path, get_repo_root, load_yaml_file, + resolve_toc_file, save_yaml_file, ) @@ -89,23 +90,17 @@ def extract_and_update() -> bool | None: continue try: - # Get the file path as a Path object file_path_str = chapter["file"] - p = Path(file_path_str) - # Ensure the file has an extension (default to .md if none) - if p.suffix not in [".md", ".ipynb"]: - p = p.with_suffix(".md") - - # Create the full path to the file - full_path = get_file_path(p, repo_root) + # Resolve the _toc.yml entry to an existing file + full_path = resolve_toc_file(file_path_str, repo_root) # Check if file exists if not full_path.exists(): missing_files.append(str(full_path)) logger.warning("Chapter file not found: %s", full_path) # Use filename as fallback title - toc_chapters.append(f"[Missing: {p.stem}]") + toc_chapters.append(f"[Missing: {Path(file_path_str).name}]") continue # Extract the chapter title from the file's first heading @@ -117,7 +112,7 @@ def extract_and_update() -> bool | None: logger.exception("Error processing chapter %s", chapter.get("file", "unknown")) # Add a placeholder with the filename if possible try: - toc_chapters.append(f"[Error: {p.stem}]") + toc_chapters.append(f"[Error: {Path(chapter['file']).name}]") except Exception: toc_chapters.append("[Error: unknown chapter]") diff --git a/quadriga/metadata/extract_from_lernziele.py b/quadriga/metadata/extract_from_lernziele.py index 90a7892..bd7aee7 100644 --- a/quadriga/metadata/extract_from_lernziele.py +++ b/quadriga/metadata/extract_from_lernziele.py @@ -1,11 +1,20 @@ """Extract learning objectives with metadata from Lernziele.md files.""" from __future__ import annotations + import logging import re from pathlib import Path from typing import Any -from .utils import get_repo_root, save_yaml_file, get_file_path, load_yaml_file, iter_toc_files + +from .utils import ( + get_file_path, + get_repo_root, + iter_toc_files, + load_yaml_file, + resolve_toc_file, + save_yaml_file, +) logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") logger = logging.getLogger(__name__) @@ -25,20 +34,20 @@ def parse_metadata_comment(comment: str) -> dict[str, str]: """Parse 'competency: X | bloom: Y' from the inner text of an HTML comment.""" metadata = {} - for part in comment.split('|'): + for part in comment.split("|"): part = part.strip() - if ':' not in part: + if ":" not in part: continue - key, value = part.split(':', 1) - key = key.strip().lower().replace(' ', '-') + key, value = part.split(":", 1) + key = key.strip().lower().replace(" ", "-") value = value.strip() if not value: continue - if key == 'bloom': - metadata['blooms-category'] = value - elif key == 'competency': + if key == "bloom": + metadata["blooms-category"] = value + elif key == "competency": metadata[key] = value - metadata['data-flow'] = derive_data_flow(value) + metadata["data-flow"] = derive_data_flow(value) return metadata @@ -47,14 +56,14 @@ def validate_objective_metadata(objective_data: dict[str, Any]) -> list[str]: """Fill in missing competency/bloom/data-flow defaults and return names of missing fields.""" missing_fields = [] - if not objective_data.get('competency'): - missing_fields.append('competency') - objective_data['competency'] = DEFAULT_COMPETENCY - objective_data['data-flow'] = DEFAULT_DATA_FLOW + if not objective_data.get("competency"): + missing_fields.append("competency") + objective_data["competency"] = DEFAULT_COMPETENCY + objective_data["data-flow"] = DEFAULT_DATA_FLOW - if not objective_data.get('blooms-category'): - missing_fields.append('blooms-category') - objective_data['blooms-category'] = DEFAULT_BLOOM + if not objective_data.get("blooms-category"): + missing_fields.append("blooms-category") + objective_data["blooms-category"] = DEFAULT_BLOOM return missing_fields @@ -92,7 +101,9 @@ def extract_admonition_blocks( validation_issues = [] # Pattern to match admonition blocks preceded by - admonition_pattern = r'\s*\n```\{admonition\}\s+(.+?)\n((?::[^\n]+\n)*)((?:(?!```).)+)```' + admonition_pattern = ( + r"\s*\n```\{admonition\}\s+(.+?)\n((?::[^\n]+\n)*)((?:(?!```).)+)```" + ) matches = re.finditer(admonition_pattern, content, re.DOTALL | re.MULTILINE) @@ -102,7 +113,7 @@ def extract_admonition_blocks( body = match.group(4).strip() # Parse title - extract text and reference - title_match = re.match(r'\[(.+?)\]\((.+?)\)(\s*\(\*(.+?)\*\))?', title_line) + title_match = re.match(r"\[(.+?)\]\((.+?)\)(\s*\(\*(.+?)\*\))?", title_line) if not title_match: logger.warning("Could not parse admonition title: %s", title_line) @@ -111,8 +122,10 @@ def extract_admonition_blocks( section_title = title_match.group(1) # Learning goal + # ((?:(?!-->).)+?) instead of (.+?) so the match can never run past + # the end of the comment into an adjacent one learning_goal_match = re.search( - r'', + r").)+?)\s*-->", body, re.DOTALL, ) @@ -120,34 +133,34 @@ def extract_admonition_blocks( learning_goal = normalize_whitespace(learning_goal_match.group(1)) else: learning_goal = "TODO" - validation_issues.append({ - 'section': section_title, - 'missing_fields': ['learning-goal'] - }) + validation_issues.append({"section": section_title, "missing_fields": ["learning-goal"]}) # Strip all START/END markers before parsing objectives - body_cleaned = re.sub(r'\s*', '', body) - body_cleaned = re.sub(r'\s*', '', body_cleaned) + body_cleaned = re.sub(r"\s*", "", body) + body_cleaned = re.sub(r"\s*", "", body_cleaned) - # Parse numbered objectives with optional inline metadata comment + # Parse numbered objectives with optional inline metadata comment. + # The comment group must not cross a --> boundary, and the lookahead + # accepts a following comment so a stray extra comment between + # objectives doesn't get swallowed into the objective text. objectives = [] - objective_pattern = r'\d+\.\s+(.+?)(?:(?:\n\s*|(?=)?(?=\n\d+\.|\n\n|$)' + objective_pattern = ( + r"\d+\.\s+(.+?)(?:(?:\n\s*|(?=).)+?)\s*-->)?(?=\n\d+\.|\n\s*