Skip to content

Commit

Permalink
Added some tools for working with GERBIL data.
Browse files Browse the repository at this point in the history
  • Loading branch information
MichaelRoeder committed Sep 2, 2024
1 parent fd83ddd commit ab8e720
Show file tree
Hide file tree
Showing 2 changed files with 132 additions and 0 deletions.
62 changes: 62 additions & 0 deletions src/main/java/org/aksw/gerbil/tools/QA2C2KBConverter.java
Original file line number Diff line number Diff line change
@@ -0,0 +1,62 @@
package org.aksw.gerbil.tools;

import java.io.BufferedOutputStream;
import java.io.FileOutputStream;
import java.io.OutputStream;
import java.util.stream.Collectors;

import org.aksw.gerbil.config.GerbilConfiguration;
import org.aksw.gerbil.dataset.Dataset;
import org.aksw.gerbil.dataset.DatasetConfiguration;
import org.aksw.gerbil.datatypes.ExperimentType;
import org.aksw.gerbil.exceptions.GerbilException;
import org.aksw.gerbil.io.nif.NIFWriter;
import org.aksw.gerbil.io.nif.impl.TurtleNIFWriter;
import org.aksw.gerbil.qa.datatypes.Property;
import org.aksw.gerbil.transfer.nif.Document;
import org.aksw.gerbil.transfer.nif.Meaning;
import org.aksw.gerbil.web.config.AdapterManager;
import org.aksw.gerbil.web.config.AnnotatorsConfig;
import org.aksw.gerbil.web.config.DatasetsConfig;

public class QA2C2KBConverter {

public static void main(String[] args) {
GerbilConfiguration.loadAdditionalProperties("src/main/properties/gerbil.properties");
run();
}

public static void run() {
Dataset dataset = loadDataset();
if (dataset == null) {
return;
}
for (Document document : dataset.getInstances()) {
// Remove all property markings
document.setMarkings(document.getMarkings(Meaning.class).stream().filter(m -> !Property.class.isInstance(m))
.collect(Collectors.toList()));
document.setDocumentURI(document.getDocumentURI().replace('#', '-'));
}
try (OutputStream out = new BufferedOutputStream(new FileOutputStream("qald10-test-nif.ttl"))) {
NIFWriter writer = new TurtleNIFWriter();
writer.writeNIF(dataset.getInstances(), out);
} catch (Exception e) {
e.printStackTrace();
}
}

public static Dataset loadDataset() {
AdapterManager adapterManager = new AdapterManager();
adapterManager.setAnnotators(AnnotatorsConfig.annotators());
adapterManager.setDatasets(DatasetsConfig.datasets(null, null));
DatasetConfiguration datasetConfig = adapterManager.getDatasetConfig("QALD10 Test Multilingual",
ExperimentType.QA, "en");
try {
return datasetConfig.getDataset(ExperimentType.QA);
} catch (GerbilException e) {
e.printStackTrace();
}
return null;
}

}
70 changes: 70 additions & 0 deletions src/main/java/org/aksw/gerbil/tools/QALDExporter.java
Original file line number Diff line number Diff line change
@@ -0,0 +1,70 @@
package org.aksw.gerbil.tools;

import java.io.FileWriter;
import java.io.Writer;

import org.aksw.gerbil.config.GerbilConfiguration;
import org.aksw.gerbil.dataset.Dataset;
import org.aksw.gerbil.dataset.DatasetConfiguration;
import org.aksw.gerbil.datatypes.ExperimentType;
import org.aksw.gerbil.exceptions.GerbilException;
import org.aksw.gerbil.transfer.nif.Document;
import org.aksw.gerbil.web.config.AdapterManager;
import org.aksw.gerbil.web.config.AnnotatorsConfig;
import org.aksw.gerbil.web.config.DatasetsConfig;
import org.apache.jena.rdf.model.Model;
import org.apache.jena.rdf.model.ModelFactory;
import org.apache.jena.rdf.model.Resource;
import org.apache.jena.vocabulary.RDF;
import org.apache.jena.vocabulary.RDFS;

public class QALDExporter {

public static final String QALD10_BASE_IRI = "https://github.com/qald_10#";

public static void main(String[] args) {
GerbilConfiguration.loadAdditionalProperties("src/main/properties/gerbil.properties");
run();
}

public static void run() {
Dataset dataset = loadDataset();
if(dataset == null) {
return;
}
Model model = ModelFactory.createDefaultModel();
for (Document document : dataset.getInstances()) {
addQuestion(model, document);
}
model.setNsPrefix("qa", QALD10_BASE_IRI);
model.setNsPrefix("rdf", RDF.getURI());
model.setNsPrefix("rdfs", RDFS.getURI());
try (Writer writer = new FileWriter("qald10-simple.nt")) {
model.write(writer, "nt");
} catch (Exception e) {
e.printStackTrace();
}
}

public static Dataset loadDataset() {
AdapterManager adapterManager = new AdapterManager();
adapterManager.setAnnotators(AnnotatorsConfig.annotators());
adapterManager.setDatasets(DatasetsConfig.datasets(null, null));
DatasetConfiguration datasetConfig = adapterManager.getDatasetConfig("QALD10 Test Multilingual", ExperimentType.QA, "en");
try {
return datasetConfig.getDataset(ExperimentType.QA);
} catch (GerbilException e) {
e.printStackTrace();
}
return null;
}

private static void addQuestion(Model model, Document document) {
Resource questionClass = model.getResource("http://quans-namespace.org/#QUESTION");

Resource questionResource = model.createResource(QALD10_BASE_IRI + "QUE" + document.getDocumentURI().substring("http://qa.gerbil.aksw.org/QALD10+Test+Multilingual/question#".length()));
model.add(questionResource, RDF.type, questionClass);
model.add(questionResource, RDF.value, document.getText());
model.add(questionResource, RDFS.comment, document.getText());
}
}

0 comments on commit ab8e720

Please sign in to comment.