2021-02-25 16:45:46 +01:00
|
|
|
# https://github.com/opencultureconsulting/openrefine-task-runner
|
2020-08-01 02:04:39 +02:00
|
|
|
|
|
|
|
version: '3'
|
|
|
|
|
2021-02-25 16:45:46 +01:00
|
|
|
includes:
|
|
|
|
alephino: alephino
|
|
|
|
barcodes: barcodes
|
|
|
|
bibliotheca: bibliotheca
|
|
|
|
pica+: pica+
|
2020-08-01 02:04:39 +02:00
|
|
|
|
2021-02-25 16:45:46 +01:00
|
|
|
silent: true
|
|
|
|
output: prefixed
|
2020-08-01 02:04:39 +02:00
|
|
|
|
|
|
|
env:
|
2021-02-25 16:45:46 +01:00
|
|
|
OPENREFINE:
|
|
|
|
sh: readlink -m .openrefine/refine
|
|
|
|
CLIENT:
|
|
|
|
sh: readlink -m .openrefine/client
|
2020-08-01 02:04:39 +02:00
|
|
|
|
|
|
|
tasks:
|
|
|
|
default:
|
2021-02-25 16:45:46 +01:00
|
|
|
desc: Datenverarbeitung sequentiell
|
2020-08-01 02:04:39 +02:00
|
|
|
cmds:
|
2021-02-25 16:45:46 +01:00
|
|
|
- task: alephino:main
|
|
|
|
- task: bibliotheca:main
|
|
|
|
- task: pica+:refine
|
2020-08-01 02:04:39 +02:00
|
|
|
|
2021-02-25 16:45:46 +01:00
|
|
|
install:
|
|
|
|
desc: (re)install OpenRefine and openrefine-client into subdirectory .openrefine
|
2020-11-09 16:12:35 +01:00
|
|
|
cmds:
|
2021-02-25 16:45:46 +01:00
|
|
|
- | # delete existing install and recreate folder
|
|
|
|
rm -rf .openrefine
|
|
|
|
mkdir -p .openrefine
|
|
|
|
- > # download OpenRefine archive
|
|
|
|
wget --no-verbose -O openrefine.tar.gz
|
|
|
|
https://github.com/OpenRefine/OpenRefine/releases/download/3.4.1/openrefine-linux-3.4.1.tar.gz
|
|
|
|
- | # install OpenRefine into subdirectory .openrefine
|
|
|
|
tar -xzf openrefine.tar.gz -C .openrefine --strip 1
|
|
|
|
rm openrefine.tar.gz
|
|
|
|
- | # optimize OpenRefine for batch processing
|
|
|
|
sed -i 's/cd `dirname $0`/cd "$(dirname "$0")"/' ".openrefine/refine" # fix path issue in OpenRefine startup file
|
|
|
|
sed -i '$ a JAVA_OPTIONS=-Drefine.headless=true' ".openrefine/refine.ini" # do not try to open OpenRefine in browser
|
|
|
|
sed -i 's/#REFINE_AUTOSAVE_PERIOD=60/REFINE_AUTOSAVE_PERIOD=1440/' ".openrefine/refine.ini" # set autosave period from 5 minutes to 25 hours
|
|
|
|
- > # download openrefine-client into subdirectory .openrefine
|
|
|
|
wget --no-verbose -O .openrefine/client
|
|
|
|
https://github.com/opencultureconsulting/openrefine-client/releases/download/v0.3.10/openrefine-client_0-3-10_linux
|
|
|
|
- chmod +x .openrefine/client # make client executable
|
|
|
|
|
|
|
|
start:
|
|
|
|
dir: ./{{.DIR}}
|
2020-11-09 16:12:35 +01:00
|
|
|
cmds:
|
2021-02-25 16:45:46 +01:00
|
|
|
- | # verify that OpenRefine is installed
|
|
|
|
if [ ! -f "$OPENREFINE" ]; then
|
|
|
|
echo 1>&2 "OpenRefine missing; try task install"; exit 1
|
|
|
|
fi
|
|
|
|
- | # delete temporary files and log file of previous run
|
|
|
|
rm -rf ./*.project* workspace.json
|
|
|
|
rm -rf "{{.PROJECT}}.log"
|
|
|
|
- > # launch OpenRefine with specific data directory and redirect its output to a log file
|
|
|
|
"$OPENREFINE" -v warn -p {{.PORT}} -m {{.RAM}}
|
|
|
|
-d ../{{.DIR}}
|
|
|
|
>> "{{.PROJECT}}.log" 2>&1 &
|
|
|
|
- | # wait until OpenRefine API is available
|
|
|
|
timeout 30s bash -c "until
|
|
|
|
wget -q -O - http://localhost:{{.PORT}} | cat | grep -q -o OpenRefine
|
|
|
|
do sleep 1
|
|
|
|
done"
|
|
|
|
|
|
|
|
stop:
|
|
|
|
dir: ./{{.DIR}}
|
2020-08-01 02:04:39 +02:00
|
|
|
cmds:
|
2021-02-25 16:45:46 +01:00
|
|
|
- | # shut down OpenRefine gracefully
|
|
|
|
PID=$(lsof -t -i:{{.PORT}})
|
|
|
|
kill $PID
|
|
|
|
while ps -p $PID > /dev/null; do sleep 1; done
|
|
|
|
- > # archive the OpenRefine project
|
|
|
|
tar cfz
|
|
|
|
"{{.PROJECT}}.openrefine.tar.gz"
|
|
|
|
-C $(grep -l "{{.PROJECT}}" *.project/metadata.json | cut -d '/' -f 1)
|
|
|
|
.
|
|
|
|
- rm -rf ./*.project* workspace.json # delete temporary files
|
|
|
|
|
|
|
|
kill:
|
|
|
|
dir: ./{{.DIR}}
|
2020-08-01 02:04:39 +02:00
|
|
|
cmds:
|
2021-02-25 16:45:46 +01:00
|
|
|
- | # shut down OpenRefine immediately to save time and disk space
|
|
|
|
PID=$(lsof -t -i:{{.PORT}})
|
|
|
|
kill -9 $PID
|
|
|
|
while ps -p $PID > /dev/null; do sleep 1; done
|
|
|
|
- rm -rf ./*.project* workspace.json # delete temporary files
|
|
|
|
|
|
|
|
check:
|
|
|
|
desc: check OpenRefine log for any warnings and exit on error
|
|
|
|
dir: ./{{.DIR}}
|
2021-02-03 11:54:09 +01:00
|
|
|
cmds:
|
2021-02-25 16:45:46 +01:00
|
|
|
- | # find log file(s) and check for "exception" or "error"
|
|
|
|
if grep -i 'exception\|error' $(find . -name '*.log'); then
|
|
|
|
echo 1>&2 "log contains warnings!"; exit 1
|
|
|
|
fi
|