diff --git a/content/posts/2020-03.md b/content/posts/2020-03.md index 890980c1a..c7549079d 100644 --- a/content/posts/2020-03.md +++ b/content/posts/2020-03.md @@ -37,4 +37,32 @@ $ ./fix-metadata-values.py -i 2020-03-04-fix-1-ilri-subject.csv -db dspace -u ds - But I have not run it on CGSpace yet because we want to ask Peter if he is sure about it... - Send a message to Macaroni Bros to ask them about their Drupal module and its readiness for DSpace 6 UUIDs +## 2020-03-05 + +- I found a very [interesting comment on the Solr 8.1 guide](https://lucene.apache.org/solr/guide/8_1/solr-system-requirements.html#lucene-solr-prior-to-7-0) about Java compatibility: + +> Lucene/Solr 7.0 was the first version that successfully passed our tests using Java 9 and higher. You should avoid Java 9 or later for Lucene/Solr 6.x or earlier. + +## 2020-03-08 + +- I want to try to consolidate our yearly Solr statistics cores back into one `statistics` core using the solr-import-export-json tool +- I will try it on DSpace test, doing one year at a time: + +``` +$ ./run.sh -s http://localhost:8081/solr/statistics-2010 -a export -o /tmp/statistics-2010.json -k uid +$ ./run.sh -s http://localhost:8081/solr/statistics -a import -o /tmp/statistics-2010.json -k uid +$ curl -s "http://localhost:8081/solr/statistics-2010/update?softCommit=true" -H "Content-Type: text/xml" --data-binary "time:2010*" +$ ./run.sh -s http://localhost:8081/solr/statistics-2011 -a export -o /tmp/statistics-2011.json -k uid +$ ./run.sh -s http://localhost:8081/solr/statistics -a import -o /tmp/statistics-2011.json -k uid +$ curl -s "http://localhost:8081/solr/statistics-2011/update?softCommit=true" -H "Content-Type: text/xml" --data-binary "time:2011*" +$ ./run.sh -s http://localhost:8081/solr/statistics -a import -o /tmp/statistics-2012.json -k uid +$ curl -s 'http://localhost:8081/solr/statistics/select?q=time:2012*&rows=0&wt=json&indent=true' | grep numFound + "response":{"numFound":3761989,"start":0,"docs":[] +$ curl -s 'http://localhost:8081/solr/statistics-2012/select?q=time:2012*&rows=0&wt=json&indent=true' | grep numFound + "response":{"numFound":3761989,"start":0,"docs":[] +$ curl -s "http://localhost:8081/solr/statistics-2012/update?softCommit=true" -H "Content-Type: text/xml" --data-binary "time:2012*" +``` + +- I will do this for as many cores as I can (disk space limited) and then monitor the effect on the system and JVM memory usage + diff --git a/docs/2020-02/index.html b/docs/2020-02/index.html index 58204068c..59c18ccaa 100644 --- a/docs/2020-02/index.html +++ b/docs/2020-02/index.html @@ -20,7 +20,7 @@ The code finally builds and runs with a fresh install - + @@ -47,7 +47,7 @@ The code finally builds and runs with a fresh install "url": "https:\/\/alanorth.github.io\/cgspace-notes\/2020-02\/", "wordCount": "7238", "datePublished": "2020-02-02T11:56:30+02:00", - "dateModified": "2020-03-03T18:56:09+02:00", + "dateModified": "2020-03-04T18:02:54+02:00", "author": { "@type": "Person", "name": "Alan Orth" diff --git a/docs/2020-03/index.html b/docs/2020-03/index.html index e734f9886..81157f80d 100644 --- a/docs/2020-03/index.html +++ b/docs/2020-03/index.html @@ -22,7 +22,7 @@ You need to download this into the DSpace 6.x source and compile it - + @@ -49,9 +49,9 @@ You need to download this into the DSpace 6.x source and compile it "@type": "BlogPosting", "headline": "March, 2020", "url": "https:\/\/alanorth.github.io\/cgspace-notes\/2020-03\/", - "wordCount": "162", + "wordCount": "358", "datePublished": "2020-03-02T12:31:30+02:00", - "dateModified": "2020-03-02T12:38:10+02:00", + "dateModified": "2020-03-04T18:02:54+02:00", "author": { "@type": "Person", "name": "Alan Orth" @@ -162,6 +162,33 @@ $ ~/dspace63/bin/dspace solr-upgrade-statistics-6x
  • But I have not run it on CGSpace yet because we want to ask Peter if he is sure about it…
  • Send a message to Macaroni Bros to ask them about their Drupal module and its readiness for DSpace 6 UUIDs
  • +

    2020-03-05

    + +
    +

    Lucene/Solr 7.0 was the first version that successfully passed our tests using Java 9 and higher. You should avoid Java 9 or later for Lucene/Solr 6.x or earlier.

    +
    +

    2020-03-08

    + +
    $ ./run.sh -s http://localhost:8081/solr/statistics-2010 -a export -o /tmp/statistics-2010.json -k uid
    +$ ./run.sh -s http://localhost:8081/solr/statistics -a import -o /tmp/statistics-2010.json -k uid
    +$ curl -s "http://localhost:8081/solr/statistics-2010/update?softCommit=true" -H "Content-Type: text/xml" --data-binary "<delete><query>time:2010*</query></delete>"
    +$ ./run.sh -s http://localhost:8081/solr/statistics-2011 -a export -o /tmp/statistics-2011.json -k uid
    +$ ./run.sh -s http://localhost:8081/solr/statistics -a import -o /tmp/statistics-2011.json -k uid
    +$ curl -s "http://localhost:8081/solr/statistics-2011/update?softCommit=true" -H "Content-Type: text/xml" --data-binary "<delete><query>time:2011*</query></delete>"
    +$ ./run.sh -s http://localhost:8081/solr/statistics -a import -o /tmp/statistics-2012.json -k uid
    +$ curl -s 'http://localhost:8081/solr/statistics/select?q=time:2012*&rows=0&wt=json&indent=true' | grep numFound
    +  "response":{"numFound":3761989,"start":0,"docs":[]
    +$ curl -s 'http://localhost:8081/solr/statistics-2012/select?q=time:2012*&rows=0&wt=json&indent=true' | grep numFound
    +  "response":{"numFound":3761989,"start":0,"docs":[]
    +$ curl -s "http://localhost:8081/solr/statistics-2012/update?softCommit=true" -H "Content-Type: text/xml" --data-binary "<delete><query>time:2012*</query></delete>"
    +
    diff --git a/docs/sitemap.xml b/docs/sitemap.xml index 0ffa5b306..00f279a9e 100644 --- a/docs/sitemap.xml +++ b/docs/sitemap.xml @@ -4,32 +4,32 @@ https://alanorth.github.io/cgspace-notes/categories/ - 2020-03-03T18:56:09+02:00 + 2020-03-04T18:02:54+02:00 https://alanorth.github.io/cgspace-notes/ - 2020-03-03T18:56:09+02:00 + 2020-03-04T18:02:54+02:00 https://alanorth.github.io/cgspace-notes/2020-03/ - 2020-03-02T12:38:10+02:00 + 2020-03-04T18:02:54+02:00 https://alanorth.github.io/cgspace-notes/categories/notes/ - 2020-03-03T18:56:09+02:00 + 2020-03-04T18:02:54+02:00 https://alanorth.github.io/cgspace-notes/posts/ - 2020-03-03T18:56:09+02:00 + 2020-03-04T18:02:54+02:00 https://alanorth.github.io/cgspace-notes/2020-02/ - 2020-03-03T18:56:09+02:00 + 2020-03-04T18:02:54+02:00