Readable workflows need simple data | F1000Research "use strict";function _typeof(t){return(_typeof="function"==typeof Symbol&&"symbol"==typeof Symbol.iterator?function(t){return typeof t}:function(t){return t&&"function"==typeof Symbol&&t.constructor===Symbol&&t!==Symbol.prototype?"symbol":typeof t})(t)}!function(){var t=function(){var t,e,o=[],n=window,r=n;for(;r;){try{if(r.frames.__tcfapiLocator){t=r;break}}catch(t){}if(r===n.top)break;r=r.parent}t||(!function t(){var e=n.document,o=!!n.frames.__tcfapiLocator;if(!o)if(e.body){var r=e.createElement("iframe");r.style.cssText="display:none",r.name="__tcfapiLocator",e.body.appendChild(r)}else setTimeout(t,5);return!o}(),n.__tcfapi=function(){for(var t=arguments.length,n=new Array(t),r=0;r 3&&2===parseInt(n[1],10)&&"boolean"==typeof n[3]&&(e=n[3],"function"==typeof n[2]&&n[2]("set",!0)):"ping"===n[0]?"function"==typeof n[2]&&n[2]({gdprApplies:e,cmpLoaded:!1,cmpStatus:"stub"}):o.push(n)},n.addEventListener("message",(function(t){var e="string"==typeof t.data,o={};if(e)try{o=JSON.parse(t.data)}catch(t){}else o=t.data;var n="object"===_typeof(o)&&null!==o?o.__tcfapiCall:null;n&&window.__tcfapi(n.command,n.version,(function(o,r){var a={__tcfapiReturn:{returnValue:o,success:r,callId:n.callId}};t&&t.source&&t.source.postMessage&&t.source.postMessage(e?JSON.stringify(a):a,"*")}),n.parameter)}),!1))};"undefined"!=typeof module?module.exports=t:t()}(); dataLayer = dataLayer || []; // Standard GTM initialization - Google Consent Mode handles consent automatically (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start': new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0], j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src= 'https://www.googletagmanager.com/gtm.js?id='+i+dl+ '>m_auth=hzk0Vc3qFsQYhCrIoHz68A>m_preview=env-1>m_cookies_win=x';f.parentNode.insertBefore(j,f); })(window,document,'script','dataLayer','GTM-MWFK8L5J'); ;window.NREUM||(NREUM={});NREUM.init={distributed_tracing:{enabled:true},privacy:{cookies_enabled:true},ajax:{deny_list:["bam.nr-data.net"]}}; ;NREUM.loader_config={accountID:"438030",trustKey:"438030",agentID:"772317073",licenseKey:"97f8f67f26",applicationID:"772317073"} ;NREUM.info={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net",licenseKey:"97f8f67f26",applicationID:"772317073",sa:1} ;/*! For license information please see nr-loader-spa-1.236.0.min.js.LICENSE.txt */ (()=>{"use strict";var e,t,r={5763:(e,t,r)=>{r.d(t,{P_:()=>l,Mt:()=>g,C5:()=>s,DL:()=>v,OP:()=>T,lF:()=>D,Yu:()=>y,Dg:()=>h,CX:()=>c,GE:()=>b,sU:()=>_});var n=r(8632),i=r(9567);const o={beacon:n.ce.beacon,errorBeacon:n.ce.errorBeacon,licenseKey:void 0,applicationID:void 0,sa:void 0,queueTime:void 0,applicationTime:void 0,ttGuid:void 0,user:void 0,account:void 0,product:void 0,extra:void 0,jsAttributes:{},userAttributes:void 0,atts:void 0,transactionName:void 0,tNamePlain:void 0},a={};function s(e){if(!e)throw new Error("All info objects require an agent identifier!");if(!a[e])throw new Error("Info for ".concat(e," was never set"));return a[e]}function c(e,t){if(!e)throw new Error("All info objects require an agent identifier!");a[e]=(0,i.D)(t,o),(0,n.Qy)(e,a[e],"info")}var u=r(7056);const d=()=>{const e={blockSelector:"[data-nr-block]",maskInputOptions:{password:!0}};return{allow_bfcache:!0,privacy:{cookies_enabled:!0},ajax:{deny_list:void 0,enabled:!0,harvestTimeSeconds:10},distributed_tracing:{enabled:void 0,exclude_newrelic_header:void 0,cors_use_newrelic_header:void 0,cors_use_tracecontext_headers:void 0,allowed_origins:void 0},session:{domain:void 0,expiresMs:u.oD,inactiveMs:u.Hb},ssl:void 0,obfuscate:void 0,jserrors:{enabled:!0,harvestTimeSeconds:10},metrics:{enabled:!0},page_action:{enabled:!0,harvestTimeSeconds:30},page_view_event:{enabled:!0},page_view_timing:{enabled:!0,harvestTimeSeconds:30,long_task:!1},session_trace:{enabled:!0,harvestTimeSeconds:10},harvest:{tooManyRequestsDelay:60},session_replay:{enabled:!1,harvestTimeSeconds:60,sampleRate:.1,errorSampleRate:.1,maskTextSelector:"*",maskAllInputs:!0,get blockClass(){return"nr-block"},get ignoreClass(){return"nr-ignore"},get maskTextClass(){return"nr-mask"},get blockSelector(){return e.blockSelector},set blockSelector(t){e.blockSelector+=",".concat(t)},get maskInputOptions(){return e.maskInputOptions},set maskInputOptions(t){e.maskInputOptions={...t,password:!0}}},spa:{enabled:!0,harvestTimeSeconds:10}}},f={};function l(e){if(!e)throw new Error("All configuration objects require an agent identifier!");if(!f[e])throw new Error("Configuration for ".concat(e," was never set"));return f[e]}function h(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");f[e]=(0,i.D)(t,d()),(0,n.Qy)(e,f[e],"config")}function g(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");var r=l(e);if(r){for(var n=t.split("."),i=0;i {r.d(t,{D:()=>i});var n=r(50);function i(e,t){try{if(!e||"object"!=typeof e)return(0,n.Z)("Setting a Configurable requires an object as input");if(!t||"object"!=typeof t)return(0,n.Z)("Setting a Configurable requires a model to set its initial properties");const r=Object.create(Object.getPrototypeOf(t),Object.getOwnPropertyDescriptors(t)),o=0===Object.keys(r).length?e:r;for(let a in o)if(void 0!==e[a])try{"object"==typeof e[a]&&"object"==typeof t[a]?r[a]=i(e[a],t[a]):r[a]=e[a]}catch(e){(0,n.Z)("An error occurred while setting a property of a Configurable",e)}return r}catch(e){(0,n.Z)("An error occured while setting a Configurable",e)}}},6818:(e,t,r)=>{r.d(t,{Re:()=>i,gF:()=>o,q4:()=>n});const n="1.236.0",i="PROD",o="CDN"},385:(e,t,r)=>{r.d(t,{FN:()=>a,IF:()=>u,Nk:()=>f,Tt:()=>s,_A:()=>o,il:()=>n,pL:()=>c,v6:()=>i,w1:()=>d});const n="undefined"!=typeof window&&!!window.document,i="undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self.navigator instanceof WorkerNavigator||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis.navigator instanceof WorkerNavigator),o=n?window:"undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis),a=""+o?.location,s=/iPad|iPhone|iPod/.test(navigator.userAgent),c=s&&"undefined"==typeof SharedWorker,u=(()=>{const e=navigator.userAgent.match(/Firefox[/\s](\d+\.\d+)/);return Array.isArray(e)&&e.length>=2?+e[1]:0})(),d=Boolean(n&&window.document.documentMode),f=!!navigator.sendBeacon},1117:(e,t,r)=>{r.d(t,{w:()=>o});var n=r(50);const i={agentIdentifier:"",ee:void 0};class o{constructor(e){try{if("object"!=typeof e)return(0,n.Z)("shared context requires an object as input");this.sharedContext={},Object.assign(this.sharedContext,i),Object.entries(e).forEach((e=>{let[t,r]=e;Object.keys(i).includes(t)&&(this.sharedContext[t]=r)}))}catch(e){(0,n.Z)("An error occured while setting SharedContext",e)}}}},8e3:(e,t,r)=>{r.d(t,{L:()=>d,R:()=>c});var n=r(2177),i=r(1284),o=r(4322),a=r(3325);const s={};function c(e,t){const r={staged:!1,priority:a.p[t]||0};u(e),s[e].get(t)||s[e].set(t,r)}function u(e){e&&(s[e]||(s[e]=new Map))}function d(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:"",t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:"feature";if(u(e),!e||!s[e].get(t))return a(t);s[e].get(t).staged=!0;const r=[...s[e]];function a(t){const r=e?n.ee.get(e):n.ee,a=o.X.handlers;if(r.backlog&&a){var s=r.backlog[t],c=a[t];if(c){for(var u=0;s&&u {let[t,r]=e;return r.staged}))&&(r.sort(((e,t)=>e[1].priority-t[1].priority)),r.forEach((e=>{let[t]=e;a(t)})))}function f(e,t){var r=e[1];(0,i.D)(t[r],(function(t,r){var n=e[0];if(r[0]===n){var i=r[1],o=e[3],a=e[2];i.apply(o,a)}}))}},2177:(e,t,r)=>{r.d(t,{c:()=>f,ee:()=>u});var n=r(8632),i=r(2210),o=r(1284),a=r(5763),s="nr@context";let c=(0,n.fP)();var u;function d(){}function f(e){return(0,i.X)(e,s,l)}function l(){return new d}function h(){u.aborted=!0,u.backlog={}}c.ee?u=c.ee:(u=function e(t,r){var n={},c={},f={},g=!1;try{g=16===r.length&&(0,a.OP)(r).isolatedBacklog}catch(e){}var p={on:b,addEventListener:b,removeEventListener:y,emit:v,get:x,listeners:w,context:m,buffer:A,abort:h,aborted:!1,isBuffering:E,debugId:r,backlog:g?{}:t&&"object"==typeof t.backlog?t.backlog:{}};return p;function m(e){return e&&e instanceof d?e:e?(0,i.X)(e,s,l):l()}function v(e,r,n,i,o){if(!1!==o&&(o=!0),!u.aborted||i){t&&o&&t.emit(e,r,n);for(var a=m(n),s=w(e),d=s.length,f=0;fn,p:()=>i});var n=r(2177).ee.get("handle");function i(e,t,r,i,o){o?(o.buffer([e],i),o.emit(e,t,r)):(n.buffer([e],i),n.emit(e,t,r))}},4322:(e,t,r)=>{r.d(t,{X:()=>o});var n=r(5546);o.on=a;var i=o.handlers={};function o(e,t,r,o){a(o||n.E,i,e,t,r)}function a(e,t,r,i,o){o||(o="feature"),e||(e=n.E);var a=t[o]=t[o]||{};(a[r]=a[r]||[]).push([e,i])}},3239:(e,t,r)=>{r.d(t,{bP:()=>s,iz:()=>c,m$:()=>a});var n=r(385);let i=!1,o=!1;try{const e={get passive(){return i=!0,!1},get signal(){return o=!0,!1}};n._A.addEventListener("test",null,e),n._A.removeEventListener("test",null,e)}catch(e){}function a(e,t){return i||o?{capture:!!e,passive:i,signal:t}:!!e}function s(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;window.addEventListener(e,t,a(r,n))}function c(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;document.addEventListener(e,t,a(r,n))}},4402:(e,t,r)=>{r.d(t,{Ht:()=>u,M:()=>c,Rl:()=>a,ky:()=>s});var n=r(385);const i="xxxxxxxx-xxxx-4xxx-yxxx-xxxxxxxxxxxx";function o(e,t){return e?15&e[t]:16*Math.random()|0}function a(){const e=n._A?.crypto||n._A?.msCrypto;let t,r=0;return e&&e.getRandomValues&&(t=e.getRandomValues(new Uint8Array(31))),i.split("").map((e=>"x"===e?o(t,++r).toString(16):"y"===e?(3&o()|8).toString(16):e)).join("")}function s(e){const t=n._A?.crypto||n._A?.msCrypto;let r,i=0;t&&t.getRandomValues&&(r=t.getRandomValues(new Uint8Array(31)));const a=[];for(var s=0;s {r.d(t,{Bq:()=>n,Hb:()=>o,oD:()=>i});const n="NRBA",i=144e5,o=18e5},7894:(e,t,r)=>{function n(){return Math.round(performance.now())}r.d(t,{z:()=>n})},7243:(e,t,r)=>{r.d(t,{e:()=>o});var n=r(385),i={};function o(e){if(e in i)return i[e];if(0===(e||"").indexOf("data:"))return{protocol:"data"};let t;var r=n._A?.location,o={};if(n.il)t=document.createElement("a"),t.href=e;else try{t=new URL(e,r.href)}catch(e){return o}o.port=t.port;var a=t.href.split("://");!o.port&&a[1]&&(o.port=a[1].split("/")[0].split("@").pop().split(":")[1]),o.port&&"0"!==o.port||(o.port="https"===a[0]?"443":"80"),o.hostname=t.hostname||r.hostname,o.pathname=t.pathname,o.protocol=a[0],"/"!==o.pathname.charAt(0)&&(o.pathname="/"+o.pathname);var s=!t.protocol||":"===t.protocol||t.protocol===r.protocol,c=t.hostname===r.hostname&&t.port===r.port;return o.sameOrigin=s&&(!t.hostname||c),"/"===o.pathname&&(i[e]=o),o}},50:(e,t,r)=>{function n(e,t){"function"==typeof console.warn&&(console.warn("New Relic: ".concat(e)),t&&console.warn(t))}r.d(t,{Z:()=>n})},2587:(e,t,r)=>{r.d(t,{N:()=>c,T:()=>u});var n=r(2177),i=r(5546),o=r(8e3),a=r(3325);const s={stn:[a.D.sessionTrace],err:[a.D.jserrors,a.D.metrics],ins:[a.D.pageAction],spa:[a.D.spa],sr:[a.D.sessionReplay,a.D.sessionTrace]};function c(e,t){const r=n.ee.get(t);e&&"object"==typeof e&&(Object.entries(e).forEach((e=>{let[t,n]=e;void 0===u[t]&&(s[t]?s[t].forEach((e=>{n?(0,i.p)("feat-"+t,[],void 0,e,r):(0,i.p)("block-"+t,[],void 0,e,r),(0,i.p)("rumresp-"+t,[Boolean(n)],void 0,e,r)})):n&&(0,i.p)("feat-"+t,[],void 0,void 0,r),u[t]=Boolean(n))})),Object.keys(s).forEach((e=>{void 0===u[e]&&(s[e]?.forEach((t=>(0,i.p)("rumresp-"+e,[!1],void 0,t,r))),u[e]=!1)})),(0,o.L)(t,a.D.pageViewEvent))}const u={}},2210:(e,t,r)=>{r.d(t,{X:()=>i});var n=Object.prototype.hasOwnProperty;function i(e,t,r){if(n.call(e,t))return e[t];var i=r();if(Object.defineProperty&&Object.keys)try{return Object.defineProperty(e,t,{value:i,writable:!0,enumerable:!1}),i}catch(e){}return e[t]=i,i}},1284:(e,t,r)=>{r.d(t,{D:()=>n});const n=(e,t)=>Object.entries(e||{}).map((e=>{let[r,n]=e;return t(r,n)}))},4351:(e,t,r)=>{r.d(t,{P:()=>o});var n=r(2177);const i=()=>{const e=new WeakSet;return(t,r)=>{if("object"==typeof r&&null!==r){if(e.has(r))return;e.add(r)}return r}};function o(e){try{return JSON.stringify(e,i())}catch(e){try{n.ee.emit("internal-error",[e])}catch(e){}}}},3960:(e,t,r)=>{r.d(t,{K:()=>a,b:()=>o});var n=r(3239);function i(){return"undefined"==typeof document||"complete"===document.readyState}function o(e,t){if(i())return e();(0,n.bP)("load",e,t)}function a(e){if(i())return e();(0,n.iz)("DOMContentLoaded",e)}},8632:(e,t,r)=>{r.d(t,{EZ:()=>u,Qy:()=>c,ce:()=>o,fP:()=>a,gG:()=>d,mF:()=>s});var n=r(7894),i=r(385);const o={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net"};function a(){return i._A.NREUM||(i._A.NREUM={}),void 0===i._A.newrelic&&(i._A.newrelic=i._A.NREUM),i._A.NREUM}function s(){let e=a();return e.o||(e.o={ST:i._A.setTimeout,SI:i._A.setImmediate,CT:i._A.clearTimeout,XHR:i._A.XMLHttpRequest,REQ:i._A.Request,EV:i._A.Event,PR:i._A.Promise,MO:i._A.MutationObserver,FETCH:i._A.fetch}),e}function c(e,t,r){let i=a();const o=i.initializedAgents||{},s=o[e]||{};return Object.keys(s).length||(s.initializedAt={ms:(0,n.z)(),date:new Date}),i.initializedAgents={...o,[e]:{...s,[r]:t}},i}function u(e,t){a()[e]=t}function d(){return function(){let e=a();const t=e.info||{};e.info={beacon:o.beacon,errorBeacon:o.errorBeacon,...t}}(),function(){let e=a();const t=e.init||{};e.init={...t}}(),s(),function(){let e=a();const t=e.loader_config||{};e.loader_config={...t}}(),a()}},7956:(e,t,r)=>{r.d(t,{N:()=>i});var n=r(3239);function i(e){let t=arguments.length>1&&void 0!==arguments[1]&&arguments[1],r=arguments.length>2?arguments[2]:void 0,i=arguments.length>3?arguments[3]:void 0;return void(0,n.iz)("visibilitychange",(function(){if(t)return void("hidden"==document.visibilityState&&e());e(document.visibilityState)}),r,i)}},1214:(e,t,r)=>{r.d(t,{em:()=>v,u5:()=>N,QU:()=>S,_L:()=>I,Gm:()=>L,Lg:()=>M,gy:()=>U,BV:()=>Q,Kf:()=>ee});var n=r(2177);const i="nr@original";var o=Object.prototype.hasOwnProperty,a=!1;function s(e,t){return e||(e=n.ee),r.inPlace=function(e,t,n,i,o){n||(n="");var a,s,c,u="-"===n.charAt(0);for(c=0;c 2?n-2:0),o=2;o {r(A[T],e,w),r(E[T],e,w)})),r(l._A,"fetch",y),t.on(y+"end",(function(e,r){var n=this;if(r){var i=r.headers.get("content-length");null!==i&&(n.rxSize=i),t.emit(y+"done",[null,r],n)}else t.emit(y+"done",[e],n)})),t}const O={},j=["pushState","replaceState"];function S(e){const t=function(e){return(e||n.ee).get("history")}(e);return!l.il||O[t.debugId]++||(O[t.debugId]=1,s(t).inPlace(window.history,j,"-")),t}var P=r(3239);const C={},R=["appendChild","insertBefore","replaceChild"];function I(e){const t=function(e){return(e||n.ee).get("jsonp")}(e);if(!l.il||C[t.debugId])return t;C[t.debugId]=!0;var r=s(t),i=/[?&](?:callback|cb)=([^&#]+)/,o=/(.*)\.([^.]+)/,a=/^(\w+)(\.|$)(.*)$/;function c(e,t){var r=e.match(a),n=r[1],i=r[3];return i?c(i,t[n]):t[n]}return r.inPlace(Node.prototype,R,"dom-"),t.on("dom-start",(function(e){!function(e){if(!e||"string"!=typeof e.nodeName||"script"!==e.nodeName.toLowerCase())return;if("function"!=typeof e.addEventListener)return;var n=(a=e.src,s=a.match(i),s?s[1]:null);var a,s;if(!n)return;var u=function(e){var t=e.match(o);if(t&&t.length>=3)return{key:t[2],parent:c(t[1],window)};return{key:e,parent:window}}(n);if("function"!=typeof u.parent[u.key])return;var d={};function f(){t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}function l(){t.emit("jsonp-error",[],d),t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}r.inPlace(u.parent,[u.key],"cb-",d),e.addEventListener("load",f,(0,P.m$)(!1)),e.addEventListener("error",l,(0,P.m$)(!1)),t.emit("new-jsonp",[e.src],d)}(e[0])})),t}var k=r(5763);const H={};function L(e){const t=function(e){return(e||n.ee).get("mutation")}(e);if(!l.il||H[t.debugId])return t;H[t.debugId]=!0;var r=s(t),i=k.Yu.MO;return i&&(window.MutationObserver=function(e){return this instanceof i?new i(r(e,"fn-")):i.apply(this,arguments)},MutationObserver.prototype=i.prototype),t}const z={};function M(e){const t=function(e){return(e||n.ee).get("promise")}(e);if(z[t.debugId])return t;z[t.debugId]=!0;var r=n.c,o=s(t),a=k.Yu.PR;return a&&function(){function e(r){var n=t.context(),i=o(r,"executor-",n,null,!1);const s=Reflect.construct(a,[i],e);return t.context(s).getCtx=function(){return n},s}l._A.Promise=e,Object.defineProperty(e,"name",{value:"Promise"}),e.toString=function(){return a.toString()},Object.setPrototypeOf(e,a),["all","race"].forEach((function(r){const n=a[r];e[r]=function(e){let i=!1;[...e||[]].forEach((e=>{this.resolve(e).then(a("all"===r),a(!1))}));const o=n.apply(this,arguments);return o;function a(e){return function(){t.emit("propagate",[null,!i],o,!1,!1),i=i||!e}}}})),["resolve","reject"].forEach((function(r){const n=a[r];e[r]=function(e){const r=n.apply(this,arguments);return e!==r&&t.emit("propagate",[e,!0],r,!1,!1),r}})),e.prototype=a.prototype;const n=a.prototype.then;a.prototype.then=function(){var e=this,i=r(e);i.promise=e;for(var a=arguments.length,s=new Array(a),c=0;c e())),t};function m(e,t){i.inPlace(t,["onreadystatechange"],"fn-",E)}function b(){var e=this,t=r.context(e);e.readyState>3&&!t.resolved&&(t.resolved=!0,r.emit("xhr-resolved",[],e)),i.inPlace(e,f,"fn-",E)}if(function(e,t){for(var r in e)t[r]=e[r]}(o,p),p.prototype=o.prototype,i.inPlace(p.prototype,J,"-xhr-",E),r.on("send-xhr-start",(function(e,t){m(e,t),function(e){h.push(e),a&&(y?y.then(A):u?u(A):(w=-w,x.data=w))}(t)})),r.on("open-xhr-start",m),a){var y=c&&c.resolve();if(!u&&!c){var w=1,x=document.createTextNode(w);new a(A).observe(x,{characterData:!0})}}else t.on("fn-end",(function(e){e[0]&&e[0].type===d||A()}));function A(){for(var e=0;e {r.d(t,{t:()=>n});const n=r(3325).D.ajax},6660:(e,t,r)=>{r.d(t,{A:()=>i,t:()=>n});const n=r(3325).D.jserrors,i="nr@seenError"},3081:(e,t,r)=>{r.d(t,{gF:()=>o,mY:()=>i,t9:()=>n,vz:()=>s,xS:()=>a});const n=r(3325).D.metrics,i="sm",o="cm",a="storeSupportabilityMetrics",s="storeEventMetrics"},4649:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageAction},7633:(e,t,r)=>{r.d(t,{Dz:()=>i,OJ:()=>a,qw:()=>o,t9:()=>n});const n=r(3325).D.pageViewEvent,i="firstbyte",o="domcontent",a="windowload"},9251:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageViewTiming},3614:(e,t,r)=>{r.d(t,{BST_RESOURCE:()=>i,END:()=>s,FEATURE_NAME:()=>n,FN_END:()=>u,FN_START:()=>c,PUSH_STATE:()=>d,RESOURCE:()=>o,START:()=>a});const n=r(3325).D.sessionTrace,i="bstResource",o="resource",a="-start",s="-end",c="fn"+a,u="fn"+s,d="pushState"},7836:(e,t,r)=>{r.d(t,{BODY:()=>A,CB_END:()=>E,CB_START:()=>u,END:()=>x,FEATURE_NAME:()=>i,FETCH:()=>_,FETCH_BODY:()=>v,FETCH_DONE:()=>m,FETCH_START:()=>p,FN_END:()=>c,FN_START:()=>s,INTERACTION:()=>l,INTERACTION_API:()=>d,INTERACTION_EVENTS:()=>o,JSONP_END:()=>b,JSONP_NODE:()=>g,JS_TIME:()=>T,MAX_TIMER_BUDGET:()=>a,REMAINING:()=>f,SPA_NODE:()=>h,START:()=>w,originalSetTimeout:()=>y});var n=r(5763);const i=r(3325).D.spa,o=["click","submit","keypress","keydown","keyup","change"],a=999,s="fn-start",c="fn-end",u="cb-start",d="api-ixn-",f="remaining",l="interaction",h="spaNode",g="jsonpNode",p="fetch-start",m="fetch-done",v="fetch-body-",b="jsonp-end",y=n.Yu.ST,w="-start",x="-end",A="-body",E="cb"+x,T="jsTime",_="fetch"},5938:(e,t,r)=>{r.d(t,{W:()=>o});var n=r(5763),i=r(2177);class o{constructor(e,t,r){this.agentIdentifier=e,this.aggregator=t,this.ee=i.ee.get(e,(0,n.OP)(this.agentIdentifier).isolatedBacklog),this.featureName=r,this.blocked=!1}}},9144:(e,t,r)=>{r.d(t,{j:()=>m});var n=r(3325),i=r(5763),o=r(5546),a=r(2177),s=r(7894),c=r(8e3),u=r(3960),d=r(385),f=r(50),l=r(3081),h=r(8632);function g(){const e=(0,h.gG)();["setErrorHandler","finished","addToTrace","inlineHit","addRelease","addPageAction","setCurrentRouteName","setPageViewName","setCustomAttribute","interaction","noticeError","setUserId"].forEach((t=>{e[t]=function(){for(var r=arguments.length,n=new Array(r),i=0;i 1?r-1:0),i=1;i {e.exposed&&e.api[t]&&o.push(e.api[t](...n))})),o.length>1?o:o[0]}(t,...n)}}))}var p=r(2587);function m(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:{},m=arguments.length>2?arguments[2]:void 0,v=arguments.length>3?arguments[3]:void 0,{init:b,info:y,loader_config:w,runtime:x={loaderType:m},exposed:A=!0}=t;const E=(0,h.gG)();y||(b=E.init,y=E.info,w=E.loader_config),(0,i.Dg)(e,b||{}),(0,i.GE)(e,w||{}),(0,i.sU)(e,x),y.jsAttributes??={},d.v6&&(y.jsAttributes.isWorker=!0),(0,i.CX)(e,y),g();const T=function(e,t){t||(0,c.R)(e,"api");const h={};var g=a.ee.get(e),p=g.get("tracer"),m="api-",v=m+"ixn-";function b(t,r,n,o){const a=(0,i.C5)(e);return null===r?delete a.jsAttributes[t]:(0,i.CX)(e,{...a,jsAttributes:{...a.jsAttributes,[t]:r}}),x(m,n,!0,o||null===r?"session":void 0)(t,r)}function y(){}["setErrorHandler","finished","addToTrace","inlineHit","addRelease"].forEach((e=>h[e]=x(m,e,!0,"api"))),h.addPageAction=x(m,"addPageAction",!0,n.D.pageAction),h.setCurrentRouteName=x(m,"routeName",!0,n.D.spa),h.setPageViewName=function(t,r){if("string"==typeof t)return"/"!==t.charAt(0)&&(t="/"+t),(0,i.OP)(e).customTransaction=(r||"http://custom.transaction")+t,x(m,"setPageViewName",!0)()},h.setCustomAttribute=function(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2];if("string"==typeof e){if(["string","number"].includes(typeof t)||null===t)return b(e,t,"setCustomAttribute",r);(0,f.Z)("Failed to execute setCustomAttribute.\nNon-null value must be a string or number type, but a type of was provided."))}else(0,f.Z)("Failed to execute setCustomAttribute.\nName must be a string type, but a type of was provided."))},h.setUserId=function(e){if("string"==typeof e||null===e)return b("enduser.id",e,"setUserId",!0);(0,f.Z)("Failed to execute setUserId.\nNon-null value must be a string type, but a type of was provided."))},h.interaction=function(){return(new y).get()};var w=y.prototype={createTracer:function(e,t){var r={},i=this,a="function"==typeof t;return(0,o.p)(v+"tracer",[(0,s.z)(),e,r],i,n.D.spa,g),function(){if(p.emit((a?"":"no-")+"fn-start",[(0,s.z)(),i,a],r),a)try{return t.apply(this,arguments)}catch(e){throw p.emit("fn-err",[arguments,this,"string"==typeof e?new Error(e):e],r),e}finally{p.emit("fn-end",[(0,s.z)()],r)}}}};function x(e,t,r,i){return function(){return(0,o.p)(l.xS,["API/"+t+"/called"],void 0,n.D.metrics,g),i&&(0,o.p)(e+t,[(0,s.z)(),...arguments],r?null:this,i,g),r?void 0:this}}function A(){r.e(439).then(r.bind(r,7438)).then((t=>{let{setAPI:r}=t;r(e),(0,c.L)(e,"api")})).catch((()=>(0,f.Z)("Downloading runtime APIs failed...")))}return["actionText","setName","setAttribute","save","ignore","onEnd","getContext","end","get"].forEach((e=>{w[e]=x(v,e,void 0,n.D.spa)})),h.noticeError=function(e,t){"string"==typeof e&&(e=new Error(e)),(0,o.p)(l.xS,["API/noticeError/called"],void 0,n.D.metrics,g),(0,o.p)("err",[e,(0,s.z)(),!1,t],void 0,n.D.jserrors,g)},d.il?(0,u.b)((()=>A()),!0):A(),h}(e,v);return(0,h.Qy)(e,T,"api"),(0,h.Qy)(e,A,"exposed"),(0,h.EZ)("activatedFeatures",p.T),T}},3325:(e,t,r)=>{r.d(t,{D:()=>n,p:()=>i});const n={ajax:"ajax",jserrors:"jserrors",metrics:"metrics",pageAction:"page_action",pageViewEvent:"page_view_event",pageViewTiming:"page_view_timing",sessionReplay:"session_replay",sessionTrace:"session_trace",spa:"spa"},i={[n.pageViewEvent]:1,[n.pageViewTiming]:2,[n.metrics]:3,[n.jserrors]:4,[n.ajax]:5,[n.sessionTrace]:6,[n.pageAction]:7,[n.spa]:8,[n.sessionReplay]:9}}},n={};function i(e){var t=n[e];if(void 0!==t)return t.exports;var o=n[e]={exports:{}};return r[e](o,o.exports,i),o.exports}i.m=r,i.d=(e,t)=>{for(var r in t)i.o(t,r)&&!i.o(e,r)&&Object.defineProperty(e,r,{enumerable:!0,get:t[r]})},i.f={},i.e=e=>Promise.all(Object.keys(i.f).reduce(((t,r)=>(i.f[r](e,t),t)),[])),i.u=e=>(({78:"page_action-aggregate",147:"metrics-aggregate",242:"session-manager",317:"jserrors-aggregate",348:"page_view_timing-aggregate",412:"lazy-feature-loader",439:"async-api",538:"recorder",590:"session_replay-aggregate",675:"compressor",733:"session_trace-aggregate",786:"page_view_event-aggregate",873:"spa-aggregate",898:"ajax-aggregate"}[e]||e)+"."+{78:"ac76d497",147:"3dc53903",148:"1a20d5fe",242:"2a64278a",317:"49e41428",348:"bd6de33a",412:"2f55ce66",439:"30bd804e",538:"1b18459f",590:"cf0efb30",675:"ae9f91a8",733:"83105561",786:"06482edd",860:"03a8b7a5",873:"e6b09d52",898:"998ef92b"}[e]+"-1.236.0.min.js"),i.o=(e,t)=>Object.prototype.hasOwnProperty.call(e,t),e={},t="NRBA:",i.l=(r,n,o,a)=>{if(e[r])e[r].push(n);else{var s,c;if(void 0!==o)for(var u=document.getElementsByTagName("script"),d=0;d {s.onerror=s.onload=null,clearTimeout(h);var i=e[r];if(delete e[r],s.parentNode&&s.parentNode.removeChild(s),i&&i.forEach((e=>e(n))),t)return t(n)},h=setTimeout(l.bind(null,void 0,{type:"timeout",target:s}),12e4);s.onerror=l.bind(null,s.onerror),s.onload=l.bind(null,s.onload),c&&document.head.appendChild(s)}},i.r=e=>{"undefined"!=typeof Symbol&&Symbol.toStringTag&&Object.defineProperty(e,Symbol.toStringTag,{value:"Module"}),Object.defineProperty(e,"__esModule",{value:!0})},i.j=364,i.p="https://js-agent.newrelic.com/",(()=>{var e={364:0,953:0};i.f.j=(t,r)=>{var n=i.o(e,t)?e[t]:void 0;if(0!==n)if(n)r.push(n[2]);else{var o=new Promise(((r,i)=>n=e[t]=[r,i]));r.push(n[2]=o);var a=i.p+i.u(t),s=new Error;i.l(a,(r=>{if(i.o(e,t)&&(0!==(n=e[t])&&(e[t]=void 0),n)){var o=r&&("load"===r.type?"missing":r.type),a=r&&r.target&&r.target.src;s.message="Loading chunk "+t+" failed.\n("+o+": "+a+")",s.name="ChunkLoadError",s.type=o,s.request=a,n[1](s)}}),"chunk-"+t,t)}};var t=(t,r)=>{var n,o,[a,s,c]=r,u=0;if(a.some((t=>0!==e[t]))){for(n in s)i.o(s,n)&&(i.m[n]=s[n]);if(c)c(i)}for(t&&t(r);u {i.r(o);var e=i(3325),t=i(5763);const r=Object.values(e.D);function n(e){const n={};return r.forEach((r=>{n[r]=function(e,r){return!1!==(0,t.Mt)(r,"".concat(e,".enabled"))}(r,e)})),n}var a=i(9144);var s=i(5546),c=i(385),u=i(8e3),d=i(5938),f=i(3960),l=i(50);class h extends d.W{constructor(e,t,r){let n=!(arguments.length>3&&void 0!==arguments[3])||arguments[3];super(e,t,r),this.auto=n,this.abortHandler,this.featAggregate,this.onAggregateImported,n&&(0,u.R)(e,r)}importAggregator(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:{};if(this.featAggregate||!this.auto)return;const r=c.il&&!0===(0,t.Mt)(this.agentIdentifier,"privacy.cookies_enabled");let n;this.onAggregateImported=new Promise((e=>{n=e}));const o=async()=>{let t;try{if(r){const{setupAgentSession:e}=await Promise.all([i.e(860),i.e(242)]).then(i.bind(i,3228));t=e(this.agentIdentifier)}}catch(e){(0,l.Z)("A problem occurred when starting up session manager. This page will not start or extend any session.",e)}try{if(!this.shouldImportAgg(this.featureName,t))return void(0,u.L)(this.agentIdentifier,this.featureName);const{lazyFeatureLoader:r}=await i.e(412).then(i.bind(i,8582)),{Aggregate:o}=await r(this.featureName,"aggregate");this.featAggregate=new o(this.agentIdentifier,this.aggregator,e),n(!0)}catch(e){(0,l.Z)("Downloading and initializing ".concat(this.featureName," failed..."),e),this.abortHandler?.(),n(!1)}};c.il?(0,f.b)((()=>o()),!0):o()}shouldImportAgg(r,n){return r!==e.D.sessionReplay||!1!==(0,t.Mt)(this.agentIdentifier,"session_trace.enabled")&&(!!n?.isNew||!!n?.state.sessionReplay)}}var g=i(7633),p=i(7894);class m extends h{static featureName=g.t9;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];if(super(r,n,g.t9,i),("undefined"==typeof PerformanceNavigationTiming||c.Tt)&&"undefined"!=typeof PerformanceTiming){const n=(0,t.OP)(r);n[g.Dz]=Math.max(Date.now()-n.offset,0),(0,f.K)((()=>n[g.qw]=Math.max((0,p.z)()-n[g.Dz],0))),(0,f.b)((()=>{const t=(0,p.z)();n[g.OJ]=Math.max(t-n[g.Dz],0),(0,s.p)("timing",["load",t],void 0,e.D.pageViewTiming,this.ee)}))}this.importAggregator()}}var v=i(1117),b=i(1284);class y extends v.w{constructor(e){super(e),this.aggregatedData={}}store(e,t,r,n,i){var o=this.getBucket(e,t,r,i);return o.metrics=function(e,t){t||(t={count:0});return t.count+=1,(0,b.D)(e,(function(e,r){t[e]=w(r,t[e])})),t}(n,o.metrics),o}merge(e,t,r,n,i){var o=this.getBucket(e,t,n,i);if(o.metrics){var a=o.metrics;a.count+=r.count,(0,b.D)(r,(function(e,t){if("count"!==e){var n=a[e],i=r[e];i&&!i.c?a[e]=w(i.t,n):a[e]=function(e,t){if(!t)return e;t.c||(t=x(t.t));return t.min=Math.min(e.min,t.min),t.max=Math.max(e.max,t.max),t.t+=e.t,t.sos+=e.sos,t.c+=e.c,t}(i,a[e])}}))}else o.metrics=r}storeMetric(e,t,r,n){var i=this.getBucket(e,t,r);return i.stats=w(n,i.stats),i}getBucket(e,t,r,n){this.aggregatedData[e]||(this.aggregatedData[e]={});var i=this.aggregatedData[e][t];return i||(i=this.aggregatedData[e][t]={params:r||{}},n&&(i.custom=n)),i}get(e,t){return t?this.aggregatedData[e]&&this.aggregatedData[e][t]:this.aggregatedData[e]}take(e){for(var t={},r="",n=!1,i=0;i t.max&&(t.max=e),e 2&&void 0!==arguments[2])||arguments[2];super(e,r,j.t,n),c.il&&((0,t.OP)(e).initHidden=Boolean("hidden"===document.visibilityState),(0,N.N)((()=>(0,s.p)("docHidden",[(0,p.z)()],void 0,j.t,this.ee)),!0),(0,O.bP)("pagehide",(()=>(0,s.p)("winPagehide",[(0,p.z)()],void 0,j.t,this.ee))),this.importAggregator())}}var P=i(3081);class C extends h{static featureName=P.t9;constructor(e,t){let r=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(e,t,P.t9,r),this.importAggregator()}}var R,I=i(2210),k=i(1214),H=i(2177),L={};try{R=localStorage.getItem("__nr_flags").split(","),console&&"function"==typeof console.log&&(L.console=!0,-1!==R.indexOf("dev")&&(L.dev=!0),-1!==R.indexOf("nr_dev")&&(L.nrDev=!0))}catch(e){}function z(e){try{L.console&&z(e)}catch(e){}}L.nrDev&&H.ee.on("internal-error",(function(e){z(e.stack)})),L.dev&&H.ee.on("fn-err",(function(e,t,r){z(r.stack)})),L.dev&&(z("NR AGENT IN DEVELOPMENT MODE"),z("flags: "+(0,b.D)(L,(function(e,t){return e})).join(", ")));var M=i(6660);class B extends h{static featureName=M.t;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(r,n,M.t,i),this.skipNext=0;try{this.removeOnAbort=new AbortController}catch(e){}const o=this;o.ee.on("fn-start",(function(e,t,r){o.abortHandler&&(o.skipNext+=1)})),o.ee.on("fn-err",(function(t,r,n){o.abortHandler&&!n[M.A]&&((0,I.X)(n,M.A,(function(){return!0})),this.thrown=!0,(0,s.p)("err",[n,(0,p.z)()],void 0,e.D.jserrors,o.ee))})),o.ee.on("fn-end",(function(){o.abortHandler&&!this.thrown&&o.skipNext>0&&(o.skipNext-=1)})),o.ee.on("internal-error",(function(t){(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,o.ee)})),this.origOnerror=c._A.onerror,c._A.onerror=this.onerrorHandler.bind(this),c._A.addEventListener("unhandledrejection",(t=>{const r=function(e){let t="Unhandled Promise Rejection: ";if(e instanceof Error)try{return e.message=t+e.message,e}catch(t){return e}if(void 0===e)return new Error(t);try{return new Error(t+(0,D.P)(e))}catch(e){return new Error(t)}}(t.reason);(0,s.p)("err",[r,(0,p.z)(),!1,{unhandledPromiseRejection:1}],void 0,e.D.jserrors,this.ee)}),(0,O.m$)(!1,this.removeOnAbort?.signal)),(0,k.gy)(this.ee),(0,k.BV)(this.ee),(0,k.em)(this.ee),(0,t.OP)(r).xhrWrappable&&(0,k.Kf)(this.ee),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}onerrorHandler(t,r,n,i,o){"function"==typeof this.origOnerror&&this.origOnerror(...arguments);try{this.skipNext?this.skipNext-=1:(0,s.p)("err",[o||new F(t,r,n),(0,p.z)()],void 0,e.D.jserrors,this.ee)}catch(t){try{(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,this.ee)}catch(e){}}return!1}}function F(e,t,r){this.message=e||"Uncaught error with no additional information",this.sourceURL=t,this.line=r}let U=1;const q="nr@id";function G(e){const t=typeof e;return!e||"object"!==t&&"function"!==t?-1:e===c._A?0:(0,I.X)(e,q,(function(){return U++}))}function V(e){if("string"==typeof e&&e.length)return e.length;if("object"==typeof e){if("undefined"!=typeof ArrayBuffer&&e instanceof ArrayBuffer&&e.byteLength)return e.byteLength;if("undefined"!=typeof Blob&&e instanceof Blob&&e.size)return e.size;if(!("undefined"!=typeof FormData&&e instanceof FormData))try{return(0,D.P)(e).length}catch(e){return}}}var X=i(7243);class W{constructor(e){this.agentIdentifier=e,this.generateTracePayload=this.generateTracePayload.bind(this),this.shouldGenerateTrace=this.shouldGenerateTrace.bind(this)}generateTracePayload(e){if(!this.shouldGenerateTrace(e))return null;var r=(0,t.DL)(this.agentIdentifier);if(!r)return null;var n=(r.accountID||"").toString()||null,i=(r.agentID||"").toString()||null,o=(r.trustKey||"").toString()||null;if(!n||!i)return null;var a=(0,_.M)(),s=(0,_.Ht)(),c=Date.now(),u={spanId:a,traceId:s,timestamp:c};return(e.sameOrigin||this.isAllowedOrigin(e)&&this.useTraceContextHeadersForCors())&&(u.traceContextParentHeader=this.generateTraceContextParentHeader(a,s),u.traceContextStateHeader=this.generateTraceContextStateHeader(a,c,n,i,o)),(e.sameOrigin&&!this.excludeNewrelicHeader()||!e.sameOrigin&&this.isAllowedOrigin(e)&&this.useNewrelicHeaderForCors())&&(u.newrelicHeader=this.generateTraceHeader(a,s,c,n,i,o)),u}generateTraceContextParentHeader(e,t){return"00-"+t+"-"+e+"-01"}generateTraceContextStateHeader(e,t,r,n,i){return i+"@nr=0-1-"+r+"-"+n+"-"+e+"----"+t}generateTraceHeader(e,t,r,n,i,o){if(!("function"==typeof c._A?.btoa))return null;var a={v:[0,1],d:{ty:"Browser",ac:n,ap:i,id:e,tr:t,ti:r}};return o&&n!==o&&(a.d.tk=o),btoa((0,D.P)(a))}shouldGenerateTrace(e){return this.isDtEnabled()&&this.isAllowedOrigin(e)}isAllowedOrigin(e){var r=!1,n={};if((0,t.Mt)(this.agentIdentifier,"distributed_tracing")&&(n=(0,t.P_)(this.agentIdentifier).distributed_tracing),e.sameOrigin)r=!0;else if(n.allowed_origins instanceof Array)for(var i=0;i 2&&void 0!==arguments[2])||arguments[2];super(r,n,Z.t,i),(0,t.OP)(r).xhrWrappable&&(this.dt=new W(r),this.handler=(e,t,r,n)=>(0,s.p)(e,t,r,n,this.ee),(0,k.u5)(this.ee),(0,k.Kf)(this.ee),function(r,n,i,o){function a(e){var t=this;t.totalCbs=0,t.called=0,t.cbTime=0,t.end=E,t.ended=!1,t.xhrGuids={},t.lastSize=null,t.loadCaptureCalled=!1,t.params=this.params||{},t.metrics=this.metrics||{},e.addEventListener("load",(function(r){_(t,e)}),(0,O.m$)(!1)),c.IF||e.addEventListener("progress",(function(e){t.lastSize=e.loaded}),(0,O.m$)(!1))}function s(e){this.params={method:e[0]},T(this,e[1]),this.metrics={}}function u(e,n){var i=(0,t.DL)(r);i.xpid&&this.sameOrigin&&n.setRequestHeader("X-NewRelic-ID",i.xpid);var a=o.generateTracePayload(this.parsedOrigin);if(a){var s=!1;a.newrelicHeader&&(n.setRequestHeader("newrelic",a.newrelicHeader),s=!0),a.traceContextParentHeader&&(n.setRequestHeader("traceparent",a.traceContextParentHeader),a.traceContextStateHeader&&n.setRequestHeader("tracestate",a.traceContextStateHeader),s=!0),s&&(this.dt=a)}}function d(e,t){var r=this.metrics,i=e[0],o=this;if(r&&i){var a=V(i);a&&(r.txSize=a)}this.startTime=(0,p.z)(),this.listener=function(e){try{"abort"!==e.type||o.loadCaptureCalled||(o.params.aborted=!0),("load"!==e.type||o.called===o.totalCbs&&(o.onloadCalled||"function"!=typeof t.onload)&&"function"==typeof o.end)&&o.end(t)}catch(e){try{n.emit("internal-error",[e])}catch(e){}}};for(var s=0;s 1?e[1]=i:e.push(i)}else e[0]&&e[0].headers&&s(e[0].headers,n)&&(this.dt=n);function s(e,t){var r=!1;return t.newrelicHeader&&(e.set("newrelic",t.newrelicHeader),r=!0),t.traceContextParentHeader&&(e.set("traceparent",t.traceContextParentHeader),t.traceContextStateHeader&&e.set("tracestate",t.traceContextStateHeader),r=!0),r}}function x(e,t){this.params={},this.metrics={},this.startTime=(0,p.z)(),this.dt=t,e.length>=1&&(this.target=e[0]),e.length>=2&&(this.opts=e[1]);var r,n=this.opts||{},i=this.target;"string"==typeof i?r=i:"object"==typeof i&&i instanceof Y?r=i.url:c._A?.URL&&"object"==typeof i&&i instanceof URL&&(r=i.href),T(this,r);var o=(""+(i&&i instanceof Y&&i.method||n.method||"GET")).toUpperCase();this.params.method=o,this.txSize=V(n.body)||0}function A(t,r){var n;this.endTime=(0,p.z)(),this.params||(this.params={}),this.params.status=r?r.status:0,"string"==typeof this.rxSize&&this.rxSize.length>0&&(n=+this.rxSize);var o={txSize:this.txSize,rxSize:n,duration:(0,p.z)()-this.startTime};i("xhr",[this.params,o,this.startTime,this.endTime,"fetch"],this,e.D.ajax)}function E(t){var r=this.params,n=this.metrics;if(!this.ended){this.ended=!0;for(var o=0;o 2&&void 0!==arguments[2])||arguments[2];super(e,t,we.t,r),this.importAggregator()}}new class{constructor(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:(0,_.ky)(16);c._A?(this.agentIdentifier=t,this.sharedAggregator=new y({agentIdentifier:this.agentIdentifier}),this.features={},this.desiredFeatures=new Set(e.features||[]),this.desiredFeatures.add(m),Object.assign(this,(0,a.j)(this.agentIdentifier,e,e.loaderType||"agent")),this.start()):(0,l.Z)("Failed to initial the agent. Could not determine the runtime environment.")}get config(){return{info:(0,t.C5)(this.agentIdentifier),init:(0,t.P_)(this.agentIdentifier),loader_config:(0,t.DL)(this.agentIdentifier),runtime:(0,t.OP)(this.agentIdentifier)}}start(){const t="features";try{const r=n(this.agentIdentifier),i=[...this.desiredFeatures];i.sort(((t,r)=>e.p[t.featureName]-e.p[r.featureName])),i.forEach((t=>{if(r[t.featureName]||t.featureName===e.D.pageViewEvent){const n=function(t){switch(t){case e.D.ajax:return[e.D.jserrors];case e.D.sessionTrace:return[e.D.ajax,e.D.pageViewEvent];case e.D.sessionReplay:return[e.D.sessionTrace];case e.D.pageViewTiming:return[e.D.pageViewEvent];default:return[]}}(t.featureName);n.every((e=>r[e]))||(0,l.Z)("".concat(t.featureName," is enabled but one or more dependent features has been disabled (").concat((0,D.P)(n),"). This may cause unintended consequences or missing data...")),this.features[t.featureName]=new t(this.agentIdentifier,this.sharedAggregator)}})),(0,T.Qy)(this.agentIdentifier,this.features,t)}catch(e){(0,l.Z)("Failed to initialize all enabled instrument classes (agent aborted) -",e);for(const e in this.features)this.features[e].abortHandler?.();const r=(0,T.fP)();return delete r.initializedAgents[this.agentIdentifier]?.api,delete r.initializedAgents[this.agentIdentifier]?.[t],delete this.sharedAggregator,r.ee?.abort(),delete r.ee?.get(this.agentIdentifier),!1}}}({features:[J,m,S,class extends h{static featureName=oe;constructor(t,r){if(super(t,r,oe,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;const n=this.ee;let i;(0,k.QU)(n),this.eventsEE=(0,k.em)(n),this.eventsEE.on(se,(function(e,t){this.bstStart=(0,p.z)()})),this.eventsEE.on(ae,(function(t,r){(0,s.p)("bst",[t[0],r,this.bstStart,(0,p.z)()],void 0,e.D.sessionTrace,n)})),n.on(ce+ne,(function(e){this.time=(0,p.z)(),this.startPath=location.pathname+location.hash})),n.on(ce+ie,(function(t){(0,s.p)("bstHist",[location.pathname+location.hash,this.startPath,this.time],void 0,e.D.sessionTrace,n)}));try{i=new PerformanceObserver((t=>{const r=t.getEntries();(0,s.p)(te,[r],void 0,e.D.sessionTrace,n)})),i.observe({type:re,buffered:!0})}catch(e){}this.importAggregator({resourceObserver:i})}},C,xe,B,class extends h{static featureName=de;constructor(e,r){if(super(e,r,de,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;if(!(0,t.OP)(e).xhrWrappable)return;try{this.removeOnAbort=new AbortController}catch(e){}let n,i=0;const o=this.ee.get("tracer"),a=(0,k._L)(this.ee),s=(0,k.Lg)(this.ee),u=(0,k.BV)(this.ee),d=(0,k.Kf)(this.ee),f=this.ee.get("events"),l=(0,k.u5)(this.ee),h=(0,k.QU)(this.ee),g=(0,k.Gm)(this.ee);function m(e,t){h.emit("newURL",[""+window.location,t])}function v(){i++,n=window.location.hash,this[ve]=(0,p.z)()}function b(){i--,window.location.hash!==n&&m(0,!0);var e=(0,p.z)();this[pe]=~~this[pe]+e-this[ve],this[ye]=e}function y(e,t){e.on(t,(function(){this[t]=(0,p.z)()}))}this.ee.on(ve,v),s.on(be,v),a.on(be,v),this.ee.on(ye,b),s.on(ge,b),a.on(ge,b),this.ee.buffer([ve,ye,"xhr-resolved"],this.featureName),f.buffer([ve],this.featureName),u.buffer(["setTimeout"+le,"clearTimeout"+fe,ve],this.featureName),d.buffer([ve,"new-xhr","send-xhr"+fe],this.featureName),l.buffer([me+fe,me+"-done",me+he+fe,me+he+le],this.featureName),h.buffer(["newURL"],this.featureName),g.buffer([ve],this.featureName),s.buffer(["propagate",be,ge,"executor-err","resolve"+fe],this.featureName),o.buffer([ve,"no-"+ve],this.featureName),a.buffer(["new-jsonp","cb-start","jsonp-error","jsonp-end"],this.featureName),y(l,me+fe),y(l,me+"-done"),y(a,"new-jsonp"),y(a,"jsonp-end"),y(a,"cb-start"),h.on("pushState-end",m),h.on("replaceState-end",m),window.addEventListener("hashchange",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("load",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("popstate",(function(){m(0,i>1)}),(0,O.m$)(!0,this.removeOnAbort?.signal)),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}}],loaderType:"spa"})})(),window.NRBA=o})(); window.jQuery || document.write(' ') CKEDITOR_BASEPATH='https://f1000research.com/js/vendor/ckeditor/' window.reactTheme = 'research'; window.MathJax = { CommonHTML: { linebreaks: { automatic: true } }, 'HTML-CSS': { linebreaks: { automatic: true } }, SVG: { linebreaks: { automatic: true } }, AuthorInit: function() { MathJax.Hub.Register.MessageHook('End Process', function () { let timeout = false; // holder for timeout id const delay = 250; // delay after event is "complete" to run callback const reflowMath = function() { const dispFormulas = document.querySelectorAll('.disp-formula.panel'); if (!dispFormulas) { return; } for (const dispFormula of dispFormulas) { const child = dispFormula.querySelector('.MathJax_Preview').nextSibling.firstChild; const isMultiline = MathJax.Hub.getAllJax(dispFormula)[0].root.isMultiline; if (dispFormula.offsetWidth < child.offsetWidth || isMultiline) { MathJax.Hub.Queue(['Rerender', MathJax.Hub, dispFormula]); } } }; window.addEventListener('resize', function() { clearTimeout(timeout); // clear the timeout timeout = setTimeout(reflowMath, delay); // start timing for event "completion" }); }); }, }; if (window.location.hash == '#_=_'){ window.location = window.location.href.split('#')[0] } !function(f,b,e,v,n,t,s){if(f.fbq)return;n=f.fbq=function() {n.callMethod? n.callMethod.apply(n,arguments):n.queue.push(arguments)} ;if(!f._fbq)f._fbq=n; n.push=n;n.loaded=!0;n.version='2.0';n.queue=[];t=b.createElement(e);t.async=!0; t.src=v;s=b.getElementsByTagName(e)[0];s.parentNode.insertBefore(t,s)}(window, document,'script','https://connect.facebook.net/en_US/fbevents.js'); fbq('init', '1641728616063202'); fbq('track', "PixelInitialized", {}); (function(h,o,t,j,a,r){ h.hj=h.hj||function(){(h.hj.q=h.hj.q||[]).push(arguments)}; h._hjSettings={hjid:2318163,hjsv:6}; a=o.getElementsByTagName('head')[0]; r=o.createElement('script');r.async=1; r.src=t+h._hjSettings.hjid+j+h._hjSettings.hjsv; a.appendChild(r); })(window,document,'https://static.hotjar.com/c/hotjar-','.js?sv='); search file_upload Submit your research search menu close search Browse Gateways & Collections How to Publish Submit your Research My Submissions Article Guidelines Article Guidelines (New Versions) Open Data, Software and Code Guidelines Open Data and Accessible Source Materials Guidelines (HSS) Open Data, Software and Code Guidelines (PSE) Prepublication Checks Production Process Posters and Slides Guidelines Document Guidelines Article Processing Charges Peer Review Finding Article Reviewers About How it Works For Reviewers Our Advisors Policies Glossary FAQs For Developers Newsroom Contact My Research Submissions Content and Tracking Alerts My Details Sign In file_upload Submit your research { "@context": "https://schema.org", "@type": "ScholarlyArticle", "mainEntityOfPage": { "@type": "WebPage", "@id": "https://f1000research.com/articles/3-110" }, "headline": "Readable workflows need simple data", "datePublished": "2014-05-14T15:16:38", "dateModified": "2014-11-17T15:05:19", "author": [ { "@type": "Person", "name": "Claas-Thido Pfaff" }, { "@type": "Person", "name": "Karin Nadrowski" }, { "@type": "Person", "name": "Sophia Ratcliffe" }, { "@type": "Person", "name": "Christian Wirth" }, { "@type": "Person", "name": "Helge Bruelheide" } ], "publisher": { "@type": "Organization", "name": "F1000Research", "logo": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 480, "width": 60 } }, "image": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 1200, "width": 150 }, "description": "Sharing scientific analyses via workflows has great potential to improve the reproducibility of science as well as communicating research results. This is particularly useful for trans-disciplinary research fields such as biodiversity - ecosystem functioning (BEF), where syntheses need to merge data ranging from genes to the biosphere. Here we argue that enabling simplicity in the very beginning of workflows, at the point of data description and merging, offers huge potentials in reducing workflow complexity and in fostering data and workflow reuse. We illustrate our points using a typical analysis in BEF research, the aggregation of carbon pools in a forest ecosystem. We introduce indicators for the complexity of workflow components including data sources. We show that workflow complexity decreases exponentially during the course of the analysis and that simple text-based measures help to identify bottlenecks in a workflow and group workflow components according to tasks. We thus suggest that focusing on simplifying steps of data aggregation and imputation will greatly improve workflow readability and thus reproducibility. Providing feedback to data providers about the complexity of their datasets may help to produce better focused data that can be used more easily in further studies. At the same time, providing feedback about the complexity of workflow components may help to exchange shorter and simpler workflows for easier reuse. Additionally, identifying repetitive tasks informs software development in providing automated solutions. We discuss current initiatives in software and script development that implement quality control for simplicity and social tools of script valuation. Taken together we argue that focusing on simplifying data sources and workflow components will improve and accelerate data and workflow reuse and simplify the reproducibility of data-driven science." } { "@context": "http://schema.org", "@type": "BreadcrumbList", "itemListElement": [ { "@type": "ListItem", "position": "1", "item": { "@id": "https://f1000research.com/", "name": "Home" } }, { "@type": "ListItem", "position": "2", "item": { "@id": "https://f1000research.com/browse/articles", "name": "Browse" } }, { "@type": "ListItem", "position": "3", "item": { "@id": "https://f1000research.com/articles/3-110/v1", "name": "Readable workflows need simple data" } } ] } Home Browse Readable workflows need simple data ALL Metrics - Views Downloads Get PDF Get XML Cite How to cite this article Pfaff CT, Nadrowski K, Ratcliffe S et al. Readable workflows need simple data [version 1; peer review: 1 approved with reservations] . F1000Research 2014, 3 :110 ( https://doi.org/10.12688/f1000research.3940.1 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. Close Copy Citation Details Export Export Citation Sciwheel EndNote Ref. Manager Bibtex ProCite Sente EXPORT Select a format first Track Share ▬ ✚ Research Article Readable workflows need simple data [version 1; peer review: 1 approved with reservations] Claas-Thido Pfaff 1 , Karin Nadrowski 1 , Sophia Ratcliffe 1 , Christian Wirth 1 , Helge Bruelheide 2 Claas-Thido Pfaff 1 , Karin Nadrowski 1 , [...] Sophia Ratcliffe 1 , Christian Wirth 1 , Helge Bruelheide 2 PUBLISHED 14 May 2014 Author details Author details 1 Institute of Special Botany and Functional Biodiversity, University of Leipzig, 04103, Leipzig, Germany 2 Institute of Biology, Martin Luther University Halle-Wittenberg, 06108, Halle, Germany OPEN PEER REVIEW DETAILS REVIEWER STATUS Abstract Sharing scientific analyses via workflows has great potential to improve the reproducibility of science as well as communicating research results. This is particularly useful for trans-disciplinary research fields such as biodiversity - ecosystem functioning (BEF), where syntheses need to merge data ranging from genes to the biosphere. Here we argue that enabling simplicity in the very beginning of workflows, at the point of data description and merging, offers huge potentials in reducing workflow complexity and in fostering data and workflow reuse. We illustrate our points using a typical analysis in BEF research, the aggregation of carbon pools in a forest ecosystem. We introduce indicators for the complexity of workflow components including data sources. We show that workflow complexity decreases exponentially during the course of the analysis and that simple text-based measures help to identify bottlenecks in a workflow and group workflow components according to tasks. We thus suggest that focusing on simplifying steps of data aggregation and imputation will greatly improve workflow readability and thus reproducibility. Providing feedback to data providers about the complexity of their datasets may help to produce better focused data that can be used more easily in further studies. At the same time, providing feedback about the complexity of workflow components may help to exchange shorter and simpler workflows for easier reuse. Additionally, identifying repetitive tasks informs software development in providing automated solutions. We discuss current initiatives in software and script development that implement quality control for simplicity and social tools of script valuation. Taken together we argue that focusing on simplifying data sources and workflow components will improve and accelerate data and workflow reuse and simplify the reproducibility of data-driven science. READ ALL READ LESS Corresponding Author(s) Claas-Thido Pfaff ( [email protected] ) Close Corresponding author: Claas-Thido Pfaff Competing interests: No competing interests were disclosed. Grant information: The data was collected by 7 independent projects of the biodiversity - ecosystem functioning - China (BEF-China) research group funded by the German Research Foundation (DFG, FOR 891). The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript. Copyright: © 2014 Pfaff CT et al . This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. Data associated with the article are available under the terms of the Creative Commons Zero "No rights reserved" data waiver (CC0 1.0 Public domain dedication). How to cite: Pfaff CT, Nadrowski K, Ratcliffe S et al. Readable workflows need simple data [version 1; peer review: 1 approved with reservations] . F1000Research 2014, 3 :110 ( https://doi.org/10.12688/f1000research.3940.1 ) First published: 14 May 2014, 3 :110 ( https://doi.org/10.12688/f1000research.3940.1 ) Latest published: 17 Nov 2014, 3 :110 ( https://doi.org/10.12688/f1000research.3940.2 ) There is a newer version of this article available. Suppress this message for one day. Introduction Interdisciplinary approaches, new tools and technologies, and the increasing availability of online accessible data have changed the way researchers pose questions and perform analyses 1 . Workflow software enables access to distributed web services providing data 1 , and enables automation of the repetitive tasks that occur in every scientific analysis. Workflow tools such as Kepler or Pegasus help to break down complex tasks into smaller pieces 2 , 3 . However, an increase in the complexity of analyses and datasets packed into workflows can render them difficult to understand and to reuse. This is particularly true for the “long tail” of big data 4 , consisting of small and highly heterogeneous files that don’t result from automated loggers but from scientific experiments, observations, or interviews. The difficulty in reusing workflows and research data is not only a waste of time, money and effort but also represents a threat to the basic scientific principle of reproducibility. Current literature on workflows deals with different tools to create and manipulate workflows 5 – 7 as well as to keep track of data provenance 2 and how semantics can be integrated into workflows 8 . However, there is a lack of papers that discuss workflow components within an analysis including data processing. In the following we 1) introduce the concepts of workflow component complexity and identity as well as data complexity. We then use 2) a workflow from the research domain of biodiversity-ecosystem functioning (BEF) to illustrate these concepts. The analysis combines small and heterogeneous datasets from different working groups to quantify the effect of biodiversity and stand age on carbon stocks in a subtropical forest. In the third and last part of the paper we 3) discuss the opportunities for quantifying the complexity and identity of workflow components and data for developing useful features of data sharing platforms and fostering scientific reproducibility. In particular, we are convinced that simplicity and a clear focus are the key to adequate reuse and finally to the reproducibility of science. We use our findings to illustrate bottlenecks and opportunities for data sharing and the implementation and reuse of scientific workflows. Complexity and identity Workflows consist of components that communicate with each other. Data can be assembled from different sources and different techniques can be used to analyse the data. Components perform anything from simple data import and transformation tasks to the execution of complex statistical scripts or calls to remotely running data manipulation or information retrieval services 2 . The complexity of software or code in workflow components increases with the number of linearly independent paths 9 . Thus the complexity increases with every decision of a programmer or analyst that is introduced by an if-else or case statement. However, it is our experience as data managers and researchers in biodiversity sciences that most workflows shared between researchers do not include such if-else statements but contain one single path only. The Code Climate initiative provides code complexity feedback to programmers for many different programming languages https://codeclimate.com/?v=b . Their complexity measures include the number of lines used for methods as well as the repetition of identical code lines. Quantifying workflow complexity along the sequence of components may help to identify parts of workflows that need simplification. Workflows often begin with a series of steps that contain data preparation, merging and imputation. These first steps can make up to 70% of the whole workflow 10 . In 10 , they identify common motifs in workflows including data and workflow oriented motifs. Identifying common and recurring tasks or motifs in workflows may allow for an improved sharing of code snippets and workflow components. There are many examples of sharing code snippets, including (e.g gist, stackoverflow). Providing quantitative complexity measures together with automated tagging may further increase component and data reuse. Identification of tasks may also support the use of semantic tools in workflow creation 8 , 11 . Quantifying data complexity is not as straight forward as workflow component complexity. Datasets used for synthesis in research collaborations often consist of "dark" data, lacking sufficient metadata for reuse 4 , 12 , 13 . Here we suggest that data complexity can be quantified by looking at the workflow components needed to aggregate and focus the data for analysis. One of the paradigms of data-driven science is that the analysis should be accompanied by it’s data. We argue that at the same time, data should be accompanied by workflows that offer meaningful aggregation of the data. Data complexity could then be measured by the complexity of their workflows. In our experience as data managers of research collaborations, many datasets contain a complete representation of a certain study and thus allow us to answer more than just one single question. This is due to a "space efficient" use of sheets of papers and computer screens during the field period of the study. Thus, many data columns are used for different measurements, color is used to code for study sites without explicitly naming them in a separate column which constitutes bad quality data management. Later in the process of writing up, each analysis makes use of a subset of the data only. Thus, data needs to be transformed, imputed, aggregated, or merged with data from other columns to be used in an analysis 12 . Thus, not only the metadata but also the data columns in datasets differ in their quality and usage in a workflow. To date, we lack a suitable feedback mechanism for data providers about the quality and re-usability of their data 14 . Such feedback could potentially lead to more simple and focused datasets and thus to more focused workflows that can be shared and reused more efficiently. Focused workflow components have the potential to be used as basic building blocks in a semantically guided way of workflow creation 8 or to be targets of automation. In the following we illustrate the concepts mentioned above within a typical BEF workflow. The workflow combines datasets from different working groups to assess the influence of diversity and stand age on the carbon pool in a subtropical forest. We analyse the complexity and the identity of workflow components as well as the data sources. The effect of biodiversity on subtropical carbon stocks Our workflow performs a representative analysis in BEF. It aggregates carbon biomass from different pools of the ecosystem and compares plots along a gradient of biodiversity. The workflow combines data from 8 datasets to perform a linear regression model on the effect of biodiversity on carbon stocks in a subtropical forest. It takes into account carbon pools from soil, litter, woody debris, herb layer plants and trees and shrubs surpassing 3 cm diameter at breast height measured in the years 2008 and early 2009. The data was collected by 7 independent projects of the biodiversity - ecosystem functioning - China (BEF-China) research group funded by the German Research Foundation (DFG, FOR 891). The BEF-China research group ( www.bef-china.de ) uses two main research platforms. An experimental forest diversity gradient of 50 ha, and 27 observational plots of 30×30 m each located in the province of Gutianshan China. The plots are situated in the Nature Reserve of Gutianshan. The observational plots were selected according to a crossed sampling design along tree species richness and stand age. The data for the workflow on carbon pools stems from observational plots spanning a gradient from 22 to 116 years consisting of 14 to 35 species 15 . BEF-China uses the BEFdata platform 12 , https://github.com/befdata/befdata ) for managing and distributing data which also offers an Ecological-Metadata-Language (EML) export. We used the portal to retrieve the data and the according EML files which then were used to import the data into the Kepler Workflow system 2 for analysis ( Figure 1 ). Figure 1. The EML 2 dataset component in use for the integration of the wood density dataset in the workflow. On the right side the opened metadata window which displays all the additional information available for the dataset. As the underlying analysis of the workflow continues in the projects we only provide a short insight in the still preliminary findings here. The carbon pool in the observational plots ranged from 5321.18 kg to 51,095.95 kg. The linear model revealed both species richness as well as stand age increased the carbon pool. In addition, there was a significant interaction between stand age and species richness in that the increase of carbon with stand age was less steep in plots with higher species richness (p-values for stand age: 0.0006, species richness: 0.0568 and their interaction: 0.0236). Workflow design We use the Kepler workflow system (version 2.4) to build our workflow. The components in Kepler fall into two categories: “actors”, which handle all kinds of data related tasks, and “directors”, which direct the execution of components in the workflow. The components in Kepler can “talk” to each other via a port system. Output ports of components hand over their data to input ports of another component 2 . The “SDF” director was used to execute our workflow as it handles sequential workflows. The data was imported into Kepler using the “eml2dataset” actor. This actor can import datasets along the conventions of the Ecological Metadata Language 16 . The component reads the information available in the metadata file and uses it to automatically set up output ports to allow a direct consumption of the related data by other components in the workflow. The data in the underlying carbon stock analysis is manipulated mainly by using the rich statistics environment R 17 . From within Kepler we use the “RExpression” actor that offers an interface to R. We aimed at a uniform and low complexity for each workflow component. As a rule of thumb we set a limit of 5 lines of code per component. Quantifying workflow complexity To quantify the component complexity we used the number of code lines (loc), the number of R commands (cc) and R packages (pc) used, as well as the number of input and output ports (cp) of the component ( equation 1 ). We further calculated a relative component complexity as the ratio of absolute complexity to total workflow complexity given by the sum of all component complexities ( equation 2 ). a c = p c + l o c + c c + c p ( 1 ) r a c = a c ∑ i = 1 n a c i ( 2 ) As each component in the workflow starts its operation only if all input port variables have arrived, the longest port connection of a component back to a data source defines its absolute position in the workflow sequence ( Figure 2 ). We could thus explore total workflow complexity, individual component complexities, the number of components, and the number of identical tasks (see below) along the sequence of the workflow. For this we used linear models and compared them using Akaike Information Criteria (AIC) 18 . Figure 2. This figure shows assigned component positions using the example of the herb layer dataset of the workflow. The absolute position in a workflow is defined by the distance back to the data source. The numbers on the components display the distance count back to the data source. The assignment of positions starts with 0. Quantifying component identity Based on our analysis, we classified workflow components into 12 tasks a priori ( Table 1 ). We then used text mining tools to characterize the components automatically. For this we used the presence/absence of R commands and libraries as qualitative values, the number of input and output ports, the number of datasets a component is connected with, as well as the count of code lines. This allowed us to match the a priori tasks with the automatically gathered characteristics. We used non metric multidimensional scaling (NMDS) 19 to find the two main axes of variation in the multidimensional space defined by the characteristics. We then performed linear regression to identify which of the characteristics and which of the a-priori tasks could explain variation of the two NMDS axes. We could further compare task complexities. For this we used a Kruskal-Wallis test and a post-hoc Wilcoxon test, since residuals where not normally distributed (Shapiro-Wilk test). Table 1. Workflow component tasks defined a priori in the analysis of a biodiversity effect on forest carbon pools and their relation to the data oriented motifs identified by 10 . Identities Description Motif data source Access (remote or local data) Data retrieval type transformation Transform the type of a variable (e.g to numeric) Data preparation merge data Match and merge data Data organization data aggregation Aggregate of data Data organization create new vector Create a vector filled with new data Data curat./clean data imputation Impute data (e.g linear regressions on data subsets) Data curat./clean modify a vector Modify a complete vector by a factor or basic arithmetic operation Data organization create new factor Create a new factor Data organization data extraction Extract data values (e.g from comment strings) Data curat./clean sort data Sort data Data organization data modeling All kinds of model comparison related operations (ANOVA, AIC) Data analysis Quantifying quality and usage of data sources Data for the workflow comes from several data sources, differing in the number of columns as well as the number of processing steps needed within the workflow. We here introduce two measures of data column usage, one relative to the data source and one relative to the number of workflow components processing the data. We further introduce a quality measure of a data column by identifying a critical component within the workflow that signifies the actual analyses that answers our scientific question. We thus have workflow components that prepare data for the analysis and we have (few) workflow components that consume data for the analyses ( Figure 3 ). Figure 3. The usage and quality measure on an example dataset. The components marked with a P represent preparation steps of a variable. Here we see three preparation steps so the quality is 4. The components marked with C and I represent direct consumption and an indirect influence. Those together build the variable usage together with all following influenced components. As explained above, output ports of a data source in the workflow directly relate to data columns in the data set. Thus, the number of available ports of a data source is the “width” of a dataset, or the number of data columns. Thus, the usage of a data column in relation to the data source was calculated as the ratio of ports actually used in the workflow to the ports that were not used. This allowed us to relate the number of unused ports to the number of available ports of a data source. In contrast, the usage of a data column in relation to the workflow is quantified by the total number of workflow components processing the data, before and after the actual analysis ( equation 3 ). Similarly, the quality of a data column in relation to the workflow is quantified by the number of workflow components before the critical analysis, including this workflow component itself ( equation 4 ). Thus, the higher the quality value, the lower the column’s quality. We can now compare datasets based to the usage and quality of their data as it is processed in the workflow. We did this using Kruskal-Wallis and post-hoc Wilcoxon tests. u s a g e = ∑ i = 1 n c o n i + ∑ j = 1 n i n f j ( 3 ) q u a l i t y = ∑ i = 1 n p r e p i + 1 ( 4 ) The beginning matters - results from our workflow meta analysis The workflow analysing carbon pools on a gradient of biodiversity consisted of 71 components in 16 workflow positions consuming the data of 8 datasets ( Table 2 ). The data in the workflow was manipulated via 234 lines of R code. The number of code lines per component ranged between 1 (e.g component plot_2_numeric) and 23 (component impute_missing_tree_heights) with an overall mean of 3.3 (± 3.98 SD). See Figure 3 for a graphical representation of the workflow. Table 2. The workflow positions listed along with the unique component tasks they contain and the count of components per position. Position Tasks Component count 0 data source 8 1 data type transformation, data extraction, create new vector 20 2 merge data, data imputation, modify a vector, create new vector, data aggregation, sort data 10 3 create new factor, merge data, data aggregation 7 4 merge data, create new vector, data imputation, modify a vector 7 5 create new vector, merge data, modify a vector 5 6 merge data, data imputation, modify a vector 3 7 data imputation, create new vector, data aggregation 3 8 create new vector 2 9 merge data 1 10 modify a vector 1 11 data aggregation 1 12 modify a vector 1 13 create new vector 1 14 data modeling 1 Although we aimed to keep the components streamlined and simple, the absolute and relative component complexity varied markedly. The absolute complexity ranged between 4 and 41 with an overall mean of 9.25 (± 6.77 SD), (Summary: Min. 4.0, 1st Qu. 4.0, Median 8.0, Mean 9.2, 3rd Qu. 12.0, Max. 41.0). Relative component complexity ranged between 0.69% (e.g component calculate_carbon_mass_from_biomass) and 7.03% (e.g component: add_missing_height _-broken_trees) with an overall mean of 1.59 (± 1.16 SD), (Summary: Min. 0.69, 1st Qu. 0.69, Median 1.37, Mean 1.59, 3rd Qu. 2.05, Max. 7.03). Total workflow complexity decreased exponentially from the beginning to the end ( Figure 4 ). The exponential decrease means that the decrease in complexity is steeper in the beginning of the workflow than at the end of the workflow, showing that complexity at the end of the workflow did not differ as much as at the beginning of the workflow. From the three models relating the sum of relative component complexities to workflow position, the one including position as logarithm (AIC = 64.69) was preferred over the one including a linear and a quadratic term for position (AIC = 65.67, delta AIC = 0.71), and the one including position as linear term only (AIC = 72.26, delta AIC = 7.3). Figure 4. Relative workflow complexity along workflow positions could be best described by an exponential model including position as logarithm (R-squared=0.9, F-statistic: 29.16 on 3 and 10 DF, p-value: smaller than 0.001 ***). This figure shows the model back transformed to the original workflow positions. The gray shading displays the standard error. At the same time, relative complexity increased in the course of the analysis ( Figure 5 ), since our model with an intercept and linear term for workflow position (AIC = 325.85) was preferred. However, we will argue later, that this increase was mainly due to a group of workflow components of extreme simplicity at the very beginning of data import, visible in the bottom left of Figure 5 . These workflow components convert text columns into numeric columns in the “data type transformation” task. As we will outline later, we took this as an opportunity to program a feature for our data portal, to convert columns mixed with text and numbers to numeric columns for the EML output. Figure 5. Relative component complexities along the workflow of the carbon analysis. The points are slightly jittered to handle over plotting. At each position in the workflow there are components of different type and complexity. Linear model with positions as predictor for relative components complexity. R-squared: 0.09, F-statistic: 6.57 on 1 and 61 DF, p-value: 0.01285. We could group workflow components according to their a priori assigned tasks using text mining. The non metric multidimensional scaling had a stress value of 0.17 using 2 main axes of variation. Several of the parameters, including specific commands of R code, were correlated to axes scores ( Table 3 ). Our a priori defined tasks could be significantly separated in the parameter space ( r 2 0.58, p-value 0.001): the first axis spans between the workflow tasks “data aggregation” and “modify a vector” while the second one spans between the tasks “data extraction” and “data type transformation” ( Figure 6 ). Table 3. Results of the non metric multidimensional scaling of the component characteristics. Signif. codes: 0 ‘***’ 0.001 ‘**’ 0.01 ‘*’ 0.05 ‘.’ 0.1 ‘ ’ 1. P-values based on 999 permutations. Characteristics NMDS1 NMDS2 r2 Pr(>r) sig. abline -0.595118 0.803639 0.0879 0.024 * as.numeric 0.229379 -0.973337 0.3536 0.001 *** attach -0.485526 0.874222 0.0535 0.113 data.frame -0.902072 -0.431586 0.8074 0.001 *** ddply -0.802096 -0.597195 0.3456 0.001 *** detach -0.485526 0.874222 0.0535 0.113 grep -0.211367 0.977407 0.2885 0.001 *** ifelse -0.663695 0.748004 0.3222 0.001 *** is.na -0.759568 0.650428 0.2800 0.001 *** length -0.601172 0.799120 0.0526 0.182 lm -0.199313 0.979936 0.0578 0.157 match 0.445922 0.895072 0.0057 0.849 mean -0.994320 0.106435 0.1346 0.016 * none 0.977510 0.210891 0.5095 0.001 *** plot -0.689676 0.724118 0.0478 0.254 predict -0.595118 0.803639 0.0879 0.024 * sort -0.417914 -0.908487 0.0071 0.954 strsplit 0.016877 0.999858 0.0633 0.059 . subset -0.788938 -0.614472 0.0212 0.821 sum -0.658793 -0.752324 0.1278 0.004 ** summary 0.022488 -0.999747 0.0043 0.969 unique -0.485526 0.874222 0.0535 0.113 unlist 0.016877 0.999858 0.0633 0.059 . vector -0.269756 0.962929 0.4583 0.001 *** which -0.421341 0.906902 0.4582 0.001 *** write.csv 0.022488 -0.999747 0.0043 0.969 count of R functions -0.796351 0.604834 0.4893 0.001 *** count of codelines -0.530470 0.847704 0.5392 0.001 *** domain count 0.920778 -0.390087 0.0142 0.704 count packages per component -0.781994 -0.623286 0.4004 0.001 *** Figure 6. Non metric multi-dimensional scaling using the qualitative and quantitative component characteristics. The scaling was created using the R package vegan with the Bray-Curtis distance. The large labels represent the workflow tasks. The smaller text annotations represent the characteristics used. They are slightly jittered by a factor of 0.2 horizontally and vertically to handle over plotting. Our workflow tasks had similar complexity, with only one exception: the task “data type transformation” was less complex (Kruskal-Wallis chi-squared = 41.97, df = 9, p < 0.001) than the tasks create new vector, data aggregation, data imputation and merge data ( Figure 7 ). Again, data type transformation was only used at the beginning of the workflow to transform columns mixing numbers and text to numbers. Figure 7. The median, 25% and 75% quantiles of the relative component complexities for the component tasks. Letters refer to: a=create new factor, b=create new vector, c=data aggregation, d=data extraction, e=data imputation, f=data modeling, g=data type transformation, h=merge data, i=modify a vector, j=sort data. The small dots are the relative complexities, the diamonds the means. The whiskers are 25% quantile - 1.5 * IQR and 75% quantile + 1.5 * IQR, big black circles are outliers. Signific.: * = 0.05, ** = 0.001. Data usage in relation to data sources was higher in smaller data sources. “Wide” data sources, those consisting of many columns, contributed less to the analysis than “smaller” data sources with fewer columns. While on average 37.4% of the columns in the data sources were used, a linear regression showed that the number of columns not used increased with the total number of columns available per data source ( Figure 8 ). Figure 8. The linear regression shows the dependency between the total column count of the datasets and the count of unused columns. The gray shaded area represents the standard error. Linear model with total columns as predictor for unused columns: R-squared: 0.925, F-statistic: 74.02 on 1 and 6 DF, p-value: 0.0001. At the same time, data usage in relation to the workflow was similar for all data sources. Data column usage within the workflow ranged between a minimum of 1 and a maximum of 16 with an overall mean of 6.38 (± 4.25 SD). Although usage was different between datasets (Kruskal-Wallis, chi-squared = 18.05, df = 7, p-value = 0.012), a post hoc group wise comparison could not identify the differences (Wilcoxon test). The data column quality, the amount of processing steps needed to transform data for the analysis (see above), was also similar for all data sources. It ranged between a minimum of 1 and a maximum of 10 with an overall mean of 3.54 (± 2.44 SD). There were no differences in the data column quality between data sources (Kruskal-Wallis, chi-squared = 10.9, df = 7, p-value = 0.14). Discussion We showed that workflow complexity and data usage of a typical analysis in BEF can be quantified using relatively simple qualitative and quantitative measures based on commands, code lines, and variable numbers. It is the data aggregation, merging, and subsetting part at the beginning that complicates workflows. In our case, workflow complexity decreased exponentially in the course of the analysis ( Figure 4 ). Similarly, 10 found that the data transformation, merging and aggregation steps at the beginning of an analysis complicate workflows. Thus, simplifying data processing steps would greatly increase workflow simplicity. Here we argue that data simplicity could be fostered by providing feedback to data providers on the usage and quality values of the columns in their datasets 14 . In our workflow, “wide” datasets consisting of many columns, contributed less to the analysis than smaller datasets with fewer columns ( Figure 8 ). The more data columns a dataset has, the more difficult it is to understand what the dataset is about. The more data columns it has, the more difficult it is to describe it. The high number of columns in datasets resulting from fieldwork in ecology is a result of the effort to provide comprehensive information in one file only, often including different experimental designs and methodologies. These datasets result from copying field notes that are related to the same research objects, but combine information from different experiments. For example, a field campaign on estimating the amount of woody debris on a study site might count the number and size of branches found. At the same time, as one is already in the field, other branches might be used to find general rules for branch allometries. Thus, the same sheet of paper will be used for two different purposes. While this approach is efficient in terms of time and field work effort, it leads to highly complicated datasets. Separating the datasets into two, one for the dead matter, the other for branch allometries would decrease the number of columns of a dataset and increase the value of the dataset for the analysis of carbon budgets. Combining data from different sources for meta-analyses could especially benefit from a more atomic way of storing data. Atomic means data particles (e.g. columns) stored separately, described via metadata and linked to ontological concepts. But in ecology the linking of data is rarely performed due to the high heterogeneity of data and concepts. With emerging technologies and a broader acceptance of metadata and ontological frameworks in ecology, datasets could be created automatically using logical constraints built from available atomic data particles. So, a query could return horizontal and vertically subsetted data products (facets) that, in the best case scenario, represent a 100% match directly usable in a meta-analysis 20 . Providing feedback to data providers about the complexity of their data may thus be an important step in leveraging the readability of scientific workflows and supporting the reproducibility of data-driven science. This is especially true for “dark” data, the small and complex datasets in the long tail of big data 4 . We are presently witnessing a growing concern over the loss of data 21 , which is mostly due to the illegibility of datasets due to missing metadata and the lack of adherence to standard formats. Researchers still do not have training in data management. This concern in losing complex data has led to the invention of tools like DataUP that help to annotate data within Excel, or BEFdata to import Excel files, since this spreadsheet software is mainly used for data storage by researchers. At the same time, opportunities are emerging to publish datasets ( Ecological Archives is only one alternative, there are also data journals http://www.hindawi.com/dpis/ecology/ ) and to provide measures of impact for data 22 . Providing means for data quality feedback may also be helpful for propagating data ownership, which remains an unsolved problem 14 and a major concern in data sharing 23 , 24 . We show that in our analysis, all data columns had a similar usage factor in relation to the workflow. Such usage factors could help to quantify data ownership as they allow one to quantify the amount a certain column or dataset has contributed to derive the results of an analysis. Since we used the Kepler workflow software to execute R scripts, we made use of Kepler’s interface components. Our text-based approach of quantifying complexity will thus be useful mainly in the context of workflows that work with custom scripts. However, the Kepler workflows are stored as XML files and our approach could thus be generalized to other components in Kepler, or other workflow systems working with XML as an exchange format. Even if workflow programs do not store their workflows in human readable form, their source code could be analysed using similar text-based measures. Providing complexity measures at the level of workflow components might help in re-using and adapting workflows. To date, the workflow platform “MyExperiment” is used by 7500 members and presents 2500 workflows for reuse and adaption, however, there is only a rating for the workflow. Offering complexity measures for workflow components may help to identify bottlenecks in existing workflows and help users to adapt components of workflows 25 , http://www.myexperiment.org/workflows?query=ecology ). A further step in finding and adapting workflows would be the possibility to identify useful workflow components. Here we show that we could identify workflow tasks using text mining ( Figure 6 ). In our case, we could identify one task of very low complexity (data type transformation). This task was very simple and constituted the second axis of our NMDS ( Figure 6 ). Components of this task mostly convert text vectors into numeric vectors. Having text vectors that could actually be interpreted as numbers stems from a “weakness” of EML, in that it does not allow text in data columns that store numbers. However, it is very common that scientists comment missing values or numbers below or above a measuring uncertainty threshold using text. To store the datasets in EML format forces the data provider to label the whole column as a text column. As a consequence of having identified this simple and repetitive task of converting text to numbers, we have now added a feature to the BEFdata platform application that automates the conversion. We now offer two ways of exporting the data as comma separated values (CSV), one using the original data and one duplicating numeric columns that contain text, one column containing only the numbers, the other containing only the text. This is also the procedure suggested by DataUP for dealing with columns mixing text and numbers 26 . The BEFdata EML export now only offers the data in the latter format, so that numbers are no longer mixed with text. This is an example of how the analysis of a scientific workflow can guide towards useful automation features for data repositories. Summary Simplicity of data sources is the key to simple workflows, but we currently lack feedback mechanisms for quantifying data simplicity. We show that simple text-based measures could already be helpful in quantifying data and workflow complexity. Providing feedback on data complexity as well as the complexity of workflow components may not only foster simplicity and reuse, but additionally may present a means of propagating data ownership through interdisciplinary synthesis efforts and highlights the importance of the underlying primary research data. Data availability figshare: Data used to quantify the complexity of the workflow on biodiversity-ecosystem functioning, http://dx.doi.org/10.6084/m9.figshare.1008319 27 Author contributions C.T.P., K.N., S.R., C.W. and H.B. substantially contributed to the work including the conceptualization of the work, the acquisition and analysis of data as well as critical revision of the draft towards the final manuscript. Competing interests No competing interests were disclosed. Grant information The data was collected by 7 independent projects of the biodiversity - ecosystem functioning - China (BEF-China) research group funded by the German Research Foundation (DFG, FOR 891). Acknowledgements Thanks to all the data owners from the BEF-China experiment who contributed their data to make this analysis possible. The research data is not all publicly available currently but will be in future. The datasets are linked and the ones publicly available are marked accordingly and can be downloaded using the following links. By dataset these are: Wood density of tree species in the Comparative study plot (CSPs) : David Eichenberg, Martin Böhnke, Helge Bruelheide. Tree size in the CSPs in 2008 and 2009 : Bernhard Schmid, Martin Baruffol. Biomass of herb layer plants in the CSPs, separated into functional groups (public): Alexandra Erfmeier, Sabine Both. Gravimetric Water Content of the Mineral Soil in the CSPs : Stefan Trogisch, Michael Scherer-Lorenzen. Coarse woody debris (CWD): Collection of data on dead wood with special regard to snow break (public): Goddert von Oheimb, Karin Nadrowski, Christian Wirth. CSP information to be shared with all BEF-China scientists : Helge Bruelheide, Karin Nadrowski. CNS and pH analyses of soils depth, increments of 27 Comparative Study Plots: Peter Kühn, Thomas Scholten, Christian Geißler. Faculty Opinions recommended References 1. Michener WK, Jones MB: Ecoinformatics: supporting ecology as a data-intensive science. Trends Ecol Evol. 2012; 27 (2): 85–93. PubMed Abstract | Publisher Full Text 2. Altintas I, Berkley C, Jaeger E, et al. : Kepler: an extensible system for design and execution of scientific workflows. Proceedings. 16th International Conference on Scientific and Statistical Database Management . 2004; 423–424. Publisher Full Text 3. Ewa D, Gurmeet S, Mei-hui S, et al. : Pegasus: a framework for mapping complex scientific workflows onto distributed systems. 2005; 13 : 219–237. Reference Source 4. Heidorn PB: Shedding light on the dark data in the long tail of science. Library Trends. 2008; 57 (2): 280–299. Publisher Full Text 5. Gries C, Porter JH: Moving from custom scripts with extensive instructions to a workflow system: use of the Kepler workflow engine in environmental information management. In M B Jones and C Gries, editors, Environmental Information Management Conference 2011 . Santa Barbara, CA University of California. 2011; 70–75. Reference Source 6. Ewa D, Gurmeet S, Mei-hui S, et al. : Pegasus: a framework for mapping complex scientific workflows onto distributed systems. 2005; 13 : 219–237. Reference Source 7. Oinn T, Greenwood M, Addis M, et al. : Taverna: lessons in creating a workflow environment for the life sciences. Concurrency Computation: Pract Exp. 2006; 18 (10): 1067–1100. Publisher Full Text 8. Bowers S, Ludäscher B: Towards Automatic Generation of Semantic Types in Scientific Workflows. Web Information Systems Engineering–WISE 2005 Workshops Proceedings . 2005; 3807 . : 207–216. Publisher Full Text 9. McCabe TJ: A complexity measure. In Proceedings of the 2nd international conference on Software engineering . ICSE ’76, Los Alamitos, CA USA. IEEE Computer Society Press. 1976; 2 (4): 308–320. Publisher Full Text 10. Garijo D, Alper P, Belhajjame K, et al. : Common motifs in scientific workflows: An empirical analysis. IEEE 8th International Conference on E-Science . 2012; 1–8. Publisher Full Text 11. Gil Y, González-Calero PA, Kim J, et al. : A semantic framework for automatic generation of computational workflows using distributed data and component catalogues. J Experimental Theoretical Artificial Intelligence. 2011; 23 (4): 389–467. Publisher Full Text 12. Nadrowski K, Ratcliffe S, Bönisch G, et al. : Harmonizing, annotating and sharing data in biodiversityecosystem functioning research. Methods Ecol Evol. 2013; 4 (2): 201–205. Publisher Full Text 13. Parsons MA, Godoy O, LeDrew E, et al. : A conceptual framework for managing very diverse data for complex, interdisciplinary science. J Info Sci. 2011; 37 (6): 555–569. Publisher Full Text 14. Ingwersen P, Chavan V: Indicators for the Data Usage Index (DUI): an incentive for publishing primary biodiversity data through global information infrastructure. BMC Bioinformatics. 2011; 12 (Suppl 15): S3. PubMed Abstract | Publisher Full Text | Free Full Text 15. Bruelheide H: The role of tree and shrub diversity for production, erosion control, element cycling, and species conservation in Chinese subtropical forest ecosystems. 2010. Reference Source 16. Fegraus EH, Andelman S, Jones MB, et al. : Maximizing the value of ecological data with structured metadata: an introduction to ecological metadata language (eml) and principles for metadata creation. Bulletin of the Ecological Society of America. 2005; 86 (3): 158–168. Publisher Full Text 17. R Development Core Team. R: A Language and Environment for Statistical Computing. R Foundation for Statistical Computing, Vienna, Austria, ISBN 3-90005107-0, 2008. Reference Source 18. Burnham KP, Anderson DR: Model selection and multimodel inference: a practical information-theoretic approach. Springer. 2002; 172 . Reference Source 19. Dixon P: Vegan, a package of r functions for community ecology. J Vegetation Sci. 2003; 14 (6): 927–930. Publisher Full Text 20. Leinfelder B, Bowers S, Jones MB, et al. : Using Semantic Metadata for Discovery and Integration of Heterogeneous Ecological Data. Language. 2011; 92–97. 21. Nelson B: Data sharing: Empty archives. Nature. 2009; 461 (7261): 160–163. PubMed Abstract | Publisher Full Text 22. Piwowar H: Altmetrics: Value all research products. Nature. 2013; 493 (7431): 159–159. PubMed Abstract | Publisher Full Text 23. Cragin MH, Palmer CL, Carlson JR, et al. : Data sharing, small science and institutional repositories. Philos Trans A Math Phys Eng Sci. 2010; 368 (1926): 4023–38. PubMed Abstract | Publisher Full Text 24. Xiaolei H, Hawkins BA, Fumin L, et al. : Willing or unwilling to share primary biodiversity data: results and implications of an international survey. Conservation Letters. 2012; 5 (5): 399–406. Publisher Full Text 25. De Roure D, Goble C, Bhagat J, et al. : myexperiment: Defining the social virtual research environment. In eScience, 2008. eScience ’08. IEEE Fourth International Conference on . 2008; 182–189. Publisher Full Text 26. DataUp. The dataup tool. developed by the california digital library and microsoft research connections with funding from gordon and betty moore foundation. 2013. Reference Source 27. Pfaff CT, Nadrowski K, Ratcliffe S, et al. : Data used to quantify the complexity of the workflow on biodiversity-ecosystem functioning. Figshare. 2014. Data Source Comments on this article Comments (0) Version 2 VERSION 2 PUBLISHED 14 May 2014 ADD YOUR COMMENT Comment Author details Author details 1 Institute of Special Botany and Functional Biodiversity, University of Leipzig, 04103, Leipzig, Germany 2 Institute of Biology, Martin Luther University Halle-Wittenberg, 06108, Halle, Germany Competing interests No competing interests were disclosed. Grant information The data was collected by 7 independent projects of the biodiversity - ecosystem functioning - China (BEF-China) research group funded by the German Research Foundation (DFG, FOR 891). The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript. Article Versions (2) version 2 Revised Published: 17 Nov 2014, 3:110 https://doi.org/10.12688/f1000research.3940.2 version 1 Published: 14 May 2014, 3:110 https://doi.org/10.12688/f1000research.3940.1 Copyright © 2014 Pfaff CT et al . This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. Data associated with the article are available under the terms of the Creative Commons Zero "No rights reserved" data waiver (CC0 1.0 Public domain dedication). Download Export To Sciwheel Bibtex EndNote ProCite Ref. Manager (RIS) Sente metrics Views Downloads F1000Research - - PubMed Central info_outline Data from PMC are received and updated monthly. - - Citations open_in_new 0 open_in_new 0 open_in_new SEE MORE DETAILS CITE how to cite this article Pfaff CT, Nadrowski K, Ratcliffe S et al. Readable workflows need simple data [version 1; peer review: 1 approved with reservations] . F1000Research 2014, 3 :110 ( https://doi.org/10.12688/f1000research.3940.1 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS track receive updates on this article Track an article to receive email alerts on any updates to this article. TRACK THIS ARTICLE Share Open Peer Review Current Reviewer Status: ? Key to Reviewer Statuses VIEW HIDE Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Version 1 VERSION 1 PUBLISHED 14 May 2014 Views 0 Cite How to cite this report: Missier P. Reviewer Report For: Readable workflows need simple data [version 1; peer review: 1 approved with reservations] . F1000Research 2014, 3 :110 ( https://doi.org/10.5256/f1000research.4221.r4788 ) The direct URL for this report is: https://f1000research.com/articles/3-110/v1#referee-response-4788 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 04 Jun 2014 Paolo Missier , School of Computing Science, Newcastle University, Newcastle upon Tyne, UK Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.4221.r4788 The title and abstract do indeed summarise the purpose and content of the paper adequately. While the goals of the work are laudable, the framework proposed to go about them is not convincing. I see two main problems, firstly with using a ... Continue reading READ ALL The title and abstract do indeed summarise the purpose and content of the paper adequately. While the goals of the work are laudable, the framework proposed to go about them is not convincing. I see two main problems, firstly with using a single case study to drive the definition and analysis of data and process complexity, and thus to derive results can hardly have general validity. Secondly, by basing the analysis on some questionable assumptions. In what follows, I try to elaborate on these points. The paper makes a strong case for simplicity of data and components, however that is based on a single sample. This is hardly justified, and at odds with the wealth of quantitative research methods machinery deployed to analyse workflows and data and derive the metrics proposed in the paper. It would be good to clarify whether the paper's focus is on the method -- whereby the case study is just an illustrative example, and without any pretense of drawing general conclusions, or on the actual results, which given the very limited evaluation, are questionable. Regarding assumptions, it is stated "data complexity could be measured by the complexity of their workflow". How general is this meant to be? I am not sure a process-independent notion of data complexity is given in the paper, but I believe it should be, to clarify the argument. Here complexity seems to be based on how many different usages (and reuse) the data supports, which is fine perhaps, but only one of many possible criteria. I am also suspicious of process complexity criteria based on lines of code, especially in workflows that are composed of discrete components, often pre-existing and part of libraries. Kepler is idiosyncratic in this, as it assumes most actors are ad hoc programs. More generally, workflow is about coarse-grained composition (eg of third party services), and local coding decisions matter a lot less than in hand-crafted code. LOC is a very crude measure of complexity. Just as old, but perhaps more appropriate, is the notion of "function points" whereby you express complexity in terms of functionality realised by a component -- regardless of how much code is required to implement a certain function. LOC alone is also at odds with the idea that languages like R sit on powerful packages, which make for succinct but expressive code. How do you compare R code that implements a whole algorithm in R with one that simply invokes a lib function to achieve the same result? One could also argue, reading on pg 7 (col 2), that you may be measuring personal coding style rather that actual process complexity. Other assumptions along the way seem contrived and overfit the (single) example, for instance "output ports of a data source in the workflow directly relate to data columns in the data set". (pg 5,6) In the same section, questionable conclusions follow from this assumption. So overall, I think the quantitative methods used in the paper are interesting, but they are applied to a framework where a number of initial assumptions are questionable, and seem to be driven by one single example. A few specific comments: pg 3 - Complexity: The point is about programs with control structures, but scientific workflows traditionally are dataflows. So does the same notion of complexity apply here? pg 4: I feel there is probably too much detail on the science and its results here, which is not the focus of the paper and can be distracting (and uninteresting unless you know the specific science). pg 5 col 2: Need to explain AIC. pg 7: I found table 2 interesting and generally useful. In contrast, Table 3 is a bit of a mystery to me. Competing Interests: No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Missier P. Reviewer Report For: Readable workflows need simple data [version 1; peer review: 1 approved with reservations] . F1000Research 2014, 3 :110 ( https://doi.org/10.5256/f1000research.4221.r4788 ) The direct URL for this report is: https://f1000research.com/articles/3-110/v1#referee-response-4788 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Author Response 03 Nov 2014 Claas-Thido Pfaff , University of Leipzig, Germany 03 Nov 2014 Author Response Dear Paolo Missier, First of all thank you for your valuable input which gave us the opportunity to sharpen the focus of our paper. Your main argument was that we cannot ... Continue reading Dear Paolo Missier, First of all thank you for your valuable input which gave us the opportunity to sharpen the focus of our paper. Your main argument was that we cannot prove our points because we are using a single case study. At the same time you said we should clarify whether our focus is on the results of the analysis - based on only one use case - or the metrics derived for illustrating complexity. However, our main focus is neither on the specific results of this use case, nor on the metrics. We are writing an opinion paper, and both, the use case and the metrics, are illustrations of our opinion. As you say in your comment, - and we take this as a compliment -, we want to “make a strong case for simplicity of data and workflow components”. Although it is not our intention to use the case study as proof, our paper is accompanied by many statistical analyses and plots. This may fool the reader in believing that we want to present a research article. However, we think that our plots are very useful for other data managers and scientists in illustrating why it is worthwhile to invest energy into simplifying datasets. This is especially the case for files from the long tail of big data, which are handcrafted, and relatively small data sets resulting from fieldwork and not from automated sensors. To be able to illustrate the problem of merging these files - which is our day to day work as hybrids of data managers and researchers - we chose this case study, as it is representative for our work and the work of our fellow data managers we spoke to. We also think that it is highly useful to illustrate our difficulties in data re-use. Reworking our text in response to your questions, we scaled down the method descriptions and put a stronger focus on the opinion parts of the paper. We completely reworked the text in many passages and provide here some examples: For example, in the abstract we changed the sentence (page 1): “We illustrate our points using a typical analysis in BEF research...” to: “To illustrate our points we chose a typical analysis in BEF research...”. At the beginning of the introduction we sharpened the opinion aspect of the paper, instead of the sentence (page 2): “However, there is a lack of papers that discuss workflow components within an analysis including data processing.” we now write : “Here we argue that there is a need for quality measures of workflow components, which include scripts, as well as for the underlying data sources. Failure to reuse workflows and available research data is not only a waste of time, money and effort but also represents a threat to the basic scientific principle of reproducibility. Providing feedback mechanisms on the data and workflow component complexity has the great potential to increase the readability and the reuse of workflows and its components.” and other small changes like: before: “We thus suggest that focusing on simplifying ...” after: “We argue that focusing on simplifying...” We further added a new paragraph to the discussion on our methods. We explain, that we want to illustrate the possibility to use simple text mining techniques in providing immediate feedback to data providers or workflow creators. We also add additional avenues that could be taken to quantify complexity of further workflow components or scriptlets (last paragraph discussion): “We here exemplify how to quantify the complexity as well as the quality and the usage of data in scientific workflows, using simple qualitative and quantitative measures. Our means are not meant to be exhaustive but rather could serve as a starting point for discussion towards the development of more sophisticated complexity feedback mechanisms for data providers and workflows creators. Our example workflow strongly relies on the interface component of Kepler connecting to the R statistical environment for the purpose of data manipulation and analysis. Thus the means we provide to measure complexity and quality are adapted to that specific workflow situation. However, adapting our means to further components that work as interfaces to other programming languages should be straightforward. Further complexity attributes could be the inclusion of the variable types of workflow components or a ratio capturing the enrichment or reduction of data consumed by the component. Providing complexity measures at the level of workflow components might help in adapting workflows towards a better readability and reusability and thus improve their value for reuse. Additionally it can guide the restructuring and simplification of data for a better use in workflows, a better understandability and reuse.” We further agree, that we have provided too little explanation of what we mean by “complexity”. We thus added a paragraph in the introduction, section complexity and identity, to define the aspect of complexity we are concerned about (section complexity and identity, first paragraph): “Here we are interested in workflows that begin with the cleaning, the aggregation, and the imputation of research data. These first steps can make up to 70% of the whole workflow. As data managers and researchers, we want to improve the readability of such workflows whether they are scripts or graphs. Our concept of complexity thus should capture the effort and time needed to understand and reuse such workflows. Regarding the complexity of source code we found similar incentives that provide quality measures. The Code Climate service for example provides code complexity feedback to programmers in many different programming languages (https://codeclimate.com/?v=b). Their complexity measures take the number of lines of code as well as the repetition of identical code lines into account.” Our operationalisation of data complexity is based on this approach to workflow complexity. We explain in the same section (paragraph 2 - 3): “Quantifying data complexity is not as straight forward as workflow component complexity. Datasets used for synthesis in research collaborations often consist of “dark” data, lacking sufficient meta- data for reuse…. … Here we argue that data complexity can be quantified by looking at the workflow components needed to aggregate and focus the data for analysis. One of the paradigms of data- driven science is that the analysis should be accompanied by it’s data. We argue that at the same time, data should be accompanied by workflows that offer meaningful aggregation of the data. Data complexity could then be measured by the complexity of their workflows." More technically speaking, you asked for a process independent complexity measure and criticised the use of line of codes, asking how we deal with hidden complexity when using whole script packages with only one line of code. However, the overwhelming majority of data merging efforts we see in our work as data managers are script based, and are not meant to be reused in the same way as software programs. For this reason, function points do not make sense for them. In addition, we do not only use lines of code in our complexity measure. We also include the number of packages used, for example. In our paper, in the part on the Example workflow, section Quantifying workflow complexity, we explain: To quantify the complexity of the components we used the number of code lines (loc), the number of R commands (cc) and R packages used (pc), as well as the number of input and output ports (cp) of the components (equation 1). However, we are aware that we only use simple and crude methods to assess complexity. As this is not a research paper, but an illustration for an opinion paper, we do not want to focus on the methods too much. On the other hand, we think that it would be good to develop complexity measures for these type of data merging scripts as well as their components and data sources. For this reason we added a whole new paragraph on our methods to the discussion, as stated above. We agree that our measure of complexity is within one personal coding style only. This has the disadvantage that there is only one person or coding style, but the advantage that the differences between the complexities of components is not confused by different coding styles. In most cases, data merging efforts will be done by one person only. We do not want to generalise for all researchers as to which commands or packages they choose. But from our experience, our coding example is representative for data merging exercises. Independently from coding style, most effort goes into the first data cleaning and aggregation steps, including the effort to understand the different data sets. Whatever means we find to give a feedback on how much effort is needed to reuse this data, it is worthwhile giving it back to the data providers. In the following, we answer to specific comments: Paolo Missier: " Other assumptions along the way seem contrived and overfit the (single) example, for instance "output ports of a data source in the workflow directly relate to data columns in the data set". (pg 5,6) In the same section, questionable conclusions follow from this assumption ." We agree that this formulation is misleading. Since we use the EML actor of Kepler to import data, the “output ports” are always the data columns. We did not want to imply a causality here. Data columns appear as output ports in the Kepler actor, because this is how the EML actor works. We reformulate this sentence accordingly. Indeed, the paragraph works without even using the whole sentence (see page 3, section: quantify quality and usage of data) Before: “As explained above, output ports of a data source in the workflow directly relate to data columns in the data set. Thus, the number of available ports of a data source is the “width” of a dataset, or the number of data columns. Thus, the usage of a data column in rela- tion to the data source was calculated as the ratio of ports actually used in the workflow to the ports that were not used. This allowed us to relate the number of unused ports to the number of available ports of a data source." Now: “For our analysis we only used a subset of the data columns available in each data source. We therefore quantified the “data usage” of a data source as the ratio of data columns used for the analysis to the total number of data columns in that data source." Paolo Missier: "pg 3 - Complexity: The point is about programs with control structures, but scientific workflows traditionally are dataflows. So does the same notion of complexity apply here?" No, it doesn’t. We now provide a definition of complexity that clarifies that we are interested in the amount of effort and time that has to be invested in data or workflow reuse (see above). Paolo Missier: "pg 4: I feel there is probably too much detail on the science and its results here, which is not the focus of the paper and can be distracting (and uninteresting unless you know the specific science)." We have reordered and shortened the paragraphs on our workflow example. However some the information is interesting for the general reader and are required for the overall understanding. For example that the data sources come from independent projects and are archived in a common platform as well as some basics on workflows. But we have shortened the information on the scientific analysis to one paragraph. page 3, section: biodiversity effects on subtropical carbon stocks: “Our example workflow is part of an ongoing study that measures biodiversity effects on subtropical carbon stocks and flows. It is typical for synthesis tasks in collaborative research projects in that it combines eight datasets collected by seven independent research groups collaborating within the BEF-China research platform ( www.bef-china.de ). Data is archived, harmonized, and exchanged using the BEFdata web application (citation!!!). Data is exported in EML format and as such imported into the Kepler Workflow system. The data describes carbon pools from soil, litter, […] from the years 2008 and 2009 on the observational plots of the BEF-China research platform spanning a gradient from 22 to 116 years of plot age and 15 to 35 tree species. Our example workflow merges the data and terminates in a linear model relating biomass pools to plot age and plot diversity. It shows that carbon pools increase with stand age, however, in plots with high species richness this increase was less steep (p-values).” Paolo Missier: "pg 5 col 2: Need to explain AIC." AIC is explained and cited in the methods part (Akaikes Information criterion). page 3, right column, section: quantify component identity: “For this we used linear models which have been compared using the Akaike Information Criteria (AIC) to select for the most parsimonious model.” Paolo Missier: "pg 7: I found table 2 interesting and generally useful. In contrast, Table 3 is a bit of a mystery to me." Table 1, 3 and Figure 6 are different perspectives on the same topic. Figure 6 shows the ordination result using multidimensional scaling (NMDS) of workflow component characteristics. The NMDS results in 2 axes that span the highest variation of components in the characteristics space. Thus NMDS1, the first axis, spans the highest variation between components. The component identities in Table 1 as well as the component characteristics in Table 3 were later compared to the axes scores of the NMDS axes. These are the measures given in Table 2. We have changed the text in the captions to point out the relatedness of the figure and the tables. We additionally included and example in how to interpret measures in Table 2 when comparing them with Figure 6. Table 1: “Workflow component identities defined a priori and their relation to the data oriented motifs identified by (10). Figure 6 plots the a priori defined identities of the workflow components to the characteristics we measured from each component a posteriori. Characteristics include lines of codes or specific commands (Table 3). ” Table 2: “Characteristics of workflow components used to assess variation between components by means of non metric multidimensional scaling (NMDS). Characteristics include lines of code, use of packages, as well as specific commands (see text for further detail). Figure 6 plots the two axes of the NMDS. r2, Pr(>r) and sig. describe R square, Probability, and significance level of a correlation with the characteristic as dependent and the NMDS scores of both axes (NMDS1, NMDS2) as independent variables. For example, “count of code lines” separates workflow components in the NMDS plot, so that components with more code lines are plotted in the upper left quadrant of the plot in Figure 6. Signif. codes…” Figure 6: “Workflow components (points) in reduced component characteristics space (Table 3). We used non metric multidimensional scaling (NMDS, see text for further detail) to reduce the parameter space to two axes. Table 3 lists the regression results of the axes scores on the component characteristics, which are plotted in smaller text here. Table 2 lists the a priori tasks, which are plotted in large labels here. Points are jittered by a factor of 0.2 horizontally and vertically to handle over plotting.” Dear Paolo Missier, First of all thank you for your valuable input which gave us the opportunity to sharpen the focus of our paper. Your main argument was that we cannot prove our points because we are using a single case study. At the same time you said we should clarify whether our focus is on the results of the analysis - based on only one use case - or the metrics derived for illustrating complexity. However, our main focus is neither on the specific results of this use case, nor on the metrics. We are writing an opinion paper, and both, the use case and the metrics, are illustrations of our opinion. As you say in your comment, - and we take this as a compliment -, we want to “make a strong case for simplicity of data and workflow components”. Although it is not our intention to use the case study as proof, our paper is accompanied by many statistical analyses and plots. This may fool the reader in believing that we want to present a research article. However, we think that our plots are very useful for other data managers and scientists in illustrating why it is worthwhile to invest energy into simplifying datasets. This is especially the case for files from the long tail of big data, which are handcrafted, and relatively small data sets resulting from fieldwork and not from automated sensors. To be able to illustrate the problem of merging these files - which is our day to day work as hybrids of data managers and researchers - we chose this case study, as it is representative for our work and the work of our fellow data managers we spoke to. We also think that it is highly useful to illustrate our difficulties in data re-use. Reworking our text in response to your questions, we scaled down the method descriptions and put a stronger focus on the opinion parts of the paper. We completely reworked the text in many passages and provide here some examples: For example, in the abstract we changed the sentence (page 1): “We illustrate our points using a typical analysis in BEF research...” to: “To illustrate our points we chose a typical analysis in BEF research...”. At the beginning of the introduction we sharpened the opinion aspect of the paper, instead of the sentence (page 2): “However, there is a lack of papers that discuss workflow components within an analysis including data processing.” we now write : “Here we argue that there is a need for quality measures of workflow components, which include scripts, as well as for the underlying data sources. Failure to reuse workflows and available research data is not only a waste of time, money and effort but also represents a threat to the basic scientific principle of reproducibility. Providing feedback mechanisms on the data and workflow component complexity has the great potential to increase the readability and the reuse of workflows and its components.” and other small changes like: before: “We thus suggest that focusing on simplifying ...” after: “We argue that focusing on simplifying...” We further added a new paragraph to the discussion on our methods. We explain, that we want to illustrate the possibility to use simple text mining techniques in providing immediate feedback to data providers or workflow creators. We also add additional avenues that could be taken to quantify complexity of further workflow components or scriptlets (last paragraph discussion): “We here exemplify how to quantify the complexity as well as the quality and the usage of data in scientific workflows, using simple qualitative and quantitative measures. Our means are not meant to be exhaustive but rather could serve as a starting point for discussion towards the development of more sophisticated complexity feedback mechanisms for data providers and workflows creators. Our example workflow strongly relies on the interface component of Kepler connecting to the R statistical environment for the purpose of data manipulation and analysis. Thus the means we provide to measure complexity and quality are adapted to that specific workflow situation. However, adapting our means to further components that work as interfaces to other programming languages should be straightforward. Further complexity attributes could be the inclusion of the variable types of workflow components or a ratio capturing the enrichment or reduction of data consumed by the component. Providing complexity measures at the level of workflow components might help in adapting workflows towards a better readability and reusability and thus improve their value for reuse. Additionally it can guide the restructuring and simplification of data for a better use in workflows, a better understandability and reuse.” We further agree, that we have provided too little explanation of what we mean by “complexity”. We thus added a paragraph in the introduction, section complexity and identity, to define the aspect of complexity we are concerned about (section complexity and identity, first paragraph): “Here we are interested in workflows that begin with the cleaning, the aggregation, and the imputation of research data. These first steps can make up to 70% of the whole workflow. As data managers and researchers, we want to improve the readability of such workflows whether they are scripts or graphs. Our concept of complexity thus should capture the effort and time needed to understand and reuse such workflows. Regarding the complexity of source code we found similar incentives that provide quality measures. The Code Climate service for example provides code complexity feedback to programmers in many different programming languages (https://codeclimate.com/?v=b). Their complexity measures take the number of lines of code as well as the repetition of identical code lines into account.” Our operationalisation of data complexity is based on this approach to workflow complexity. We explain in the same section (paragraph 2 - 3): “Quantifying data complexity is not as straight forward as workflow component complexity. Datasets used for synthesis in research collaborations often consist of “dark” data, lacking sufficient meta- data for reuse…. … Here we argue that data complexity can be quantified by looking at the workflow components needed to aggregate and focus the data for analysis. One of the paradigms of data- driven science is that the analysis should be accompanied by it’s data. We argue that at the same time, data should be accompanied by workflows that offer meaningful aggregation of the data. Data complexity could then be measured by the complexity of their workflows." More technically speaking, you asked for a process independent complexity measure and criticised the use of line of codes, asking how we deal with hidden complexity when using whole script packages with only one line of code. However, the overwhelming majority of data merging efforts we see in our work as data managers are script based, and are not meant to be reused in the same way as software programs. For this reason, function points do not make sense for them. In addition, we do not only use lines of code in our complexity measure. We also include the number of packages used, for example. In our paper, in the part on the Example workflow, section Quantifying workflow complexity, we explain: To quantify the complexity of the components we used the number of code lines (loc), the number of R commands (cc) and R packages used (pc), as well as the number of input and output ports (cp) of the components (equation 1). However, we are aware that we only use simple and crude methods to assess complexity. As this is not a research paper, but an illustration for an opinion paper, we do not want to focus on the methods too much. On the other hand, we think that it would be good to develop complexity measures for these type of data merging scripts as well as their components and data sources. For this reason we added a whole new paragraph on our methods to the discussion, as stated above. We agree that our measure of complexity is within one personal coding style only. This has the disadvantage that there is only one person or coding style, but the advantage that the differences between the complexities of components is not confused by different coding styles. In most cases, data merging efforts will be done by one person only. We do not want to generalise for all researchers as to which commands or packages they choose. But from our experience, our coding example is representative for data merging exercises. Independently from coding style, most effort goes into the first data cleaning and aggregation steps, including the effort to understand the different data sets. Whatever means we find to give a feedback on how much effort is needed to reuse this data, it is worthwhile giving it back to the data providers. In the following, we answer to specific comments: Paolo Missier: " Other assumptions along the way seem contrived and overfit the (single) example, for instance "output ports of a data source in the workflow directly relate to data columns in the data set". (pg 5,6) In the same section, questionable conclusions follow from this assumption ." We agree that this formulation is misleading. Since we use the EML actor of Kepler to import data, the “output ports” are always the data columns. We did not want to imply a causality here. Data columns appear as output ports in the Kepler actor, because this is how the EML actor works. We reformulate this sentence accordingly. Indeed, the paragraph works without even using the whole sentence (see page 3, section: quantify quality and usage of data) Before: “As explained above, output ports of a data source in the workflow directly relate to data columns in the data set. Thus, the number of available ports of a data source is the “width” of a dataset, or the number of data columns. Thus, the usage of a data column in rela- tion to the data source was calculated as the ratio of ports actually used in the workflow to the ports that were not used. This allowed us to relate the number of unused ports to the number of available ports of a data source." Now: “For our analysis we only used a subset of the data columns available in each data source. We therefore quantified the “data usage” of a data source as the ratio of data columns used for the analysis to the total number of data columns in that data source." Paolo Missier: "pg 3 - Complexity: The point is about programs with control structures, but scientific workflows traditionally are dataflows. So does the same notion of complexity apply here?" No, it doesn’t. We now provide a definition of complexity that clarifies that we are interested in the amount of effort and time that has to be invested in data or workflow reuse (see above). Paolo Missier: "pg 4: I feel there is probably too much detail on the science and its results here, which is not the focus of the paper and can be distracting (and uninteresting unless you know the specific science)." We have reordered and shortened the paragraphs on our workflow example. However some the information is interesting for the general reader and are required for the overall understanding. For example that the data sources come from independent projects and are archived in a common platform as well as some basics on workflows. But we have shortened the information on the scientific analysis to one paragraph. page 3, section: biodiversity effects on subtropical carbon stocks: “Our example workflow is part of an ongoing study that measures biodiversity effects on subtropical carbon stocks and flows. It is typical for synthesis tasks in collaborative research projects in that it combines eight datasets collected by seven independent research groups collaborating within the BEF-China research platform ( www.bef-china.de ). Data is archived, harmonized, and exchanged using the BEFdata web application (citation!!!). Data is exported in EML format and as such imported into the Kepler Workflow system. The data describes carbon pools from soil, litter, […] from the years 2008 and 2009 on the observational plots of the BEF-China research platform spanning a gradient from 22 to 116 years of plot age and 15 to 35 tree species. Our example workflow merges the data and terminates in a linear model relating biomass pools to plot age and plot diversity. It shows that carbon pools increase with stand age, however, in plots with high species richness this increase was less steep (p-values).” Paolo Missier: "pg 5 col 2: Need to explain AIC." AIC is explained and cited in the methods part (Akaikes Information criterion). page 3, right column, section: quantify component identity: “For this we used linear models which have been compared using the Akaike Information Criteria (AIC) to select for the most parsimonious model.” Paolo Missier: "pg 7: I found table 2 interesting and generally useful. In contrast, Table 3 is a bit of a mystery to me." Table 1, 3 and Figure 6 are different perspectives on the same topic. Figure 6 shows the ordination result using multidimensional scaling (NMDS) of workflow component characteristics. The NMDS results in 2 axes that span the highest variation of components in the characteristics space. Thus NMDS1, the first axis, spans the highest variation between components. The component identities in Table 1 as well as the component characteristics in Table 3 were later compared to the axes scores of the NMDS axes. These are the measures given in Table 2. We have changed the text in the captions to point out the relatedness of the figure and the tables. We additionally included and example in how to interpret measures in Table 2 when comparing them with Figure 6. Table 1: “Workflow component identities defined a priori and their relation to the data oriented motifs identified by (10). Figure 6 plots the a priori defined identities of the workflow components to the characteristics we measured from each component a posteriori. Characteristics include lines of codes or specific commands (Table 3). ” Table 2: “Characteristics of workflow components used to assess variation between components by means of non metric multidimensional scaling (NMDS). Characteristics include lines of code, use of packages, as well as specific commands (see text for further detail). Figure 6 plots the two axes of the NMDS. r2, Pr(>r) and sig. describe R square, Probability, and significance level of a correlation with the characteristic as dependent and the NMDS scores of both axes (NMDS1, NMDS2) as independent variables. For example, “count of code lines” separates workflow components in the NMDS plot, so that components with more code lines are plotted in the upper left quadrant of the plot in Figure 6. Signif. codes…” Figure 6: “Workflow components (points) in reduced component characteristics space (Table 3). We used non metric multidimensional scaling (NMDS, see text for further detail) to reduce the parameter space to two axes. Table 3 lists the regression results of the axes scores on the component characteristics, which are plotted in smaller text here. Table 2 lists the a priori tasks, which are plotted in large labels here. Points are jittered by a factor of 0.2 horizontally and vertically to handle over plotting.” Competing Interests: No competing interests were disclosed. Close Report a concern Respond or Comment COMMENTS ON THIS REPORT Author Response 03 Nov 2014 Claas-Thido Pfaff , University of Leipzig, Germany 03 Nov 2014 Author Response Dear Paolo Missier, First of all thank you for your valuable input which gave us the opportunity to sharpen the focus of our paper. Your main argument was that we cannot ... Continue reading Dear Paolo Missier, First of all thank you for your valuable input which gave us the opportunity to sharpen the focus of our paper. Your main argument was that we cannot prove our points because we are using a single case study. At the same time you said we should clarify whether our focus is on the results of the analysis - based on only one use case - or the metrics derived for illustrating complexity. However, our main focus is neither on the specific results of this use case, nor on the metrics. We are writing an opinion paper, and both, the use case and the metrics, are illustrations of our opinion. As you say in your comment, - and we take this as a compliment -, we want to “make a strong case for simplicity of data and workflow components”. Although it is not our intention to use the case study as proof, our paper is accompanied by many statistical analyses and plots. This may fool the reader in believing that we want to present a research article. However, we think that our plots are very useful for other data managers and scientists in illustrating why it is worthwhile to invest energy into simplifying datasets. This is especially the case for files from the long tail of big data, which are handcrafted, and relatively small data sets resulting from fieldwork and not from automated sensors. To be able to illustrate the problem of merging these files - which is our day to day work as hybrids of data managers and researchers - we chose this case study, as it is representative for our work and the work of our fellow data managers we spoke to. We also think that it is highly useful to illustrate our difficulties in data re-use. Reworking our text in response to your questions, we scaled down the method descriptions and put a stronger focus on the opinion parts of the paper. We completely reworked the text in many passages and provide here some examples: For example, in the abstract we changed the sentence (page 1): “We illustrate our points using a typical analysis in BEF research...” to: “To illustrate our points we chose a typical analysis in BEF research...”. At the beginning of the introduction we sharpened the opinion aspect of the paper, instead of the sentence (page 2): “However, there is a lack of papers that discuss workflow components within an analysis including data processing.” we now write : “Here we argue that there is a need for quality measures of workflow components, which include scripts, as well as for the underlying data sources. Failure to reuse workflows and available research data is not only a waste of time, money and effort but also represents a threat to the basic scientific principle of reproducibility. Providing feedback mechanisms on the data and workflow component complexity has the great potential to increase the readability and the reuse of workflows and its components.” and other small changes like: before: “We thus suggest that focusing on simplifying ...” after: “We argue that focusing on simplifying...” We further added a new paragraph to the discussion on our methods. We explain, that we want to illustrate the possibility to use simple text mining techniques in providing immediate feedback to data providers or workflow creators. We also add additional avenues that could be taken to quantify complexity of further workflow components or scriptlets (last paragraph discussion): “We here exemplify how to quantify the complexity as well as the quality and the usage of data in scientific workflows, using simple qualitative and quantitative measures. Our means are not meant to be exhaustive but rather could serve as a starting point for discussion towards the development of more sophisticated complexity feedback mechanisms for data providers and workflows creators. Our example workflow strongly relies on the interface component of Kepler connecting to the R statistical environment for the purpose of data manipulation and analysis. Thus the means we provide to measure complexity and quality are adapted to that specific workflow situation. However, adapting our means to further components that work as interfaces to other programming languages should be straightforward. Further complexity attributes could be the inclusion of the variable types of workflow components or a ratio capturing the enrichment or reduction of data consumed by the component. Providing complexity measures at the level of workflow components might help in adapting workflows towards a better readability and reusability and thus improve their value for reuse. Additionally it can guide the restructuring and simplification of data for a better use in workflows, a better understandability and reuse.” We further agree, that we have provided too little explanation of what we mean by “complexity”. We thus added a paragraph in the introduction, section complexity and identity, to define the aspect of complexity we are concerned about (section complexity and identity, first paragraph): “Here we are interested in workflows that begin with the cleaning, the aggregation, and the imputation of research data. These first steps can make up to 70% of the whole workflow. As data managers and researchers, we want to improve the readability of such workflows whether they are scripts or graphs. Our concept of complexity thus should capture the effort and time needed to understand and reuse such workflows. Regarding the complexity of source code we found similar incentives that provide quality measures. The Code Climate service for example provides code complexity feedback to programmers in many different programming languages (https://codeclimate.com/?v=b). Their complexity measures take the number of lines of code as well as the repetition of identical code lines into account.” Our operationalisation of data complexity is based on this approach to workflow complexity. We explain in the same section (paragraph 2 - 3): “Quantifying data complexity is not as straight forward as workflow component complexity. Datasets used for synthesis in research collaborations often consist of “dark” data, lacking sufficient meta- data for reuse…. … Here we argue that data complexity can be quantified by looking at the workflow components needed to aggregate and focus the data for analysis. One of the paradigms of data- driven science is that the analysis should be accompanied by it’s data. We argue that at the same time, data should be accompanied by workflows that offer meaningful aggregation of the data. Data complexity could then be measured by the complexity of their workflows." More technically speaking, you asked for a process independent complexity measure and criticised the use of line of codes, asking how we deal with hidden complexity when using whole script packages with only one line of code. However, the overwhelming majority of data merging efforts we see in our work as data managers are script based, and are not meant to be reused in the same way as software programs. For this reason, function points do not make sense for them. In addition, we do not only use lines of code in our complexity measure. We also include the number of packages used, for example. In our paper, in the part on the Example workflow, section Quantifying workflow complexity, we explain: To quantify the complexity of the components we used the number of code lines (loc), the number of R commands (cc) and R packages used (pc), as well as the number of input and output ports (cp) of the components (equation 1). However, we are aware that we only use simple and crude methods to assess complexity. As this is not a research paper, but an illustration for an opinion paper, we do not want to focus on the methods too much. On the other hand, we think that it would be good to develop complexity measures for these type of data merging scripts as well as their components and data sources. For this reason we added a whole new paragraph on our methods to the discussion, as stated above. We agree that our measure of complexity is within one personal coding style only. This has the disadvantage that there is only one person or coding style, but the advantage that the differences between the complexities of components is not confused by different coding styles. In most cases, data merging efforts will be done by one person only. We do not want to generalise for all researchers as to which commands or packages they choose. But from our experience, our coding example is representative for data merging exercises. Independently from coding style, most effort goes into the first data cleaning and aggregation steps, including the effort to understand the different data sets. Whatever means we find to give a feedback on how much effort is needed to reuse this data, it is worthwhile giving it back to the data providers. In the following, we answer to specific comments: Paolo Missier: " Other assumptions along the way seem contrived and overfit the (single) example, for instance "output ports of a data source in the workflow directly relate to data columns in the data set". (pg 5,6) In the same section, questionable conclusions follow from this assumption ." We agree that this formulation is misleading. Since we use the EML actor of Kepler to import data, the “output ports” are always the data columns. We did not want to imply a causality here. Data columns appear as output ports in the Kepler actor, because this is how the EML actor works. We reformulate this sentence accordingly. Indeed, the paragraph works without even using the whole sentence (see page 3, section: quantify quality and usage of data) Before: “As explained above, output ports of a data source in the workflow directly relate to data columns in the data set. Thus, the number of available ports of a data source is the “width” of a dataset, or the number of data columns. Thus, the usage of a data column in rela- tion to the data source was calculated as the ratio of ports actually used in the workflow to the ports that were not used. This allowed us to relate the number of unused ports to the number of available ports of a data source." Now: “For our analysis we only used a subset of the data columns available in each data source. We therefore quantified the “data usage” of a data source as the ratio of data columns used for the analysis to the total number of data columns in that data source." Paolo Missier: "pg 3 - Complexity: The point is about programs with control structures, but scientific workflows traditionally are dataflows. So does the same notion of complexity apply here?" No, it doesn’t. We now provide a definition of complexity that clarifies that we are interested in the amount of effort and time that has to be invested in data or workflow reuse (see above). Paolo Missier: "pg 4: I feel there is probably too much detail on the science and its results here, which is not the focus of the paper and can be distracting (and uninteresting unless you know the specific science)." We have reordered and shortened the paragraphs on our workflow example. However some the information is interesting for the general reader and are required for the overall understanding. For example that the data sources come from independent projects and are archived in a common platform as well as some basics on workflows. But we have shortened the information on the scientific analysis to one paragraph. page 3, section: biodiversity effects on subtropical carbon stocks: “Our example workflow is part of an ongoing study that measures biodiversity effects on subtropical carbon stocks and flows. It is typical for synthesis tasks in collaborative research projects in that it combines eight datasets collected by seven independent research groups collaborating within the BEF-China research platform ( www.bef-china.de ). Data is archived, harmonized, and exchanged using the BEFdata web application (citation!!!). Data is exported in EML format and as such imported into the Kepler Workflow system. The data describes carbon pools from soil, litter, […] from the years 2008 and 2009 on the observational plots of the BEF-China research platform spanning a gradient from 22 to 116 years of plot age and 15 to 35 tree species. Our example workflow merges the data and terminates in a linear model relating biomass pools to plot age and plot diversity. It shows that carbon pools increase with stand age, however, in plots with high species richness this increase was less steep (p-values).” Paolo Missier: "pg 5 col 2: Need to explain AIC." AIC is explained and cited in the methods part (Akaikes Information criterion). page 3, right column, section: quantify component identity: “For this we used linear models which have been compared using the Akaike Information Criteria (AIC) to select for the most parsimonious model.” Paolo Missier: "pg 7: I found table 2 interesting and generally useful. In contrast, Table 3 is a bit of a mystery to me." Table 1, 3 and Figure 6 are different perspectives on the same topic. Figure 6 shows the ordination result using multidimensional scaling (NMDS) of workflow component characteristics. The NMDS results in 2 axes that span the highest variation of components in the characteristics space. Thus NMDS1, the first axis, spans the highest variation between components. The component identities in Table 1 as well as the component characteristics in Table 3 were later compared to the axes scores of the NMDS axes. These are the measures given in Table 2. We have changed the text in the captions to point out the relatedness of the figure and the tables. We additionally included and example in how to interpret measures in Table 2 when comparing them with Figure 6. Table 1: “Workflow component identities defined a priori and their relation to the data oriented motifs identified by (10). Figure 6 plots the a priori defined identities of the workflow components to the characteristics we measured from each component a posteriori. Characteristics include lines of codes or specific commands (Table 3). ” Table 2: “Characteristics of workflow components used to assess variation between components by means of non metric multidimensional scaling (NMDS). Characteristics include lines of code, use of packages, as well as specific commands (see text for further detail). Figure 6 plots the two axes of the NMDS. r2, Pr(>r) and sig. describe R square, Probability, and significance level of a correlation with the characteristic as dependent and the NMDS scores of both axes (NMDS1, NMDS2) as independent variables. For example, “count of code lines” separates workflow components in the NMDS plot, so that components with more code lines are plotted in the upper left quadrant of the plot in Figure 6. Signif. codes…” Figure 6: “Workflow components (points) in reduced component characteristics space (Table 3). We used non metric multidimensional scaling (NMDS, see text for further detail) to reduce the parameter space to two axes. Table 3 lists the regression results of the axes scores on the component characteristics, which are plotted in smaller text here. Table 2 lists the a priori tasks, which are plotted in large labels here. Points are jittered by a factor of 0.2 horizontally and vertically to handle over plotting.” Dear Paolo Missier, First of all thank you for your valuable input which gave us the opportunity to sharpen the focus of our paper. Your main argument was that we cannot prove our points because we are using a single case study. At the same time you said we should clarify whether our focus is on the results of the analysis - based on only one use case - or the metrics derived for illustrating complexity. However, our main focus is neither on the specific results of this use case, nor on the metrics. We are writing an opinion paper, and both, the use case and the metrics, are illustrations of our opinion. As you say in your comment, - and we take this as a compliment -, we want to “make a strong case for simplicity of data and workflow components”. Although it is not our intention to use the case study as proof, our paper is accompanied by many statistical analyses and plots. This may fool the reader in believing that we want to present a research article. However, we think that our plots are very useful for other data managers and scientists in illustrating why it is worthwhile to invest energy into simplifying datasets. This is especially the case for files from the long tail of big data, which are handcrafted, and relatively small data sets resulting from fieldwork and not from automated sensors. To be able to illustrate the problem of merging these files - which is our day to day work as hybrids of data managers and researchers - we chose this case study, as it is representative for our work and the work of our fellow data managers we spoke to. We also think that it is highly useful to illustrate our difficulties in data re-use. Reworking our text in response to your questions, we scaled down the method descriptions and put a stronger focus on the opinion parts of the paper. We completely reworked the text in many passages and provide here some examples: For example, in the abstract we changed the sentence (page 1): “We illustrate our points using a typical analysis in BEF research...” to: “To illustrate our points we chose a typical analysis in BEF research...”. At the beginning of the introduction we sharpened the opinion aspect of the paper, instead of the sentence (page 2): “However, there is a lack of papers that discuss workflow components within an analysis including data processing.” we now write : “Here we argue that there is a need for quality measures of workflow components, which include scripts, as well as for the underlying data sources. Failure to reuse workflows and available research data is not only a waste of time, money and effort but also represents a threat to the basic scientific principle of reproducibility. Providing feedback mechanisms on the data and workflow component complexity has the great potential to increase the readability and the reuse of workflows and its components.” and other small changes like: before: “We thus suggest that focusing on simplifying ...” after: “We argue that focusing on simplifying...” We further added a new paragraph to the discussion on our methods. We explain, that we want to illustrate the possibility to use simple text mining techniques in providing immediate feedback to data providers or workflow creators. We also add additional avenues that could be taken to quantify complexity of further workflow components or scriptlets (last paragraph discussion): “We here exemplify how to quantify the complexity as well as the quality and the usage of data in scientific workflows, using simple qualitative and quantitative measures. Our means are not meant to be exhaustive but rather could serve as a starting point for discussion towards the development of more sophisticated complexity feedback mechanisms for data providers and workflows creators. Our example workflow strongly relies on the interface component of Kepler connecting to the R statistical environment for the purpose of data manipulation and analysis. Thus the means we provide to measure complexity and quality are adapted to that specific workflow situation. However, adapting our means to further components that work as interfaces to other programming languages should be straightforward. Further complexity attributes could be the inclusion of the variable types of workflow components or a ratio capturing the enrichment or reduction of data consumed by the component. Providing complexity measures at the level of workflow components might help in adapting workflows towards a better readability and reusability and thus improve their value for reuse. Additionally it can guide the restructuring and simplification of data for a better use in workflows, a better understandability and reuse.” We further agree, that we have provided too little explanation of what we mean by “complexity”. We thus added a paragraph in the introduction, section complexity and identity, to define the aspect of complexity we are concerned about (section complexity and identity, first paragraph): “Here we are interested in workflows that begin with the cleaning, the aggregation, and the imputation of research data. These first steps can make up to 70% of the whole workflow. As data managers and researchers, we want to improve the readability of such workflows whether they are scripts or graphs. Our concept of complexity thus should capture the effort and time needed to understand and reuse such workflows. Regarding the complexity of source code we found similar incentives that provide quality measures. The Code Climate service for example provides code complexity feedback to programmers in many different programming languages (https://codeclimate.com/?v=b). Their complexity measures take the number of lines of code as well as the repetition of identical code lines into account.” Our operationalisation of data complexity is based on this approach to workflow complexity. We explain in the same section (paragraph 2 - 3): “Quantifying data complexity is not as straight forward as workflow component complexity. Datasets used for synthesis in research collaborations often consist of “dark” data, lacking sufficient meta- data for reuse…. … Here we argue that data complexity can be quantified by looking at the workflow components needed to aggregate and focus the data for analysis. One of the paradigms of data- driven science is that the analysis should be accompanied by it’s data. We argue that at the same time, data should be accompanied by workflows that offer meaningful aggregation of the data. Data complexity could then be measured by the complexity of their workflows." More technically speaking, you asked for a process independent complexity measure and criticised the use of line of codes, asking how we deal with hidden complexity when using whole script packages with only one line of code. However, the overwhelming majority of data merging efforts we see in our work as data managers are script based, and are not meant to be reused in the same way as software programs. For this reason, function points do not make sense for them. In addition, we do not only use lines of code in our complexity measure. We also include the number of packages used, for example. In our paper, in the part on the Example workflow, section Quantifying workflow complexity, we explain: To quantify the complexity of the components we used the number of code lines (loc), the number of R commands (cc) and R packages used (pc), as well as the number of input and output ports (cp) of the components (equation 1). However, we are aware that we only use simple and crude methods to assess complexity. As this is not a research paper, but an illustration for an opinion paper, we do not want to focus on the methods too much. On the other hand, we think that it would be good to develop complexity measures for these type of data merging scripts as well as their components and data sources. For this reason we added a whole new paragraph on our methods to the discussion, as stated above. We agree that our measure of complexity is within one personal coding style only. This has the disadvantage that there is only one person or coding style, but the advantage that the differences between the complexities of components is not confused by different coding styles. In most cases, data merging efforts will be done by one person only. We do not want to generalise for all researchers as to which commands or packages they choose. But from our experience, our coding example is representative for data merging exercises. Independently from coding style, most effort goes into the first data cleaning and aggregation steps, including the effort to understand the different data sets. Whatever means we find to give a feedback on how much effort is needed to reuse this data, it is worthwhile giving it back to the data providers. In the following, we answer to specific comments: Paolo Missier: " Other assumptions along the way seem contrived and overfit the (single) example, for instance "output ports of a data source in the workflow directly relate to data columns in the data set". (pg 5,6) In the same section, questionable conclusions follow from this assumption ." We agree that this formulation is misleading. Since we use the EML actor of Kepler to import data, the “output ports” are always the data columns. We did not want to imply a causality here. Data columns appear as output ports in the Kepler actor, because this is how the EML actor works. We reformulate this sentence accordingly. Indeed, the paragraph works without even using the whole sentence (see page 3, section: quantify quality and usage of data) Before: “As explained above, output ports of a data source in the workflow directly relate to data columns in the data set. Thus, the number of available ports of a data source is the “width” of a dataset, or the number of data columns. Thus, the usage of a data column in rela- tion to the data source was calculated as the ratio of ports actually used in the workflow to the ports that were not used. This allowed us to relate the number of unused ports to the number of available ports of a data source." Now: “For our analysis we only used a subset of the data columns available in each data source. We therefore quantified the “data usage” of a data source as the ratio of data columns used for the analysis to the total number of data columns in that data source." Paolo Missier: "pg 3 - Complexity: The point is about programs with control structures, but scientific workflows traditionally are dataflows. So does the same notion of complexity apply here?" No, it doesn’t. We now provide a definition of complexity that clarifies that we are interested in the amount of effort and time that has to be invested in data or workflow reuse (see above). Paolo Missier: "pg 4: I feel there is probably too much detail on the science and its results here, which is not the focus of the paper and can be distracting (and uninteresting unless you know the specific science)." We have reordered and shortened the paragraphs on our workflow example. However some the information is interesting for the general reader and are required for the overall understanding. For example that the data sources come from independent projects and are archived in a common platform as well as some basics on workflows. But we have shortened the information on the scientific analysis to one paragraph. page 3, section: biodiversity effects on subtropical carbon stocks: “Our example workflow is part of an ongoing study that measures biodiversity effects on subtropical carbon stocks and flows. It is typical for synthesis tasks in collaborative research projects in that it combines eight datasets collected by seven independent research groups collaborating within the BEF-China research platform ( www.bef-china.de ). Data is archived, harmonized, and exchanged using the BEFdata web application (citation!!!). Data is exported in EML format and as such imported into the Kepler Workflow system. The data describes carbon pools from soil, litter, […] from the years 2008 and 2009 on the observational plots of the BEF-China research platform spanning a gradient from 22 to 116 years of plot age and 15 to 35 tree species. Our example workflow merges the data and terminates in a linear model relating biomass pools to plot age and plot diversity. It shows that carbon pools increase with stand age, however, in plots with high species richness this increase was less steep (p-values).” Paolo Missier: "pg 5 col 2: Need to explain AIC." AIC is explained and cited in the methods part (Akaikes Information criterion). page 3, right column, section: quantify component identity: “For this we used linear models which have been compared using the Akaike Information Criteria (AIC) to select for the most parsimonious model.” Paolo Missier: "pg 7: I found table 2 interesting and generally useful. In contrast, Table 3 is a bit of a mystery to me." Table 1, 3 and Figure 6 are different perspectives on the same topic. Figure 6 shows the ordination result using multidimensional scaling (NMDS) of workflow component characteristics. The NMDS results in 2 axes that span the highest variation of components in the characteristics space. Thus NMDS1, the first axis, spans the highest variation between components. The component identities in Table 1 as well as the component characteristics in Table 3 were later compared to the axes scores of the NMDS axes. These are the measures given in Table 2. We have changed the text in the captions to point out the relatedness of the figure and the tables. We additionally included and example in how to interpret measures in Table 2 when comparing them with Figure 6. Table 1: “Workflow component identities defined a priori and their relation to the data oriented motifs identified by (10). Figure 6 plots the a priori defined identities of the workflow components to the characteristics we measured from each component a posteriori. Characteristics include lines of codes or specific commands (Table 3). ” Table 2: “Characteristics of workflow components used to assess variation between components by means of non metric multidimensional scaling (NMDS). Characteristics include lines of code, use of packages, as well as specific commands (see text for further detail). Figure 6 plots the two axes of the NMDS. r2, Pr(>r) and sig. describe R square, Probability, and significance level of a correlation with the characteristic as dependent and the NMDS scores of both axes (NMDS1, NMDS2) as independent variables. For example, “count of code lines” separates workflow components in the NMDS plot, so that components with more code lines are plotted in the upper left quadrant of the plot in Figure 6. Signif. codes…” Figure 6: “Workflow components (points) in reduced component characteristics space (Table 3). We used non metric multidimensional scaling (NMDS, see text for further detail) to reduce the parameter space to two axes. Table 3 lists the regression results of the axes scores on the component characteristics, which are plotted in smaller text here. Table 2 lists the a priori tasks, which are plotted in large labels here. Points are jittered by a factor of 0.2 horizontally and vertically to handle over plotting.” Competing Interests: No competing interests were disclosed. Close Report a concern COMMENT ON THIS REPORT Comments on this article Comments (0) Version 2 VERSION 2 PUBLISHED 14 May 2014 ADD YOUR COMMENT Comment keyboard_arrow_left keyboard_arrow_right Open Peer Review Reviewer Status info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Reviewer Reports Invited Reviewers 1 2 3 4 Version 2 (revision) 17 Nov 14 read read read Version 1 14 May 14 read Paolo Missier , Newcastle University, Newcastle upon Tyne, UK Barry Demchak , University of California San Diego, La Jolla, CA, USA Kristina Hettne , Leiden University, Leiden, The Netherlands David Soergel , Google Inc., Mountain View, USA Comments on this article All Comments (0) Add a comment Sign up for content alerts Sign Up You are now signed up to receive this alert Browse by related subjects keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2015 Soergel D. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 19 Feb 2015 | for Version 2 David Soergel , Google Inc., Mountain View, CA, USA 0 Views copyright © 2015 Soergel D. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Not Approved info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Wrong arguments in support of a valid point It is appropriate that this is now an opinion piece rather than a research paper. Given that, it is essential to state clearly what the opinion is, and what practical consequences it has. As far as I can tell, the opinion is that real-world data are messy and require a lot of cleaning. That much is obvious and requires no demonstration. But what do the authors actually want people to do about it? Surely the authors are not suggesting that dataset providers should proactively prune their datasets, for instance by removing columns that they consider less interesting or that are rarely used (perhaps based on some feedback mechanism). I hope it goes without saying that such an opinion would be unscientific and dangerous, and should not be published. I trust the authors instead mean that dataset providers should carefully distinguish data arising from different experiments and experimental designs, by providing them in separate files with adequate metadata, so that the meaning and provenance of each measurement is clear. That is of course true and should be entirely obvious, but it does merit repeating. To the extent that scientists do in fact mix up their datasets as the authors describe, that is very bad. But the principal argument against it is not that it requires additional work on the part of downstream data consumers (as represented here by workflow complexity). The real argument is that mixed datasets, insufficient metadata, columns of mixed type, and other forms of poor bookkeeping lead to bad science and wrong results, regardless of the workflow system or analysis methods that are used. So, while I am very much in favor of exhorting scientists to collect and publish their data in ways that are clean, rigorous, simple, and reusable, I think the arguments presented here miss the point as to why those are important goals. Workflow metrics Measuring workflow complexity is an entirely separate issue, but interesting in its own right. Any proposal and validation of workflow complexity metrics should be a separate paper. However the measures proposed here lack justification, and neglect the existing large literature on software metrics. The authors say that "complexity" should reflect how long it takes a user to understand a component, but then propose an entirely arbitrary measure (Eq. 1) with no reference to this "readability" criterion. Why should a line of code and an R package import count the same? The authors are motivated by the laudable goal of making workflow components more reusable, and assert that component complexity inhibits reuse. But is complexity really the main barrier, or even any barrier at all, to component reuse? What about basic interoperability (i.e., having input and output ports of matching data type)? What about sufficient metadata and documentation of the components? What about search and discovery of available components? etc. etc. Consider the analogous process of importing some library package in R or any other language. Such libraries are often extremely complex, but are designed and marketed with reuse in mind, and many enjoy widespread adoption. Workflow component classification Use of the terms "Identities" and "Identical tasks" is extremely misleading. The authors classified the components into very general classes such as "data extraction". This does not make the components identical! Table 1, Column 1 header should read "Classes" (and similarly throughout the text). Did the authors manually label the 71 components with their respective classes? If so, that is not clear from the text (manual labeling is not what "a priori" means, and that's the only related bit I can find). The "text mining" methods are poorly described. Do the authors mean the NMDS applied to term presence/absence vectors? In any case, this is orthogonal to the rest of the paper and entirely unnecessary. Perhaps the authors hope that their method is generalizable, so that it can be run on large numbers of workflows; indeed they state "we further show that specific workflow tasks can be identified using text mining." But in fact they don't show that. In order to make such a claim, they would need to provide some validation that the automated classification is meaningful, typically by comparing the class predictions with manually labelled data. Is that what figure 6 is meant to demonstrate? If so, 1) that is unclear from the text and caption, 2) the authors do not anywhere actually predict component classes based on feature vectors, and 3) the classes are not at all separated in the figure, so it seems unlikely that any associated class predictions would be correct. Inappropriate and overused statistical computations Overall, this paper tries to blind the reader with gratuitous use of statistical methods that are not justified and have no clear purpose. If the authors can clearly explain how the use of each statistical method supports their argument, they should do so. If not, the methods should not appear. This applies at least to every mention of a statistical test (Kruskal-Wallis, Wilcoxon, etc.), every mention of AIC, and to the NMDS analysis. Competing Interests No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to state that I do not consider it to be of an acceptable scientific standard, for reasons outlined above. reply Respond to this report Responses (0) Soergel D. Peer Review Report For: Readable workflows need simple data [version 1; peer review: 1 approved with reservations] . F1000Research 2014, 3 :110 ( https://doi.org/10.5256/f1000research.6072.r7388) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/3-110/v2#referee-response-7388 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2015 Hettne K. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 16 Feb 2015 | for Version 2 Kristina Hettne , Department of Human Genetics, Leiden University, Leiden, The Netherlands 0 Views copyright © 2015 Hettne K. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions The title and the abstract correctly summarize the article. The authors investigate an increasingly important subject, namely the role of well-designed scientific workflows in the reproducibility of science. The first step is indeed to create a workflow since that essentially should enable experimental reproducibility, but if the workflow itself is too complex it seems reasonable that that aim can indeed be missed. The tone in the abstract is right for an opinion piece. Generally, the introduction is adequate for this type of paper, but would indeed be stronger if more relevant references would be included (I second reviewer Barry Demchak here). For example, I was surprised that a sentence dealing with semantic enrichment of workflows did not mention SADI ( Wilkinson, Vandervalk and McCarthy, 2011 ). I appreciate the section “Complexity and Identity” where the authors outline what they mean by these terms in the context of scientific workflows. I however agree with reviewer Barry Demchak that it is limited in its current version and would benefit from mentioning other types of complexity. If his section would be followed by a section about the analysis strategy, including a motivation for the statistical analyses performed, the rest of the article would hopefully read more easily. The result section would benefit from use of subheadings. The point that I believe that the authors are trying to make about how to feedback complexity information to data providers is somewhat lost. They come back to this in the Discussion section, but are not being very specific in which type of information this feedback would contain. Could it simply be a recommendation about the number and type of columns in the dataset or would it also contain the background information leading to this recommendation? Also, it would be interesting to know which text mining tools the authors used to characterize the components in the workflow. Inclusion of this information would increase the reproducibility of their analysis. Competing Interests No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (0) Hettne K. Peer Review Report For: Readable workflows need simple data [version 1; peer review: 1 approved with reservations] . F1000Research 2014, 3 :110 ( https://doi.org/10.5256/f1000research.6072.r7389) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/3-110/v2#referee-response-7389 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2015 Demchak B. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 03 Feb 2015 | for Version 2 Barry Demchak , Department of Medicine, University of California San Diego, La Jolla, CA, USA 0 Views copyright © 2015 Demchak B. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions The title is appropriate, and entices a reader interested in creating "high quality" workflows. The abstract reasonably describes the problem and the general approach, and prepares me to understand evaluation methodologies I'll likely find useful across a large number of scientific projects. Somewhat disorienting, though: the characterization of giving "visual access to the flow of data" conflates a common presentation of workflows (i.e., visually) with the real nature of workflows, which is the thrust of the paper. The overall topic is of great interest, and has been dealt with (somewhat inconclusively) in the computer science literature exhaustively. Citing some of that work would lend a good foundation for the discussion. The unique value of this paper is the use of metrics and statistical analysis to make points about complexity of computation and data. While I support this, the paper would be much stronger if it could justify and give stronger foundation to the choices and formulation of metrics -- intuitively, to me, they seem to conflate the complexity of a workflow component with the overall complexity of the workflow. (The resolution lies in composition/decomposition, encapsulation, and reusability arguments from computer science or mathematics.) Also, data complexity metrics seem to focus on identifying extraneous data, which is trivially filtered -- attention to other types of complexity (e.g., data that cross-references other data) would be useful. I would also like to know whether there are other dimensions to data complexity. Given a stronger foundation for metrics, the statistical analysis approach is conceptually valuable. To drive the points, closer attention to the statistics being used would be helpful, but only with a much larger sample space. Additionally, I would like to see a section on the analysis strategy, as the use of some of the statistical techniques (e.g., AIC) seems unintuitive. Housekeeping: Figures 2 and 3 seem to have extraneous numbers (e.g., "1" or "2") or garbled text (by "4" in Figure 2). The text claims 12 tasks in Table 1, but there are 11 tasks. As a scientific paper, it could easy and usefully be twice as long if it addressed the points above. As an opinion piece, justification of the metrics and placing them on a sound theoretical foundation would be valuable, and would enable reducing the statistical analysis. Within the context above, I second all of reviewer Missier's comments, and give appreciation to both the authors and Missier for the progress so far. The great potential value of this work would be its effective targeting of the biology community, which is often best served by concrete proposals for best practices, accompanied by specific examples. Competing Interests No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (0) Demchak B. Peer Review Report For: Readable workflows need simple data [version 1; peer review: 1 approved with reservations] . F1000Research 2014, 3 :110 ( https://doi.org/10.5256/f1000research.6072.r7387) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/3-110/v2#referee-response-7387 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2014 Missier P. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 04 Jun 2014 | for Version 1 Paolo Missier , School of Computing Science, Newcastle University, Newcastle upon Tyne, UK 0 Views copyright © 2014 Missier P. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (1) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions The title and abstract do indeed summarise the purpose and content of the paper adequately. While the goals of the work are laudable, the framework proposed to go about them is not convincing. I see two main problems, firstly with using a single case study to drive the definition and analysis of data and process complexity, and thus to derive results can hardly have general validity. Secondly, by basing the analysis on some questionable assumptions. In what follows, I try to elaborate on these points. The paper makes a strong case for simplicity of data and components, however that is based on a single sample. This is hardly justified, and at odds with the wealth of quantitative research methods machinery deployed to analyse workflows and data and derive the metrics proposed in the paper. It would be good to clarify whether the paper's focus is on the method -- whereby the case study is just an illustrative example, and without any pretense of drawing general conclusions, or on the actual results, which given the very limited evaluation, are questionable. Regarding assumptions, it is stated "data complexity could be measured by the complexity of their workflow". How general is this meant to be? I am not sure a process-independent notion of data complexity is given in the paper, but I believe it should be, to clarify the argument. Here complexity seems to be based on how many different usages (and reuse) the data supports, which is fine perhaps, but only one of many possible criteria. I am also suspicious of process complexity criteria based on lines of code, especially in workflows that are composed of discrete components, often pre-existing and part of libraries. Kepler is idiosyncratic in this, as it assumes most actors are ad hoc programs. More generally, workflow is about coarse-grained composition (eg of third party services), and local coding decisions matter a lot less than in hand-crafted code. LOC is a very crude measure of complexity. Just as old, but perhaps more appropriate, is the notion of "function points" whereby you express complexity in terms of functionality realised by a component -- regardless of how much code is required to implement a certain function. LOC alone is also at odds with the idea that languages like R sit on powerful packages, which make for succinct but expressive code. How do you compare R code that implements a whole algorithm in R with one that simply invokes a lib function to achieve the same result? One could also argue, reading on pg 7 (col 2), that you may be measuring personal coding style rather that actual process complexity. Other assumptions along the way seem contrived and overfit the (single) example, for instance "output ports of a data source in the workflow directly relate to data columns in the data set". (pg 5,6) In the same section, questionable conclusions follow from this assumption. So overall, I think the quantitative methods used in the paper are interesting, but they are applied to a framework where a number of initial assumptions are questionable, and seem to be driven by one single example. A few specific comments: pg 3 - Complexity: The point is about programs with control structures, but scientific workflows traditionally are dataflows. So does the same notion of complexity apply here? pg 4: I feel there is probably too much detail on the science and its results here, which is not the focus of the paper and can be distracting (and uninteresting unless you know the specific science). pg 5 col 2: Need to explain AIC. pg 7: I found table 2 interesting and generally useful. In contrast, Table 3 is a bit of a mystery to me. Competing Interests No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (1) Author Response 03 Nov 2014 Claas-Thido Pfaff, University of Leipzig, Germany Dear Paolo Missier, First of all thank you for your valuable input which gave us the opportunity to sharpen the focus of our paper. Your main argument was that we cannot prove our points because we are using a single case study. At the same time you said we should clarify whether our focus is on the results of the analysis - based on only one use case - or the metrics derived for illustrating complexity. However, our main focus is neither on the specific results of this use case, nor on the metrics. We are writing an opinion paper, and both, the use case and the metrics, are illustrations of our opinion. As you say in your comment, - and we take this as a compliment -, we want to “make a strong case for simplicity of data and workflow components”. Although it is not our intention to use the case study as proof, our paper is accompanied by many statistical analyses and plots. This may fool the reader in believing that we want to present a research article. However, we think that our plots are very useful for other data managers and scientists in illustrating why it is worthwhile to invest energy into simplifying datasets. This is especially the case for files from the long tail of big data, which are handcrafted, and relatively small data sets resulting from fieldwork and not from automated sensors. To be able to illustrate the problem of merging these files - which is our day to day work as hybrids of data managers and researchers - we chose this case study, as it is representative for our work and the work of our fellow data managers we spoke to. We also think that it is highly useful to illustrate our difficulties in data re-use. Reworking our text in response to your questions, we scaled down the method descriptions and put a stronger focus on the opinion parts of the paper. We completely reworked the text in many passages and provide here some examples: For example, in the abstract we changed the sentence (page 1): “We illustrate our points using a typical analysis in BEF research...” to: “To illustrate our points we chose a typical analysis in BEF research...”. At the beginning of the introduction we sharpened the opinion aspect of the paper, instead of the sentence (page 2): “However, there is a lack of papers that discuss workflow components within an analysis including data processing.” we now write : “Here we argue that there is a need for quality measures of workflow components, which include scripts, as well as for the underlying data sources. Failure to reuse workflows and available research data is not only a waste of time, money and effort but also represents a threat to the basic scientific principle of reproducibility. Providing feedback mechanisms on the data and workflow component complexity has the great potential to increase the readability and the reuse of workflows and its components.” and other small changes like: before: “We thus suggest that focusing on simplifying ...” after: “We argue that focusing on simplifying...” We further added a new paragraph to the discussion on our methods. We explain, that we want to illustrate the possibility to use simple text mining techniques in providing immediate feedback to data providers or workflow creators. We also add additional avenues that could be taken to quantify complexity of further workflow components or scriptlets (last paragraph discussion): “We here exemplify how to quantify the complexity as well as the quality and the usage of data in scientific workflows, using simple qualitative and quantitative measures. Our means are not meant to be exhaustive but rather could serve as a starting point for discussion towards the development of more sophisticated complexity feedback mechanisms for data providers and workflows creators. Our example workflow strongly relies on the interface component of Kepler connecting to the R statistical environment for the purpose of data manipulation and analysis. Thus the means we provide to measure complexity and quality are adapted to that specific workflow situation. However, adapting our means to further components that work as interfaces to other programming languages should be straightforward. Further complexity attributes could be the inclusion of the variable types of workflow components or a ratio capturing the enrichment or reduction of data consumed by the component. Providing complexity measures at the level of workflow components might help in adapting workflows towards a better readability and reusability and thus improve their value for reuse. Additionally it can guide the restructuring and simplification of data for a better use in workflows, a better understandability and reuse.” We further agree, that we have provided too little explanation of what we mean by “complexity”. We thus added a paragraph in the introduction, section complexity and identity, to define the aspect of complexity we are concerned about (section complexity and identity, first paragraph): “Here we are interested in workflows that begin with the cleaning, the aggregation, and the imputation of research data. These first steps can make up to 70% of the whole workflow. As data managers and researchers, we want to improve the readability of such workflows whether they are scripts or graphs. Our concept of complexity thus should capture the effort and time needed to understand and reuse such workflows. Regarding the complexity of source code we found similar incentives that provide quality measures. The Code Climate service for example provides code complexity feedback to programmers in many different programming languages (https://codeclimate.com/?v=b). Their complexity measures take the number of lines of code as well as the repetition of identical code lines into account.” Our operationalisation of data complexity is based on this approach to workflow complexity. We explain in the same section (paragraph 2 - 3): “Quantifying data complexity is not as straight forward as workflow component complexity. Datasets used for synthesis in research collaborations often consist of “dark” data, lacking sufficient meta- data for reuse…. … Here we argue that data complexity can be quantified by looking at the workflow components needed to aggregate and focus the data for analysis. One of the paradigms of data- driven science is that the analysis should be accompanied by it’s data. We argue that at the same time, data should be accompanied by workflows that offer meaningful aggregation of the data. Data complexity could then be measured by the complexity of their workflows." More technically speaking, you asked for a process independent complexity measure and criticised the use of line of codes, asking how we deal with hidden complexity when using whole script packages with only one line of code. However, the overwhelming majority of data merging efforts we see in our work as data managers are script based, and are not meant to be reused in the same way as software programs. For this reason, function points do not make sense for them. In addition, we do not only use lines of code in our complexity measure. We also include the number of packages used, for example. In our paper, in the part on the Example workflow, section Quantifying workflow complexity, we explain: To quantify the complexity of the components we used the number of code lines (loc), the number of R commands (cc) and R packages used (pc), as well as the number of input and output ports (cp) of the components (equation 1). However, we are aware that we only use simple and crude methods to assess complexity. As this is not a research paper, but an illustration for an opinion paper, we do not want to focus on the methods too much. On the other hand, we think that it would be good to develop complexity measures for these type of data merging scripts as well as their components and data sources. For this reason we added a whole new paragraph on our methods to the discussion, as stated above. We agree that our measure of complexity is within one personal coding style only. This has the disadvantage that there is only one person or coding style, but the advantage that the differences between the complexities of components is not confused by different coding styles. In most cases, data merging efforts will be done by one person only. We do not want to generalise for all researchers as to which commands or packages they choose. But from our experience, our coding example is representative for data merging exercises. Independently from coding style, most effort goes into the first data cleaning and aggregation steps, including the effort to understand the different data sets. Whatever means we find to give a feedback on how much effort is needed to reuse this data, it is worthwhile giving it back to the data providers. In the following, we answer to specific comments: Paolo Missier: " Other assumptions along the way seem contrived and overfit the (single) example, for instance "output ports of a data source in the workflow directly relate to data columns in the data set". (pg 5,6) In the same section, questionable conclusions follow from this assumption ." We agree that this formulation is misleading. Since we use the EML actor of Kepler to import data, the “output ports” are always the data columns. We did not want to imply a causality here. Data columns appear as output ports in the Kepler actor, because this is how the EML actor works. We reformulate this sentence accordingly. Indeed, the paragraph works without even using the whole sentence (see page 3, section: quantify quality and usage of data) Before: “As explained above, output ports of a data source in the workflow directly relate to data columns in the data set. Thus, the number of available ports of a data source is the “width” of a dataset, or the number of data columns. Thus, the usage of a data column in rela- tion to the data source was calculated as the ratio of ports actually used in the workflow to the ports that were not used. This allowed us to relate the number of unused ports to the number of available ports of a data source." Now: “For our analysis we only used a subset of the data columns available in each data source. We therefore quantified the “data usage” of a data source as the ratio of data columns used for the analysis to the total number of data columns in that data source." Paolo Missier: "pg 3 - Complexity: The point is about programs with control structures, but scientific workflows traditionally are dataflows. So does the same notion of complexity apply here?" No, it doesn’t. We now provide a definition of complexity that clarifies that we are interested in the amount of effort and time that has to be invested in data or workflow reuse (see above). Paolo Missier: "pg 4: I feel there is probably too much detail on the science and its results here, which is not the focus of the paper and can be distracting (and uninteresting unless you know the specific science)." We have reordered and shortened the paragraphs on our workflow example. However some the information is interesting for the general reader and are required for the overall understanding. For example that the data sources come from independent projects and are archived in a common platform as well as some basics on workflows. But we have shortened the information on the scientific analysis to one paragraph. page 3, section: biodiversity effects on subtropical carbon stocks: “Our example workflow is part of an ongoing study that measures biodiversity effects on subtropical carbon stocks and flows. It is typical for synthesis tasks in collaborative research projects in that it combines eight datasets collected by seven independent research groups collaborating within the BEF-China research platform ( www.bef-china.de ). Data is archived, harmonized, and exchanged using the BEFdata web application (citation!!!). Data is exported in EML format and as such imported into the Kepler Workflow system. The data describes carbon pools from soil, litter, […] from the years 2008 and 2009 on the observational plots of the BEF-China research platform spanning a gradient from 22 to 116 years of plot age and 15 to 35 tree species. Our example workflow merges the data and terminates in a linear model relating biomass pools to plot age and plot diversity. It shows that carbon pools increase with stand age, however, in plots with high species richness this increase was less steep (p-values).” Paolo Missier: "pg 5 col 2: Need to explain AIC." AIC is explained and cited in the methods part (Akaikes Information criterion). page 3, right column, section: quantify component identity: “For this we used linear models which have been compared using the Akaike Information Criteria (AIC) to select for the most parsimonious model.” Paolo Missier: "pg 7: I found table 2 interesting and generally useful. In contrast, Table 3 is a bit of a mystery to me." Table 1, 3 and Figure 6 are different perspectives on the same topic. Figure 6 shows the ordination result using multidimensional scaling (NMDS) of workflow component characteristics. The NMDS results in 2 axes that span the highest variation of components in the characteristics space. Thus NMDS1, the first axis, spans the highest variation between components. The component identities in Table 1 as well as the component characteristics in Table 3 were later compared to the axes scores of the NMDS axes. These are the measures given in Table 2. We have changed the text in the captions to point out the relatedness of the figure and the tables. We additionally included and example in how to interpret measures in Table 2 when comparing them with Figure 6. Table 1: “Workflow component identities defined a priori and their relation to the data oriented motifs identified by (10). Figure 6 plots the a priori defined identities of the workflow components to the characteristics we measured from each component a posteriori. Characteristics include lines of codes or specific commands (Table 3). ” Table 2: “Characteristics of workflow components used to assess variation between components by means of non metric multidimensional scaling (NMDS). Characteristics include lines of code, use of packages, as well as specific commands (see text for further detail). Figure 6 plots the two axes of the NMDS. r2, Pr(>r) and sig. describe R square, Probability, and significance level of a correlation with the characteristic as dependent and the NMDS scores of both axes (NMDS1, NMDS2) as independent variables. For example, “count of code lines” separates workflow components in the NMDS plot, so that components with more code lines are plotted in the upper left quadrant of the plot in Figure 6. Signif. codes…” Figure 6: “Workflow components (points) in reduced component characteristics space (Table 3). We used non metric multidimensional scaling (NMDS, see text for further detail) to reduce the parameter space to two axes. Table 3 lists the regression results of the axes scores on the component characteristics, which are plotted in smaller text here. Table 2 lists the a priori tasks, which are plotted in large labels here. Points are jittered by a factor of 0.2 horizontally and vertically to handle over plotting.” View more View less Competing Interests No competing interests were disclosed. reply Respond Report a concern Missier P. Peer Review Report For: Readable workflows need simple data [version 1; peer review: 1 approved with reservations] . F1000Research 2014, 3 :110 ( https://doi.org/10.5256/f1000research.4221.r4788) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/3-110/v1#referee-response-4788 Alongside their report, reviewers assign a status to the article: Approved - the paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations - A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved - fundamental flaws in the paper seriously undermine the findings and conclusions Adjust parameters to alter display View on desktop for interactive features Includes Interactive Elements View on desktop for interactive features Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Stay Updated Sign up for content alerts and receive a weekly or monthly email with all newly published articles Register with F1000Research Already registered? Sign in Not now, thanks close PLEASE NOTE If you are an AUTHOR of this article, please check that you signed in with the account associated with this article otherwise we cannot automatically identify your role as an author and your comment will be labelled as a “User Comment”. If you are a REVIEWER of this article, please check that you have signed in with the account associated with this article and then go to your account to submit your report, please do not post your review here. If you do not have access to your original account, please contact us . All commenters must hold a formal affiliation as per our Policies . The information that you give us will be displayed next to your comment. User comments must be in English, comprehensible and relevant to the article under discussion. We reserve the right to remove any comments that we consider to be inappropriate, offensive or otherwise in breach of the User Comment Terms and Conditions . Commenters must not use a comment for personal attacks. When criticisms of the article are based on unpublished data, the data should be made available. I accept the User Comment Terms and Conditions Please confirm that you accept the User Comment Terms and Conditions. Affiliation ✕ refresh Please enter your institution. Note: To add your institution or organisation, start typing the name and then select the correct name from the list. Where applicable, the name will appear in both the original language and in English. Do not paste in the name. If the name does not appear in the drop-down list, we will display the information you have entered. ✕ refresh Country/Region * USA UK Canada China France Germany Afghanistan Aland Islands Albania Algeria American Samoa Andorra Angola Anguilla Antarctica Antigua and Barbuda Argentina Armenia Aruba Australia Austria Azerbaijan Bahamas Bahrain Bangladesh Barbados Belarus Belgium Belize Benin Bermuda Bhutan Bolivia Bosnia and Herzegovina Botswana Bouvet Island Brazil British Indian Ocean Territory British Virgin Islands Brunei Bulgaria Burkina Faso Burundi Cambodia Cameroon Canada Cape Verde Cayman Islands Central African Republic Chad Chile China Christmas Island Cocos (Keeling) Islands Colombia Comoros Congo Cook Islands Costa Rica Cote d'Ivoire Croatia Cuba Cyprus Czech Republic Democratic Republic of the Congo Denmark Djibouti Dominica Dominican Republic Ecuador Egypt El Salvador Equatorial Guinea Eritrea Estonia Ethiopia Falkland Islands Faroe Islands Federated States of Micronesia Fiji Finland France French Guiana French Polynesia French Southern Territories Gabon Georgia Germany Ghana Gibraltar Greece Greenland Grenada Guadeloupe Guam Guatemala Guernsey Guinea Guinea-Bissau Guyana Haiti Heard Island and Mcdonald Islands Holy See (Vatican City State) Honduras Hong Kong Hungary Iceland India Indonesia Iran Iraq Ireland Israel Italy Jamaica Japan Jersey Jordan Kazakhstan Kenya Kiribati Kosovo (Serbia and Montenegro) Kuwait Kyrgyzstan Lao People's Democratic Republic Latvia Lebanon Lesotho Liberia Libya Liechtenstein Lithuania Luxembourg Macao Madagascar Malawi Malaysia Maldives Mali Malta Marshall Islands Martinique Mauritania Mauritius Mayotte Mexico Minor Outlying Islands of the United States Moldova Monaco Mongolia Montenegro Montserrat Morocco Mozambique Myanmar Namibia Nauru Nepal Netherlands Antilles New Caledonia New Zealand Nicaragua Niger Nigeria Niue Norfolk Island North Korea North Macedonia Northern Mariana Islands Norway Oman Pakistan Palau Palestinian Territory Panama Papua New Guinea Paraguay Peru Philippines Pitcairn Poland Portugal Puerto Rico Qatar Reunion Romania Russian Federation Rwanda Saint Helena Saint Kitts and Nevis Saint Lucia Saint Pierre and Miquelon Saint Vincent and the Grenadines Samoa San Marino Sao Tome and Principe Saudi Arabia Senegal Serbia Seychelles Sierra Leone Singapore Slovakia Slovenia Solomon Islands Somalia South Africa South Georgia and the South Sandwich Is South Korea South Sudan Spain Sri Lanka Sudan Suriname Svalbard and Jan Mayen Swaziland Sweden Switzerland Syria Taiwan Tajikistan Tanzania Thailand The Gambia The Netherlands Timor-Leste Togo Tokelau Tonga Trinidad and Tobago Tunisia Turkey Turkmenistan Turks and Caicos Islands Tuvalu UK USA Uganda Ukraine United Arab Emirates United States Virgin Islands Uruguay Uzbekistan Vanuatu Venezuela Vietnam Wallis and Futuna West Bank and Gaza Strip Western Sahara Yemen Zambia Zimbabwe Please select your country/region. You must enter a comment. Competing Interests Please disclose any competing interests that might be construed to influence your judgment of the article's or peer review report's validity or importance. Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Please state your competing interests The comment has been saved. An error has occurred. Please try again. Cancel Post var lTitle = "Readable workflows need simple data".replace("'", ''); var linkedInUrl = "http://www.linkedin.com/shareArticle?url=https://f1000research.com/articles/3-110/v1" + "&title=" + encodeURIComponent(lTitle) + "&summary=" + encodeURIComponent('Read the article by '); var deliciousUrl = "https://del.icio.us/post?url=https://f1000research.com/articles/3-110/v1&title=" + encodeURIComponent(lTitle); var redditUrl = "http://reddit.com/submit?url=https://f1000research.com/articles/3-110/v1" + "&title=" + encodeURIComponent(lTitle); linkedInUrl += encodeURIComponent('Pfaff CT et al.'); var offsetTop = /chrome/i.test( navigator.userAgent ) ? 4 : -10; var addthis_config = { ui_offset_top: offsetTop, services_compact : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_expanded : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_custom : [ { name: "LinkedIn", url: linkedInUrl, icon:"/img/icon/at_linkedin.svg" }, { name: "Mendeley", url: "http://www.mendeley.com/import/?url=https://f1000research.com/articles/3-110/v1/mendeley", icon:"/img/icon/at_mendeley.svg" }, { name: "Reddit", url: redditUrl, icon:"/img/icon/at_reddit.svg" }, ] }; var addthis_share = { url: "https://f1000research.com/articles/3-110", templates : { twitter : "Readable workflows need simple data. Pfaff CT et al., published by " + "@F1000Research" + ", https://f1000research.com/articles/3-110/v1" } }; if (typeof(addthis) != "undefined"){ addthis.addEventListener('addthis.ready', checkCount); addthis.addEventListener('addthis.menu.share', checkCount); } $(".f1r-shares-twitter").attr("href", "https://twitter.com/intent/tweet?text=" + addthis_share.templates.twitter); $(".f1r-shares-facebook").attr("href", "https://www.facebook.com/sharer/sharer.php?u=" + addthis_share.url); $(".f1r-shares-linkedin").attr("href", addthis_config.services_custom[0].url); $(".f1r-shares-reddit").attr("href", addthis_config.services_custom[2].url); $(".f1r-shares-mendelay").attr("href", addthis_config.services_custom[1].url); function checkCount(){ setTimeout(function(){ $(".addthis_button_expanded").each(function(){ var count = $(this).text(); if (count !== "" && count != "0") $(this).removeClass("is-hidden"); else $(this).addClass("is-hidden"); }); }, 1000); } close How to cite this report {{reportCitation}} Cancel Copy Citation Details $(function(){R.ui.buttonDropdowns('.dropdown-for-downloads');}); $(function(){R.ui.toolbarDropdowns('.toolbar-dropdown-for-downloads');}); $.get("/articles/acj/3940/4221") new F1000.Clipboard(); new F1000.ThesaurusTermsDisplay("articles", "article", "4221"); $(document).ready(function() { $( "#frame1" ).on('load', function() { var mydiv = $(this).contents().find("div"); var h = mydiv.height(); console.log(h) }); var tooltipLivingFigure = jQuery(".interactive-living-figure-label .icon-more-info"), titleLivingFigure = tooltipLivingFigure.attr("title"); tooltipLivingFigure.simpletip({ fixed: true, position: ["-115", "30"], baseClass: 'small-tooltip', content:titleLivingFigure + " " }); tooltipLivingFigure.removeAttr("title"); $("body").on("click", ".cite-living-figure", function(e) { e.preventDefault(); var ref = $(this).attr("data-ref"); $(this).closest(".living-figure-list-container").find("#" + ref).fadeIn(200); }); $("body").on("click", ".close-cite-living-figure", function(e) { e.preventDefault(); $(this).closest(".popup-window-wrapper").fadeOut(200); }); $(document).on("mouseup", function(e) { var metricsContainer = $(".article-metrics-popover-wrapper"); if (!metricsContainer.is(e.target) && metricsContainer.has(e.target).length === 0) { $(".article-metrics-close-button").click(); } }); var articleId = $('#articleId').val(); if($("#main-article-count-box").attachArticleMetrics) { $("#main-article-count-box").attachArticleMetrics(articleId, { articleMetricsView: true }); } }); var figshareWidget = $(".new_figshare_widget"); if (figshareWidget.length > 0) { window.figshare.load("f1000", function(Widget) { // Select a tag/tags defined in your page. In this tag we will place the widget. _.map(figshareWidget, function(el){ var widget = new Widget({ articleId: $(el).attr("figshare_articleId") //height:300 // this is the height of the viewer part. [Default: 550] }); widget.initialize(); // initialize the widget widget.mount(el); // mount it in a tag that's on your page // this will save the widget on the global scope for later use from // your JS scripts. This line is optional. //window.widget = widget; }); }); } close Error Close Add Reset F1000.MICROSERVICES.AFFILIATION = ''; $(document).ready(function () { $('.js-affiliations-form').each((index, form) => { new AffiliationForm({ formId: form.id, institutionErrorSelector: '.comment-enter-institution', departmentErrorSelector: '.comment-enter-department', placeSelector: '.js-add-comment-place', stateSelector: '.js-add-comment-state', zipCodeSelector: '.js-add-comment-zipcode', countrySelector: '.js-add-comment-country', countryErrorSelector: '.comment-enter-country', }); }); }); $(document).ready(function () { var reportIds = { "5216": 0, "5217": 0, "6725": 0, "6726": 0, "6727": 0, "4786": 0, "4787": 0, "4788": 45, "4789": 0, "4790": 0, "7387": 20, "7388": 37, "7389": 12, "5215": 0, }; $(".referee-response-container,.js-referee-report").each(function(index, el) { var reportId = $(el).attr("data-reportid"), reportCount = reportIds[reportId] || 0; $(el).find(".comments-count-container,.js-referee-report-views").html(reportCount); }); var uuidInput = $("#article_uuid"), oldUUId = uuidInput.val(), newUUId = "2e8831bb-7182-4101-a6d5-9fa7f28840fa"; uuidInput.val(newUUId); $("a[href*='article_uuid=']").each(function(index, el) { var newHref = $(el).attr("href").replace(oldUUId, newUUId); $(el).attr("href", newHref); }); }); An innovative open access publishing platform offering rapid publication and open peer review, whilst supporting data deposition and sharing. Browse Gateways Collections How it Works Contact For Developers Cookie Notice Privacy Notice RSS Submit Your Research Follow us © 2012-2026 F1000 Research Ltd. ISSN 2046-1402 | Legal | Partner of Research4Life • CrossRef • ORCID • FAIRSharing R.templateTests.simpleTemplate = R.template(' $text $text $text $text $text '); R.templateTests.runTests(); var F1000platform = new F1000.Platform({ name: "f1000research", displayName: "F1000Research", hostName: "f1000research.com", id: "1", editorialEmail: "
[email protected]", infoEmail: "
[email protected]", usePmcStats: true }); $(function(){R.ui.dropdowns('.dropdown-for-authors, .dropdown-for-about, .dropdown-for-myresearch');}); // $(function(){R.ui.dropdowns('.dropdown-for-referees');}); $(document).ready(function () { if ($(".cookie-warning").is(":visible")) { $(".sticky").css("margin-bottom", "35px"); $(".devices").addClass("devices-and-cookie-warning"); } $(".cookie-warning .close-button").click(function (e) { $(".devices").removeClass("devices-and-cookie-warning"); $(".sticky").css("margin-bottom", "0"); }); $("#tweeter-feed .tweet-message").each(function (i, message) { var self = $(message); self.html(linkify(self.html())); }); $(".partner").on("mouseenter mouseleave", function() { $(this).find(".gray-scale, .colour").toggleClass("is-hidden"); }); }); Sign In Remember me Forgotten your password? Sign In Cancel Email or password not correct. Please try again Please wait... $(function(){ // Note: All the setup needs to run against a name attribute and *not* the id due the clonish // nature of facebox... $("a[id=googleSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("GOOGLE"); $("form[id=oAuthForm]").submit(); }); $("a[id=facebookSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("FACEBOOK"); $("form[id=oAuthForm]").submit(); }); $("a[id=orcidSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("ORCID"); $("form[id=oAuthForm]").submit(); }); }); If you've forgotten your password, please enter your email address below and we'll send you instructions on how to reset your password. The email address should be the one you originally registered with F1000. Email address not valid, please try again You registered with F1000 via Google, so we cannot reset your password. To sign in, please click here . If you still need help with your Google account password, please click here . You registered with F1000 via Facebook, so we cannot reset your password. To sign in, please click here . If you still need help with your Facebook account password, please click here . Code not correct, please try again Reset password Cancel Email us for further assistance. Server error, please try again. If your email address is registered with us, we will email you instructions to reset your password. If you think you should have received this email but it has not arrived, please check your spam filters and/or contact for further assistance. Please wait... Register $(document).ready(function () { signIn.createSignInAsRow($("#sign-in-form-gfb-popup")); $(".target-field").each(function () { var uris = $(this).val().split("/"); if (uris.pop() === "login") { $(this).val(uris.toString().replace(",","/")); } }); });
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.