Lessons learned: overcoming common challenges in... | F1000Research "use strict";function _typeof(t){return(_typeof="function"==typeof Symbol&&"symbol"==typeof Symbol.iterator?function(t){return typeof t}:function(t){return t&&"function"==typeof Symbol&&t.constructor===Symbol&&t!==Symbol.prototype?"symbol":typeof t})(t)}!function(){var t=function(){var t,e,o=[],n=window,r=n;for(;r;){try{if(r.frames.__tcfapiLocator){t=r;break}}catch(t){}if(r===n.top)break;r=r.parent}t||(!function t(){var e=n.document,o=!!n.frames.__tcfapiLocator;if(!o)if(e.body){var r=e.createElement("iframe");r.style.cssText="display:none",r.name="__tcfapiLocator",e.body.appendChild(r)}else setTimeout(t,5);return!o}(),n.__tcfapi=function(){for(var t=arguments.length,n=new Array(t),r=0;r 3&&2===parseInt(n[1],10)&&"boolean"==typeof n[3]&&(e=n[3],"function"==typeof n[2]&&n[2]("set",!0)):"ping"===n[0]?"function"==typeof n[2]&&n[2]({gdprApplies:e,cmpLoaded:!1,cmpStatus:"stub"}):o.push(n)},n.addEventListener("message",(function(t){var e="string"==typeof t.data,o={};if(e)try{o=JSON.parse(t.data)}catch(t){}else o=t.data;var n="object"===_typeof(o)&&null!==o?o.__tcfapiCall:null;n&&window.__tcfapi(n.command,n.version,(function(o,r){var a={__tcfapiReturn:{returnValue:o,success:r,callId:n.callId}};t&&t.source&&t.source.postMessage&&t.source.postMessage(e?JSON.stringify(a):a,"*")}),n.parameter)}),!1))};"undefined"!=typeof module?module.exports=t:t()}(); dataLayer = dataLayer || []; // Standard GTM initialization - Google Consent Mode handles consent automatically (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start': new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0], j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src= 'https://www.googletagmanager.com/gtm.js?id='+i+dl+ '>m_auth=hzk0Vc3qFsQYhCrIoHz68A>m_preview=env-1>m_cookies_win=x';f.parentNode.insertBefore(j,f); })(window,document,'script','dataLayer','GTM-MWFK8L5J'); ;window.NREUM||(NREUM={});NREUM.init={distributed_tracing:{enabled:true},privacy:{cookies_enabled:true},ajax:{deny_list:["bam.nr-data.net"]}}; ;NREUM.loader_config={accountID:"438030",trustKey:"438030",agentID:"772317073",licenseKey:"97f8f67f26",applicationID:"772317073"} ;NREUM.info={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net",licenseKey:"97f8f67f26",applicationID:"772317073",sa:1} ;/*! For license information please see nr-loader-spa-1.236.0.min.js.LICENSE.txt */ (()=>{"use strict";var e,t,r={5763:(e,t,r)=>{r.d(t,{P_:()=>l,Mt:()=>g,C5:()=>s,DL:()=>v,OP:()=>T,lF:()=>D,Yu:()=>y,Dg:()=>h,CX:()=>c,GE:()=>b,sU:()=>_});var n=r(8632),i=r(9567);const o={beacon:n.ce.beacon,errorBeacon:n.ce.errorBeacon,licenseKey:void 0,applicationID:void 0,sa:void 0,queueTime:void 0,applicationTime:void 0,ttGuid:void 0,user:void 0,account:void 0,product:void 0,extra:void 0,jsAttributes:{},userAttributes:void 0,atts:void 0,transactionName:void 0,tNamePlain:void 0},a={};function s(e){if(!e)throw new Error("All info objects require an agent identifier!");if(!a[e])throw new Error("Info for ".concat(e," was never set"));return a[e]}function c(e,t){if(!e)throw new Error("All info objects require an agent identifier!");a[e]=(0,i.D)(t,o),(0,n.Qy)(e,a[e],"info")}var u=r(7056);const d=()=>{const e={blockSelector:"[data-nr-block]",maskInputOptions:{password:!0}};return{allow_bfcache:!0,privacy:{cookies_enabled:!0},ajax:{deny_list:void 0,enabled:!0,harvestTimeSeconds:10},distributed_tracing:{enabled:void 0,exclude_newrelic_header:void 0,cors_use_newrelic_header:void 0,cors_use_tracecontext_headers:void 0,allowed_origins:void 0},session:{domain:void 0,expiresMs:u.oD,inactiveMs:u.Hb},ssl:void 0,obfuscate:void 0,jserrors:{enabled:!0,harvestTimeSeconds:10},metrics:{enabled:!0},page_action:{enabled:!0,harvestTimeSeconds:30},page_view_event:{enabled:!0},page_view_timing:{enabled:!0,harvestTimeSeconds:30,long_task:!1},session_trace:{enabled:!0,harvestTimeSeconds:10},harvest:{tooManyRequestsDelay:60},session_replay:{enabled:!1,harvestTimeSeconds:60,sampleRate:.1,errorSampleRate:.1,maskTextSelector:"*",maskAllInputs:!0,get blockClass(){return"nr-block"},get ignoreClass(){return"nr-ignore"},get maskTextClass(){return"nr-mask"},get blockSelector(){return e.blockSelector},set blockSelector(t){e.blockSelector+=",".concat(t)},get maskInputOptions(){return e.maskInputOptions},set maskInputOptions(t){e.maskInputOptions={...t,password:!0}}},spa:{enabled:!0,harvestTimeSeconds:10}}},f={};function l(e){if(!e)throw new Error("All configuration objects require an agent identifier!");if(!f[e])throw new Error("Configuration for ".concat(e," was never set"));return f[e]}function h(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");f[e]=(0,i.D)(t,d()),(0,n.Qy)(e,f[e],"config")}function g(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");var r=l(e);if(r){for(var n=t.split("."),i=0;i {r.d(t,{D:()=>i});var n=r(50);function i(e,t){try{if(!e||"object"!=typeof e)return(0,n.Z)("Setting a Configurable requires an object as input");if(!t||"object"!=typeof t)return(0,n.Z)("Setting a Configurable requires a model to set its initial properties");const r=Object.create(Object.getPrototypeOf(t),Object.getOwnPropertyDescriptors(t)),o=0===Object.keys(r).length?e:r;for(let a in o)if(void 0!==e[a])try{"object"==typeof e[a]&&"object"==typeof t[a]?r[a]=i(e[a],t[a]):r[a]=e[a]}catch(e){(0,n.Z)("An error occurred while setting a property of a Configurable",e)}return r}catch(e){(0,n.Z)("An error occured while setting a Configurable",e)}}},6818:(e,t,r)=>{r.d(t,{Re:()=>i,gF:()=>o,q4:()=>n});const n="1.236.0",i="PROD",o="CDN"},385:(e,t,r)=>{r.d(t,{FN:()=>a,IF:()=>u,Nk:()=>f,Tt:()=>s,_A:()=>o,il:()=>n,pL:()=>c,v6:()=>i,w1:()=>d});const n="undefined"!=typeof window&&!!window.document,i="undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self.navigator instanceof WorkerNavigator||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis.navigator instanceof WorkerNavigator),o=n?window:"undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis),a=""+o?.location,s=/iPad|iPhone|iPod/.test(navigator.userAgent),c=s&&"undefined"==typeof SharedWorker,u=(()=>{const e=navigator.userAgent.match(/Firefox[/\s](\d+\.\d+)/);return Array.isArray(e)&&e.length>=2?+e[1]:0})(),d=Boolean(n&&window.document.documentMode),f=!!navigator.sendBeacon},1117:(e,t,r)=>{r.d(t,{w:()=>o});var n=r(50);const i={agentIdentifier:"",ee:void 0};class o{constructor(e){try{if("object"!=typeof e)return(0,n.Z)("shared context requires an object as input");this.sharedContext={},Object.assign(this.sharedContext,i),Object.entries(e).forEach((e=>{let[t,r]=e;Object.keys(i).includes(t)&&(this.sharedContext[t]=r)}))}catch(e){(0,n.Z)("An error occured while setting SharedContext",e)}}}},8e3:(e,t,r)=>{r.d(t,{L:()=>d,R:()=>c});var n=r(2177),i=r(1284),o=r(4322),a=r(3325);const s={};function c(e,t){const r={staged:!1,priority:a.p[t]||0};u(e),s[e].get(t)||s[e].set(t,r)}function u(e){e&&(s[e]||(s[e]=new Map))}function d(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:"",t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:"feature";if(u(e),!e||!s[e].get(t))return a(t);s[e].get(t).staged=!0;const r=[...s[e]];function a(t){const r=e?n.ee.get(e):n.ee,a=o.X.handlers;if(r.backlog&&a){var s=r.backlog[t],c=a[t];if(c){for(var u=0;s&&u {let[t,r]=e;return r.staged}))&&(r.sort(((e,t)=>e[1].priority-t[1].priority)),r.forEach((e=>{let[t]=e;a(t)})))}function f(e,t){var r=e[1];(0,i.D)(t[r],(function(t,r){var n=e[0];if(r[0]===n){var i=r[1],o=e[3],a=e[2];i.apply(o,a)}}))}},2177:(e,t,r)=>{r.d(t,{c:()=>f,ee:()=>u});var n=r(8632),i=r(2210),o=r(1284),a=r(5763),s="nr@context";let c=(0,n.fP)();var u;function d(){}function f(e){return(0,i.X)(e,s,l)}function l(){return new d}function h(){u.aborted=!0,u.backlog={}}c.ee?u=c.ee:(u=function e(t,r){var n={},c={},f={},g=!1;try{g=16===r.length&&(0,a.OP)(r).isolatedBacklog}catch(e){}var p={on:b,addEventListener:b,removeEventListener:y,emit:v,get:x,listeners:w,context:m,buffer:A,abort:h,aborted:!1,isBuffering:E,debugId:r,backlog:g?{}:t&&"object"==typeof t.backlog?t.backlog:{}};return p;function m(e){return e&&e instanceof d?e:e?(0,i.X)(e,s,l):l()}function v(e,r,n,i,o){if(!1!==o&&(o=!0),!u.aborted||i){t&&o&&t.emit(e,r,n);for(var a=m(n),s=w(e),d=s.length,f=0;fn,p:()=>i});var n=r(2177).ee.get("handle");function i(e,t,r,i,o){o?(o.buffer([e],i),o.emit(e,t,r)):(n.buffer([e],i),n.emit(e,t,r))}},4322:(e,t,r)=>{r.d(t,{X:()=>o});var n=r(5546);o.on=a;var i=o.handlers={};function o(e,t,r,o){a(o||n.E,i,e,t,r)}function a(e,t,r,i,o){o||(o="feature"),e||(e=n.E);var a=t[o]=t[o]||{};(a[r]=a[r]||[]).push([e,i])}},3239:(e,t,r)=>{r.d(t,{bP:()=>s,iz:()=>c,m$:()=>a});var n=r(385);let i=!1,o=!1;try{const e={get passive(){return i=!0,!1},get signal(){return o=!0,!1}};n._A.addEventListener("test",null,e),n._A.removeEventListener("test",null,e)}catch(e){}function a(e,t){return i||o?{capture:!!e,passive:i,signal:t}:!!e}function s(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;window.addEventListener(e,t,a(r,n))}function c(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;document.addEventListener(e,t,a(r,n))}},4402:(e,t,r)=>{r.d(t,{Ht:()=>u,M:()=>c,Rl:()=>a,ky:()=>s});var n=r(385);const i="xxxxxxxx-xxxx-4xxx-yxxx-xxxxxxxxxxxx";function o(e,t){return e?15&e[t]:16*Math.random()|0}function a(){const e=n._A?.crypto||n._A?.msCrypto;let t,r=0;return e&&e.getRandomValues&&(t=e.getRandomValues(new Uint8Array(31))),i.split("").map((e=>"x"===e?o(t,++r).toString(16):"y"===e?(3&o()|8).toString(16):e)).join("")}function s(e){const t=n._A?.crypto||n._A?.msCrypto;let r,i=0;t&&t.getRandomValues&&(r=t.getRandomValues(new Uint8Array(31)));const a=[];for(var s=0;s {r.d(t,{Bq:()=>n,Hb:()=>o,oD:()=>i});const n="NRBA",i=144e5,o=18e5},7894:(e,t,r)=>{function n(){return Math.round(performance.now())}r.d(t,{z:()=>n})},7243:(e,t,r)=>{r.d(t,{e:()=>o});var n=r(385),i={};function o(e){if(e in i)return i[e];if(0===(e||"").indexOf("data:"))return{protocol:"data"};let t;var r=n._A?.location,o={};if(n.il)t=document.createElement("a"),t.href=e;else try{t=new URL(e,r.href)}catch(e){return o}o.port=t.port;var a=t.href.split("://");!o.port&&a[1]&&(o.port=a[1].split("/")[0].split("@").pop().split(":")[1]),o.port&&"0"!==o.port||(o.port="https"===a[0]?"443":"80"),o.hostname=t.hostname||r.hostname,o.pathname=t.pathname,o.protocol=a[0],"/"!==o.pathname.charAt(0)&&(o.pathname="/"+o.pathname);var s=!t.protocol||":"===t.protocol||t.protocol===r.protocol,c=t.hostname===r.hostname&&t.port===r.port;return o.sameOrigin=s&&(!t.hostname||c),"/"===o.pathname&&(i[e]=o),o}},50:(e,t,r)=>{function n(e,t){"function"==typeof console.warn&&(console.warn("New Relic: ".concat(e)),t&&console.warn(t))}r.d(t,{Z:()=>n})},2587:(e,t,r)=>{r.d(t,{N:()=>c,T:()=>u});var n=r(2177),i=r(5546),o=r(8e3),a=r(3325);const s={stn:[a.D.sessionTrace],err:[a.D.jserrors,a.D.metrics],ins:[a.D.pageAction],spa:[a.D.spa],sr:[a.D.sessionReplay,a.D.sessionTrace]};function c(e,t){const r=n.ee.get(t);e&&"object"==typeof e&&(Object.entries(e).forEach((e=>{let[t,n]=e;void 0===u[t]&&(s[t]?s[t].forEach((e=>{n?(0,i.p)("feat-"+t,[],void 0,e,r):(0,i.p)("block-"+t,[],void 0,e,r),(0,i.p)("rumresp-"+t,[Boolean(n)],void 0,e,r)})):n&&(0,i.p)("feat-"+t,[],void 0,void 0,r),u[t]=Boolean(n))})),Object.keys(s).forEach((e=>{void 0===u[e]&&(s[e]?.forEach((t=>(0,i.p)("rumresp-"+e,[!1],void 0,t,r))),u[e]=!1)})),(0,o.L)(t,a.D.pageViewEvent))}const u={}},2210:(e,t,r)=>{r.d(t,{X:()=>i});var n=Object.prototype.hasOwnProperty;function i(e,t,r){if(n.call(e,t))return e[t];var i=r();if(Object.defineProperty&&Object.keys)try{return Object.defineProperty(e,t,{value:i,writable:!0,enumerable:!1}),i}catch(e){}return e[t]=i,i}},1284:(e,t,r)=>{r.d(t,{D:()=>n});const n=(e,t)=>Object.entries(e||{}).map((e=>{let[r,n]=e;return t(r,n)}))},4351:(e,t,r)=>{r.d(t,{P:()=>o});var n=r(2177);const i=()=>{const e=new WeakSet;return(t,r)=>{if("object"==typeof r&&null!==r){if(e.has(r))return;e.add(r)}return r}};function o(e){try{return JSON.stringify(e,i())}catch(e){try{n.ee.emit("internal-error",[e])}catch(e){}}}},3960:(e,t,r)=>{r.d(t,{K:()=>a,b:()=>o});var n=r(3239);function i(){return"undefined"==typeof document||"complete"===document.readyState}function o(e,t){if(i())return e();(0,n.bP)("load",e,t)}function a(e){if(i())return e();(0,n.iz)("DOMContentLoaded",e)}},8632:(e,t,r)=>{r.d(t,{EZ:()=>u,Qy:()=>c,ce:()=>o,fP:()=>a,gG:()=>d,mF:()=>s});var n=r(7894),i=r(385);const o={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net"};function a(){return i._A.NREUM||(i._A.NREUM={}),void 0===i._A.newrelic&&(i._A.newrelic=i._A.NREUM),i._A.NREUM}function s(){let e=a();return e.o||(e.o={ST:i._A.setTimeout,SI:i._A.setImmediate,CT:i._A.clearTimeout,XHR:i._A.XMLHttpRequest,REQ:i._A.Request,EV:i._A.Event,PR:i._A.Promise,MO:i._A.MutationObserver,FETCH:i._A.fetch}),e}function c(e,t,r){let i=a();const o=i.initializedAgents||{},s=o[e]||{};return Object.keys(s).length||(s.initializedAt={ms:(0,n.z)(),date:new Date}),i.initializedAgents={...o,[e]:{...s,[r]:t}},i}function u(e,t){a()[e]=t}function d(){return function(){let e=a();const t=e.info||{};e.info={beacon:o.beacon,errorBeacon:o.errorBeacon,...t}}(),function(){let e=a();const t=e.init||{};e.init={...t}}(),s(),function(){let e=a();const t=e.loader_config||{};e.loader_config={...t}}(),a()}},7956:(e,t,r)=>{r.d(t,{N:()=>i});var n=r(3239);function i(e){let t=arguments.length>1&&void 0!==arguments[1]&&arguments[1],r=arguments.length>2?arguments[2]:void 0,i=arguments.length>3?arguments[3]:void 0;return void(0,n.iz)("visibilitychange",(function(){if(t)return void("hidden"==document.visibilityState&&e());e(document.visibilityState)}),r,i)}},1214:(e,t,r)=>{r.d(t,{em:()=>v,u5:()=>N,QU:()=>S,_L:()=>I,Gm:()=>L,Lg:()=>M,gy:()=>U,BV:()=>Q,Kf:()=>ee});var n=r(2177);const i="nr@original";var o=Object.prototype.hasOwnProperty,a=!1;function s(e,t){return e||(e=n.ee),r.inPlace=function(e,t,n,i,o){n||(n="");var a,s,c,u="-"===n.charAt(0);for(c=0;c 2?n-2:0),o=2;o {r(A[T],e,w),r(E[T],e,w)})),r(l._A,"fetch",y),t.on(y+"end",(function(e,r){var n=this;if(r){var i=r.headers.get("content-length");null!==i&&(n.rxSize=i),t.emit(y+"done",[null,r],n)}else t.emit(y+"done",[e],n)})),t}const O={},j=["pushState","replaceState"];function S(e){const t=function(e){return(e||n.ee).get("history")}(e);return!l.il||O[t.debugId]++||(O[t.debugId]=1,s(t).inPlace(window.history,j,"-")),t}var P=r(3239);const C={},R=["appendChild","insertBefore","replaceChild"];function I(e){const t=function(e){return(e||n.ee).get("jsonp")}(e);if(!l.il||C[t.debugId])return t;C[t.debugId]=!0;var r=s(t),i=/[?&](?:callback|cb)=([^&#]+)/,o=/(.*)\.([^.]+)/,a=/^(\w+)(\.|$)(.*)$/;function c(e,t){var r=e.match(a),n=r[1],i=r[3];return i?c(i,t[n]):t[n]}return r.inPlace(Node.prototype,R,"dom-"),t.on("dom-start",(function(e){!function(e){if(!e||"string"!=typeof e.nodeName||"script"!==e.nodeName.toLowerCase())return;if("function"!=typeof e.addEventListener)return;var n=(a=e.src,s=a.match(i),s?s[1]:null);var a,s;if(!n)return;var u=function(e){var t=e.match(o);if(t&&t.length>=3)return{key:t[2],parent:c(t[1],window)};return{key:e,parent:window}}(n);if("function"!=typeof u.parent[u.key])return;var d={};function f(){t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}function l(){t.emit("jsonp-error",[],d),t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}r.inPlace(u.parent,[u.key],"cb-",d),e.addEventListener("load",f,(0,P.m$)(!1)),e.addEventListener("error",l,(0,P.m$)(!1)),t.emit("new-jsonp",[e.src],d)}(e[0])})),t}var k=r(5763);const H={};function L(e){const t=function(e){return(e||n.ee).get("mutation")}(e);if(!l.il||H[t.debugId])return t;H[t.debugId]=!0;var r=s(t),i=k.Yu.MO;return i&&(window.MutationObserver=function(e){return this instanceof i?new i(r(e,"fn-")):i.apply(this,arguments)},MutationObserver.prototype=i.prototype),t}const z={};function M(e){const t=function(e){return(e||n.ee).get("promise")}(e);if(z[t.debugId])return t;z[t.debugId]=!0;var r=n.c,o=s(t),a=k.Yu.PR;return a&&function(){function e(r){var n=t.context(),i=o(r,"executor-",n,null,!1);const s=Reflect.construct(a,[i],e);return t.context(s).getCtx=function(){return n},s}l._A.Promise=e,Object.defineProperty(e,"name",{value:"Promise"}),e.toString=function(){return a.toString()},Object.setPrototypeOf(e,a),["all","race"].forEach((function(r){const n=a[r];e[r]=function(e){let i=!1;[...e||[]].forEach((e=>{this.resolve(e).then(a("all"===r),a(!1))}));const o=n.apply(this,arguments);return o;function a(e){return function(){t.emit("propagate",[null,!i],o,!1,!1),i=i||!e}}}})),["resolve","reject"].forEach((function(r){const n=a[r];e[r]=function(e){const r=n.apply(this,arguments);return e!==r&&t.emit("propagate",[e,!0],r,!1,!1),r}})),e.prototype=a.prototype;const n=a.prototype.then;a.prototype.then=function(){var e=this,i=r(e);i.promise=e;for(var a=arguments.length,s=new Array(a),c=0;c e())),t};function m(e,t){i.inPlace(t,["onreadystatechange"],"fn-",E)}function b(){var e=this,t=r.context(e);e.readyState>3&&!t.resolved&&(t.resolved=!0,r.emit("xhr-resolved",[],e)),i.inPlace(e,f,"fn-",E)}if(function(e,t){for(var r in e)t[r]=e[r]}(o,p),p.prototype=o.prototype,i.inPlace(p.prototype,J,"-xhr-",E),r.on("send-xhr-start",(function(e,t){m(e,t),function(e){h.push(e),a&&(y?y.then(A):u?u(A):(w=-w,x.data=w))}(t)})),r.on("open-xhr-start",m),a){var y=c&&c.resolve();if(!u&&!c){var w=1,x=document.createTextNode(w);new a(A).observe(x,{characterData:!0})}}else t.on("fn-end",(function(e){e[0]&&e[0].type===d||A()}));function A(){for(var e=0;e {r.d(t,{t:()=>n});const n=r(3325).D.ajax},6660:(e,t,r)=>{r.d(t,{A:()=>i,t:()=>n});const n=r(3325).D.jserrors,i="nr@seenError"},3081:(e,t,r)=>{r.d(t,{gF:()=>o,mY:()=>i,t9:()=>n,vz:()=>s,xS:()=>a});const n=r(3325).D.metrics,i="sm",o="cm",a="storeSupportabilityMetrics",s="storeEventMetrics"},4649:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageAction},7633:(e,t,r)=>{r.d(t,{Dz:()=>i,OJ:()=>a,qw:()=>o,t9:()=>n});const n=r(3325).D.pageViewEvent,i="firstbyte",o="domcontent",a="windowload"},9251:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageViewTiming},3614:(e,t,r)=>{r.d(t,{BST_RESOURCE:()=>i,END:()=>s,FEATURE_NAME:()=>n,FN_END:()=>u,FN_START:()=>c,PUSH_STATE:()=>d,RESOURCE:()=>o,START:()=>a});const n=r(3325).D.sessionTrace,i="bstResource",o="resource",a="-start",s="-end",c="fn"+a,u="fn"+s,d="pushState"},7836:(e,t,r)=>{r.d(t,{BODY:()=>A,CB_END:()=>E,CB_START:()=>u,END:()=>x,FEATURE_NAME:()=>i,FETCH:()=>_,FETCH_BODY:()=>v,FETCH_DONE:()=>m,FETCH_START:()=>p,FN_END:()=>c,FN_START:()=>s,INTERACTION:()=>l,INTERACTION_API:()=>d,INTERACTION_EVENTS:()=>o,JSONP_END:()=>b,JSONP_NODE:()=>g,JS_TIME:()=>T,MAX_TIMER_BUDGET:()=>a,REMAINING:()=>f,SPA_NODE:()=>h,START:()=>w,originalSetTimeout:()=>y});var n=r(5763);const i=r(3325).D.spa,o=["click","submit","keypress","keydown","keyup","change"],a=999,s="fn-start",c="fn-end",u="cb-start",d="api-ixn-",f="remaining",l="interaction",h="spaNode",g="jsonpNode",p="fetch-start",m="fetch-done",v="fetch-body-",b="jsonp-end",y=n.Yu.ST,w="-start",x="-end",A="-body",E="cb"+x,T="jsTime",_="fetch"},5938:(e,t,r)=>{r.d(t,{W:()=>o});var n=r(5763),i=r(2177);class o{constructor(e,t,r){this.agentIdentifier=e,this.aggregator=t,this.ee=i.ee.get(e,(0,n.OP)(this.agentIdentifier).isolatedBacklog),this.featureName=r,this.blocked=!1}}},9144:(e,t,r)=>{r.d(t,{j:()=>m});var n=r(3325),i=r(5763),o=r(5546),a=r(2177),s=r(7894),c=r(8e3),u=r(3960),d=r(385),f=r(50),l=r(3081),h=r(8632);function g(){const e=(0,h.gG)();["setErrorHandler","finished","addToTrace","inlineHit","addRelease","addPageAction","setCurrentRouteName","setPageViewName","setCustomAttribute","interaction","noticeError","setUserId"].forEach((t=>{e[t]=function(){for(var r=arguments.length,n=new Array(r),i=0;i 1?r-1:0),i=1;i {e.exposed&&e.api[t]&&o.push(e.api[t](...n))})),o.length>1?o:o[0]}(t,...n)}}))}var p=r(2587);function m(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:{},m=arguments.length>2?arguments[2]:void 0,v=arguments.length>3?arguments[3]:void 0,{init:b,info:y,loader_config:w,runtime:x={loaderType:m},exposed:A=!0}=t;const E=(0,h.gG)();y||(b=E.init,y=E.info,w=E.loader_config),(0,i.Dg)(e,b||{}),(0,i.GE)(e,w||{}),(0,i.sU)(e,x),y.jsAttributes??={},d.v6&&(y.jsAttributes.isWorker=!0),(0,i.CX)(e,y),g();const T=function(e,t){t||(0,c.R)(e,"api");const h={};var g=a.ee.get(e),p=g.get("tracer"),m="api-",v=m+"ixn-";function b(t,r,n,o){const a=(0,i.C5)(e);return null===r?delete a.jsAttributes[t]:(0,i.CX)(e,{...a,jsAttributes:{...a.jsAttributes,[t]:r}}),x(m,n,!0,o||null===r?"session":void 0)(t,r)}function y(){}["setErrorHandler","finished","addToTrace","inlineHit","addRelease"].forEach((e=>h[e]=x(m,e,!0,"api"))),h.addPageAction=x(m,"addPageAction",!0,n.D.pageAction),h.setCurrentRouteName=x(m,"routeName",!0,n.D.spa),h.setPageViewName=function(t,r){if("string"==typeof t)return"/"!==t.charAt(0)&&(t="/"+t),(0,i.OP)(e).customTransaction=(r||"http://custom.transaction")+t,x(m,"setPageViewName",!0)()},h.setCustomAttribute=function(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2];if("string"==typeof e){if(["string","number"].includes(typeof t)||null===t)return b(e,t,"setCustomAttribute",r);(0,f.Z)("Failed to execute setCustomAttribute.\nNon-null value must be a string or number type, but a type of was provided."))}else(0,f.Z)("Failed to execute setCustomAttribute.\nName must be a string type, but a type of was provided."))},h.setUserId=function(e){if("string"==typeof e||null===e)return b("enduser.id",e,"setUserId",!0);(0,f.Z)("Failed to execute setUserId.\nNon-null value must be a string type, but a type of was provided."))},h.interaction=function(){return(new y).get()};var w=y.prototype={createTracer:function(e,t){var r={},i=this,a="function"==typeof t;return(0,o.p)(v+"tracer",[(0,s.z)(),e,r],i,n.D.spa,g),function(){if(p.emit((a?"":"no-")+"fn-start",[(0,s.z)(),i,a],r),a)try{return t.apply(this,arguments)}catch(e){throw p.emit("fn-err",[arguments,this,"string"==typeof e?new Error(e):e],r),e}finally{p.emit("fn-end",[(0,s.z)()],r)}}}};function x(e,t,r,i){return function(){return(0,o.p)(l.xS,["API/"+t+"/called"],void 0,n.D.metrics,g),i&&(0,o.p)(e+t,[(0,s.z)(),...arguments],r?null:this,i,g),r?void 0:this}}function A(){r.e(439).then(r.bind(r,7438)).then((t=>{let{setAPI:r}=t;r(e),(0,c.L)(e,"api")})).catch((()=>(0,f.Z)("Downloading runtime APIs failed...")))}return["actionText","setName","setAttribute","save","ignore","onEnd","getContext","end","get"].forEach((e=>{w[e]=x(v,e,void 0,n.D.spa)})),h.noticeError=function(e,t){"string"==typeof e&&(e=new Error(e)),(0,o.p)(l.xS,["API/noticeError/called"],void 0,n.D.metrics,g),(0,o.p)("err",[e,(0,s.z)(),!1,t],void 0,n.D.jserrors,g)},d.il?(0,u.b)((()=>A()),!0):A(),h}(e,v);return(0,h.Qy)(e,T,"api"),(0,h.Qy)(e,A,"exposed"),(0,h.EZ)("activatedFeatures",p.T),T}},3325:(e,t,r)=>{r.d(t,{D:()=>n,p:()=>i});const n={ajax:"ajax",jserrors:"jserrors",metrics:"metrics",pageAction:"page_action",pageViewEvent:"page_view_event",pageViewTiming:"page_view_timing",sessionReplay:"session_replay",sessionTrace:"session_trace",spa:"spa"},i={[n.pageViewEvent]:1,[n.pageViewTiming]:2,[n.metrics]:3,[n.jserrors]:4,[n.ajax]:5,[n.sessionTrace]:6,[n.pageAction]:7,[n.spa]:8,[n.sessionReplay]:9}}},n={};function i(e){var t=n[e];if(void 0!==t)return t.exports;var o=n[e]={exports:{}};return r[e](o,o.exports,i),o.exports}i.m=r,i.d=(e,t)=>{for(var r in t)i.o(t,r)&&!i.o(e,r)&&Object.defineProperty(e,r,{enumerable:!0,get:t[r]})},i.f={},i.e=e=>Promise.all(Object.keys(i.f).reduce(((t,r)=>(i.f[r](e,t),t)),[])),i.u=e=>(({78:"page_action-aggregate",147:"metrics-aggregate",242:"session-manager",317:"jserrors-aggregate",348:"page_view_timing-aggregate",412:"lazy-feature-loader",439:"async-api",538:"recorder",590:"session_replay-aggregate",675:"compressor",733:"session_trace-aggregate",786:"page_view_event-aggregate",873:"spa-aggregate",898:"ajax-aggregate"}[e]||e)+"."+{78:"ac76d497",147:"3dc53903",148:"1a20d5fe",242:"2a64278a",317:"49e41428",348:"bd6de33a",412:"2f55ce66",439:"30bd804e",538:"1b18459f",590:"cf0efb30",675:"ae9f91a8",733:"83105561",786:"06482edd",860:"03a8b7a5",873:"e6b09d52",898:"998ef92b"}[e]+"-1.236.0.min.js"),i.o=(e,t)=>Object.prototype.hasOwnProperty.call(e,t),e={},t="NRBA:",i.l=(r,n,o,a)=>{if(e[r])e[r].push(n);else{var s,c;if(void 0!==o)for(var u=document.getElementsByTagName("script"),d=0;d {s.onerror=s.onload=null,clearTimeout(h);var i=e[r];if(delete e[r],s.parentNode&&s.parentNode.removeChild(s),i&&i.forEach((e=>e(n))),t)return t(n)},h=setTimeout(l.bind(null,void 0,{type:"timeout",target:s}),12e4);s.onerror=l.bind(null,s.onerror),s.onload=l.bind(null,s.onload),c&&document.head.appendChild(s)}},i.r=e=>{"undefined"!=typeof Symbol&&Symbol.toStringTag&&Object.defineProperty(e,Symbol.toStringTag,{value:"Module"}),Object.defineProperty(e,"__esModule",{value:!0})},i.j=364,i.p="https://js-agent.newrelic.com/",(()=>{var e={364:0,953:0};i.f.j=(t,r)=>{var n=i.o(e,t)?e[t]:void 0;if(0!==n)if(n)r.push(n[2]);else{var o=new Promise(((r,i)=>n=e[t]=[r,i]));r.push(n[2]=o);var a=i.p+i.u(t),s=new Error;i.l(a,(r=>{if(i.o(e,t)&&(0!==(n=e[t])&&(e[t]=void 0),n)){var o=r&&("load"===r.type?"missing":r.type),a=r&&r.target&&r.target.src;s.message="Loading chunk "+t+" failed.\n("+o+": "+a+")",s.name="ChunkLoadError",s.type=o,s.request=a,n[1](s)}}),"chunk-"+t,t)}};var t=(t,r)=>{var n,o,[a,s,c]=r,u=0;if(a.some((t=>0!==e[t]))){for(n in s)i.o(s,n)&&(i.m[n]=s[n]);if(c)c(i)}for(t&&t(r);u {i.r(o);var e=i(3325),t=i(5763);const r=Object.values(e.D);function n(e){const n={};return r.forEach((r=>{n[r]=function(e,r){return!1!==(0,t.Mt)(r,"".concat(e,".enabled"))}(r,e)})),n}var a=i(9144);var s=i(5546),c=i(385),u=i(8e3),d=i(5938),f=i(3960),l=i(50);class h extends d.W{constructor(e,t,r){let n=!(arguments.length>3&&void 0!==arguments[3])||arguments[3];super(e,t,r),this.auto=n,this.abortHandler,this.featAggregate,this.onAggregateImported,n&&(0,u.R)(e,r)}importAggregator(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:{};if(this.featAggregate||!this.auto)return;const r=c.il&&!0===(0,t.Mt)(this.agentIdentifier,"privacy.cookies_enabled");let n;this.onAggregateImported=new Promise((e=>{n=e}));const o=async()=>{let t;try{if(r){const{setupAgentSession:e}=await Promise.all([i.e(860),i.e(242)]).then(i.bind(i,3228));t=e(this.agentIdentifier)}}catch(e){(0,l.Z)("A problem occurred when starting up session manager. This page will not start or extend any session.",e)}try{if(!this.shouldImportAgg(this.featureName,t))return void(0,u.L)(this.agentIdentifier,this.featureName);const{lazyFeatureLoader:r}=await i.e(412).then(i.bind(i,8582)),{Aggregate:o}=await r(this.featureName,"aggregate");this.featAggregate=new o(this.agentIdentifier,this.aggregator,e),n(!0)}catch(e){(0,l.Z)("Downloading and initializing ".concat(this.featureName," failed..."),e),this.abortHandler?.(),n(!1)}};c.il?(0,f.b)((()=>o()),!0):o()}shouldImportAgg(r,n){return r!==e.D.sessionReplay||!1!==(0,t.Mt)(this.agentIdentifier,"session_trace.enabled")&&(!!n?.isNew||!!n?.state.sessionReplay)}}var g=i(7633),p=i(7894);class m extends h{static featureName=g.t9;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];if(super(r,n,g.t9,i),("undefined"==typeof PerformanceNavigationTiming||c.Tt)&&"undefined"!=typeof PerformanceTiming){const n=(0,t.OP)(r);n[g.Dz]=Math.max(Date.now()-n.offset,0),(0,f.K)((()=>n[g.qw]=Math.max((0,p.z)()-n[g.Dz],0))),(0,f.b)((()=>{const t=(0,p.z)();n[g.OJ]=Math.max(t-n[g.Dz],0),(0,s.p)("timing",["load",t],void 0,e.D.pageViewTiming,this.ee)}))}this.importAggregator()}}var v=i(1117),b=i(1284);class y extends v.w{constructor(e){super(e),this.aggregatedData={}}store(e,t,r,n,i){var o=this.getBucket(e,t,r,i);return o.metrics=function(e,t){t||(t={count:0});return t.count+=1,(0,b.D)(e,(function(e,r){t[e]=w(r,t[e])})),t}(n,o.metrics),o}merge(e,t,r,n,i){var o=this.getBucket(e,t,n,i);if(o.metrics){var a=o.metrics;a.count+=r.count,(0,b.D)(r,(function(e,t){if("count"!==e){var n=a[e],i=r[e];i&&!i.c?a[e]=w(i.t,n):a[e]=function(e,t){if(!t)return e;t.c||(t=x(t.t));return t.min=Math.min(e.min,t.min),t.max=Math.max(e.max,t.max),t.t+=e.t,t.sos+=e.sos,t.c+=e.c,t}(i,a[e])}}))}else o.metrics=r}storeMetric(e,t,r,n){var i=this.getBucket(e,t,r);return i.stats=w(n,i.stats),i}getBucket(e,t,r,n){this.aggregatedData[e]||(this.aggregatedData[e]={});var i=this.aggregatedData[e][t];return i||(i=this.aggregatedData[e][t]={params:r||{}},n&&(i.custom=n)),i}get(e,t){return t?this.aggregatedData[e]&&this.aggregatedData[e][t]:this.aggregatedData[e]}take(e){for(var t={},r="",n=!1,i=0;i t.max&&(t.max=e),e 2&&void 0!==arguments[2])||arguments[2];super(e,r,j.t,n),c.il&&((0,t.OP)(e).initHidden=Boolean("hidden"===document.visibilityState),(0,N.N)((()=>(0,s.p)("docHidden",[(0,p.z)()],void 0,j.t,this.ee)),!0),(0,O.bP)("pagehide",(()=>(0,s.p)("winPagehide",[(0,p.z)()],void 0,j.t,this.ee))),this.importAggregator())}}var P=i(3081);class C extends h{static featureName=P.t9;constructor(e,t){let r=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(e,t,P.t9,r),this.importAggregator()}}var R,I=i(2210),k=i(1214),H=i(2177),L={};try{R=localStorage.getItem("__nr_flags").split(","),console&&"function"==typeof console.log&&(L.console=!0,-1!==R.indexOf("dev")&&(L.dev=!0),-1!==R.indexOf("nr_dev")&&(L.nrDev=!0))}catch(e){}function z(e){try{L.console&&z(e)}catch(e){}}L.nrDev&&H.ee.on("internal-error",(function(e){z(e.stack)})),L.dev&&H.ee.on("fn-err",(function(e,t,r){z(r.stack)})),L.dev&&(z("NR AGENT IN DEVELOPMENT MODE"),z("flags: "+(0,b.D)(L,(function(e,t){return e})).join(", ")));var M=i(6660);class B extends h{static featureName=M.t;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(r,n,M.t,i),this.skipNext=0;try{this.removeOnAbort=new AbortController}catch(e){}const o=this;o.ee.on("fn-start",(function(e,t,r){o.abortHandler&&(o.skipNext+=1)})),o.ee.on("fn-err",(function(t,r,n){o.abortHandler&&!n[M.A]&&((0,I.X)(n,M.A,(function(){return!0})),this.thrown=!0,(0,s.p)("err",[n,(0,p.z)()],void 0,e.D.jserrors,o.ee))})),o.ee.on("fn-end",(function(){o.abortHandler&&!this.thrown&&o.skipNext>0&&(o.skipNext-=1)})),o.ee.on("internal-error",(function(t){(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,o.ee)})),this.origOnerror=c._A.onerror,c._A.onerror=this.onerrorHandler.bind(this),c._A.addEventListener("unhandledrejection",(t=>{const r=function(e){let t="Unhandled Promise Rejection: ";if(e instanceof Error)try{return e.message=t+e.message,e}catch(t){return e}if(void 0===e)return new Error(t);try{return new Error(t+(0,D.P)(e))}catch(e){return new Error(t)}}(t.reason);(0,s.p)("err",[r,(0,p.z)(),!1,{unhandledPromiseRejection:1}],void 0,e.D.jserrors,this.ee)}),(0,O.m$)(!1,this.removeOnAbort?.signal)),(0,k.gy)(this.ee),(0,k.BV)(this.ee),(0,k.em)(this.ee),(0,t.OP)(r).xhrWrappable&&(0,k.Kf)(this.ee),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}onerrorHandler(t,r,n,i,o){"function"==typeof this.origOnerror&&this.origOnerror(...arguments);try{this.skipNext?this.skipNext-=1:(0,s.p)("err",[o||new F(t,r,n),(0,p.z)()],void 0,e.D.jserrors,this.ee)}catch(t){try{(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,this.ee)}catch(e){}}return!1}}function F(e,t,r){this.message=e||"Uncaught error with no additional information",this.sourceURL=t,this.line=r}let U=1;const q="nr@id";function G(e){const t=typeof e;return!e||"object"!==t&&"function"!==t?-1:e===c._A?0:(0,I.X)(e,q,(function(){return U++}))}function V(e){if("string"==typeof e&&e.length)return e.length;if("object"==typeof e){if("undefined"!=typeof ArrayBuffer&&e instanceof ArrayBuffer&&e.byteLength)return e.byteLength;if("undefined"!=typeof Blob&&e instanceof Blob&&e.size)return e.size;if(!("undefined"!=typeof FormData&&e instanceof FormData))try{return(0,D.P)(e).length}catch(e){return}}}var X=i(7243);class W{constructor(e){this.agentIdentifier=e,this.generateTracePayload=this.generateTracePayload.bind(this),this.shouldGenerateTrace=this.shouldGenerateTrace.bind(this)}generateTracePayload(e){if(!this.shouldGenerateTrace(e))return null;var r=(0,t.DL)(this.agentIdentifier);if(!r)return null;var n=(r.accountID||"").toString()||null,i=(r.agentID||"").toString()||null,o=(r.trustKey||"").toString()||null;if(!n||!i)return null;var a=(0,_.M)(),s=(0,_.Ht)(),c=Date.now(),u={spanId:a,traceId:s,timestamp:c};return(e.sameOrigin||this.isAllowedOrigin(e)&&this.useTraceContextHeadersForCors())&&(u.traceContextParentHeader=this.generateTraceContextParentHeader(a,s),u.traceContextStateHeader=this.generateTraceContextStateHeader(a,c,n,i,o)),(e.sameOrigin&&!this.excludeNewrelicHeader()||!e.sameOrigin&&this.isAllowedOrigin(e)&&this.useNewrelicHeaderForCors())&&(u.newrelicHeader=this.generateTraceHeader(a,s,c,n,i,o)),u}generateTraceContextParentHeader(e,t){return"00-"+t+"-"+e+"-01"}generateTraceContextStateHeader(e,t,r,n,i){return i+"@nr=0-1-"+r+"-"+n+"-"+e+"----"+t}generateTraceHeader(e,t,r,n,i,o){if(!("function"==typeof c._A?.btoa))return null;var a={v:[0,1],d:{ty:"Browser",ac:n,ap:i,id:e,tr:t,ti:r}};return o&&n!==o&&(a.d.tk=o),btoa((0,D.P)(a))}shouldGenerateTrace(e){return this.isDtEnabled()&&this.isAllowedOrigin(e)}isAllowedOrigin(e){var r=!1,n={};if((0,t.Mt)(this.agentIdentifier,"distributed_tracing")&&(n=(0,t.P_)(this.agentIdentifier).distributed_tracing),e.sameOrigin)r=!0;else if(n.allowed_origins instanceof Array)for(var i=0;i 2&&void 0!==arguments[2])||arguments[2];super(r,n,Z.t,i),(0,t.OP)(r).xhrWrappable&&(this.dt=new W(r),this.handler=(e,t,r,n)=>(0,s.p)(e,t,r,n,this.ee),(0,k.u5)(this.ee),(0,k.Kf)(this.ee),function(r,n,i,o){function a(e){var t=this;t.totalCbs=0,t.called=0,t.cbTime=0,t.end=E,t.ended=!1,t.xhrGuids={},t.lastSize=null,t.loadCaptureCalled=!1,t.params=this.params||{},t.metrics=this.metrics||{},e.addEventListener("load",(function(r){_(t,e)}),(0,O.m$)(!1)),c.IF||e.addEventListener("progress",(function(e){t.lastSize=e.loaded}),(0,O.m$)(!1))}function s(e){this.params={method:e[0]},T(this,e[1]),this.metrics={}}function u(e,n){var i=(0,t.DL)(r);i.xpid&&this.sameOrigin&&n.setRequestHeader("X-NewRelic-ID",i.xpid);var a=o.generateTracePayload(this.parsedOrigin);if(a){var s=!1;a.newrelicHeader&&(n.setRequestHeader("newrelic",a.newrelicHeader),s=!0),a.traceContextParentHeader&&(n.setRequestHeader("traceparent",a.traceContextParentHeader),a.traceContextStateHeader&&n.setRequestHeader("tracestate",a.traceContextStateHeader),s=!0),s&&(this.dt=a)}}function d(e,t){var r=this.metrics,i=e[0],o=this;if(r&&i){var a=V(i);a&&(r.txSize=a)}this.startTime=(0,p.z)(),this.listener=function(e){try{"abort"!==e.type||o.loadCaptureCalled||(o.params.aborted=!0),("load"!==e.type||o.called===o.totalCbs&&(o.onloadCalled||"function"!=typeof t.onload)&&"function"==typeof o.end)&&o.end(t)}catch(e){try{n.emit("internal-error",[e])}catch(e){}}};for(var s=0;s 1?e[1]=i:e.push(i)}else e[0]&&e[0].headers&&s(e[0].headers,n)&&(this.dt=n);function s(e,t){var r=!1;return t.newrelicHeader&&(e.set("newrelic",t.newrelicHeader),r=!0),t.traceContextParentHeader&&(e.set("traceparent",t.traceContextParentHeader),t.traceContextStateHeader&&e.set("tracestate",t.traceContextStateHeader),r=!0),r}}function x(e,t){this.params={},this.metrics={},this.startTime=(0,p.z)(),this.dt=t,e.length>=1&&(this.target=e[0]),e.length>=2&&(this.opts=e[1]);var r,n=this.opts||{},i=this.target;"string"==typeof i?r=i:"object"==typeof i&&i instanceof Y?r=i.url:c._A?.URL&&"object"==typeof i&&i instanceof URL&&(r=i.href),T(this,r);var o=(""+(i&&i instanceof Y&&i.method||n.method||"GET")).toUpperCase();this.params.method=o,this.txSize=V(n.body)||0}function A(t,r){var n;this.endTime=(0,p.z)(),this.params||(this.params={}),this.params.status=r?r.status:0,"string"==typeof this.rxSize&&this.rxSize.length>0&&(n=+this.rxSize);var o={txSize:this.txSize,rxSize:n,duration:(0,p.z)()-this.startTime};i("xhr",[this.params,o,this.startTime,this.endTime,"fetch"],this,e.D.ajax)}function E(t){var r=this.params,n=this.metrics;if(!this.ended){this.ended=!0;for(var o=0;o 2&&void 0!==arguments[2])||arguments[2];super(e,t,we.t,r),this.importAggregator()}}new class{constructor(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:(0,_.ky)(16);c._A?(this.agentIdentifier=t,this.sharedAggregator=new y({agentIdentifier:this.agentIdentifier}),this.features={},this.desiredFeatures=new Set(e.features||[]),this.desiredFeatures.add(m),Object.assign(this,(0,a.j)(this.agentIdentifier,e,e.loaderType||"agent")),this.start()):(0,l.Z)("Failed to initial the agent. Could not determine the runtime environment.")}get config(){return{info:(0,t.C5)(this.agentIdentifier),init:(0,t.P_)(this.agentIdentifier),loader_config:(0,t.DL)(this.agentIdentifier),runtime:(0,t.OP)(this.agentIdentifier)}}start(){const t="features";try{const r=n(this.agentIdentifier),i=[...this.desiredFeatures];i.sort(((t,r)=>e.p[t.featureName]-e.p[r.featureName])),i.forEach((t=>{if(r[t.featureName]||t.featureName===e.D.pageViewEvent){const n=function(t){switch(t){case e.D.ajax:return[e.D.jserrors];case e.D.sessionTrace:return[e.D.ajax,e.D.pageViewEvent];case e.D.sessionReplay:return[e.D.sessionTrace];case e.D.pageViewTiming:return[e.D.pageViewEvent];default:return[]}}(t.featureName);n.every((e=>r[e]))||(0,l.Z)("".concat(t.featureName," is enabled but one or more dependent features has been disabled (").concat((0,D.P)(n),"). This may cause unintended consequences or missing data...")),this.features[t.featureName]=new t(this.agentIdentifier,this.sharedAggregator)}})),(0,T.Qy)(this.agentIdentifier,this.features,t)}catch(e){(0,l.Z)("Failed to initialize all enabled instrument classes (agent aborted) -",e);for(const e in this.features)this.features[e].abortHandler?.();const r=(0,T.fP)();return delete r.initializedAgents[this.agentIdentifier]?.api,delete r.initializedAgents[this.agentIdentifier]?.[t],delete this.sharedAggregator,r.ee?.abort(),delete r.ee?.get(this.agentIdentifier),!1}}}({features:[J,m,S,class extends h{static featureName=oe;constructor(t,r){if(super(t,r,oe,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;const n=this.ee;let i;(0,k.QU)(n),this.eventsEE=(0,k.em)(n),this.eventsEE.on(se,(function(e,t){this.bstStart=(0,p.z)()})),this.eventsEE.on(ae,(function(t,r){(0,s.p)("bst",[t[0],r,this.bstStart,(0,p.z)()],void 0,e.D.sessionTrace,n)})),n.on(ce+ne,(function(e){this.time=(0,p.z)(),this.startPath=location.pathname+location.hash})),n.on(ce+ie,(function(t){(0,s.p)("bstHist",[location.pathname+location.hash,this.startPath,this.time],void 0,e.D.sessionTrace,n)}));try{i=new PerformanceObserver((t=>{const r=t.getEntries();(0,s.p)(te,[r],void 0,e.D.sessionTrace,n)})),i.observe({type:re,buffered:!0})}catch(e){}this.importAggregator({resourceObserver:i})}},C,xe,B,class extends h{static featureName=de;constructor(e,r){if(super(e,r,de,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;if(!(0,t.OP)(e).xhrWrappable)return;try{this.removeOnAbort=new AbortController}catch(e){}let n,i=0;const o=this.ee.get("tracer"),a=(0,k._L)(this.ee),s=(0,k.Lg)(this.ee),u=(0,k.BV)(this.ee),d=(0,k.Kf)(this.ee),f=this.ee.get("events"),l=(0,k.u5)(this.ee),h=(0,k.QU)(this.ee),g=(0,k.Gm)(this.ee);function m(e,t){h.emit("newURL",[""+window.location,t])}function v(){i++,n=window.location.hash,this[ve]=(0,p.z)()}function b(){i--,window.location.hash!==n&&m(0,!0);var e=(0,p.z)();this[pe]=~~this[pe]+e-this[ve],this[ye]=e}function y(e,t){e.on(t,(function(){this[t]=(0,p.z)()}))}this.ee.on(ve,v),s.on(be,v),a.on(be,v),this.ee.on(ye,b),s.on(ge,b),a.on(ge,b),this.ee.buffer([ve,ye,"xhr-resolved"],this.featureName),f.buffer([ve],this.featureName),u.buffer(["setTimeout"+le,"clearTimeout"+fe,ve],this.featureName),d.buffer([ve,"new-xhr","send-xhr"+fe],this.featureName),l.buffer([me+fe,me+"-done",me+he+fe,me+he+le],this.featureName),h.buffer(["newURL"],this.featureName),g.buffer([ve],this.featureName),s.buffer(["propagate",be,ge,"executor-err","resolve"+fe],this.featureName),o.buffer([ve,"no-"+ve],this.featureName),a.buffer(["new-jsonp","cb-start","jsonp-error","jsonp-end"],this.featureName),y(l,me+fe),y(l,me+"-done"),y(a,"new-jsonp"),y(a,"jsonp-end"),y(a,"cb-start"),h.on("pushState-end",m),h.on("replaceState-end",m),window.addEventListener("hashchange",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("load",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("popstate",(function(){m(0,i>1)}),(0,O.m$)(!0,this.removeOnAbort?.signal)),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}}],loaderType:"spa"})})(),window.NRBA=o})(); window.jQuery || document.write(' ') CKEDITOR_BASEPATH='https://f1000research.com/js/vendor/ckeditor/' window.reactTheme = 'research'; window.MathJax = { CommonHTML: { linebreaks: { automatic: true } }, 'HTML-CSS': { linebreaks: { automatic: true } }, SVG: { linebreaks: { automatic: true } }, AuthorInit: function() { MathJax.Hub.Register.MessageHook('End Process', function () { let timeout = false; // holder for timeout id const delay = 250; // delay after event is "complete" to run callback const reflowMath = function() { const dispFormulas = document.querySelectorAll('.disp-formula.panel'); if (!dispFormulas) { return; } for (const dispFormula of dispFormulas) { const child = dispFormula.querySelector('.MathJax_Preview').nextSibling.firstChild; const isMultiline = MathJax.Hub.getAllJax(dispFormula)[0].root.isMultiline; if (dispFormula.offsetWidth < child.offsetWidth || isMultiline) { MathJax.Hub.Queue(['Rerender', MathJax.Hub, dispFormula]); } } }; window.addEventListener('resize', function() { clearTimeout(timeout); // clear the timeout timeout = setTimeout(reflowMath, delay); // start timing for event "completion" }); }); }, }; if (window.location.hash == '#_=_'){ window.location = window.location.href.split('#')[0] } !function(f,b,e,v,n,t,s){if(f.fbq)return;n=f.fbq=function() {n.callMethod? n.callMethod.apply(n,arguments):n.queue.push(arguments)} ;if(!f._fbq)f._fbq=n; n.push=n;n.loaded=!0;n.version='2.0';n.queue=[];t=b.createElement(e);t.async=!0; t.src=v;s=b.getElementsByTagName(e)[0];s.parentNode.insertBefore(t,s)}(window, document,'script','https://connect.facebook.net/en_US/fbevents.js'); fbq('init', '1641728616063202'); fbq('track', "PixelInitialized", {}); (function(h,o,t,j,a,r){ h.hj=h.hj||function(){(h.hj.q=h.hj.q||[]).push(arguments)}; h._hjSettings={hjid:2318163,hjsv:6}; a=o.getElementsByTagName('head')[0]; r=o.createElement('script');r.async=1; r.src=t+h._hjSettings.hjid+j+h._hjSettings.hjsv; a.appendChild(r); })(window,document,'https://static.hotjar.com/c/hotjar-','.js?sv='); search file_upload Submit your research search menu close search Browse Gateways & Collections How to Publish Submit your Research My Submissions Article Guidelines Article Guidelines (New Versions) Open Data, Software and Code Guidelines Open Data and Accessible Source Materials Guidelines (HSS) Open Data, Software and Code Guidelines (PSE) Prepublication Checks Production Process Posters and Slides Guidelines Document Guidelines Article Processing Charges Peer Review Finding Article Reviewers About How it Works For Reviewers Our Advisors Policies Glossary FAQs For Developers Newsroom Contact My Research Submissions Content and Tracking Alerts My Details Sign In file_upload Submit your research { "@context": "https://schema.org", "@type": "ScholarlyArticle", "mainEntityOfPage": { "@type": "WebPage", "@id": "https://f1000research.com/articles/12-1091" }, "headline": "Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read...", "datePublished": "2023-09-01T16:15:45", "dateModified": "2024-04-16T15:48:32", "author": [ { "@type": "Person", "name": "Marie Lataretu" }, { "@type": "Person", "name": "Oliver Drechsel" }, { "@type": "Person", "name": "René Kmiecinski" }, { "@type": "Person", "name": "Kathrin Trappe" }, { "@type": "Person", "name": "Martin Hölzer" }, { "@type": "Person", "name": "Stephan Fuchs" } ], "publisher": { "@type": "Organization", "name": "F1000Research", "logo": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 480, "width": 60 } }, "image": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 1200, "width": 150 }, "description": " Background Accurate genome sequences form the basis for genomic surveillance programs, the added value of which was impressively demonstrated during the COVID-19 pandemic by tracing transmission chains, discovering new viral lineages and mutations, and assessing them for infectiousness and resistance to available treatments. Amplicon strategies employing Illumina sequencing have become widely established for variant detection and reference-based reconstruction of SARS-CoV-2 genomes, and are routine bioinformatics tasks. Yet, specific challenges arise when analyzing amplicon data, for example, when crucial and even lineage-determining mutations occur near primer sites. Methods We present CoVpipe2, a bioinformatics workflow developed at the Public Health Institute of Germany to reconstruct SARS-CoV-2 genomes based on short-read sequencing data accurately. The decisive factor here is the reliable, accurate, and rapid reconstruction of genomes, considering the specifics of the used sequencing protocol. Besides fundamental tasks like quality control, mapping, variant calling, and consensus generation, we also implemented additional features to ease the detection of mixed samples and recombinants. Results We highlight common pitfalls in primer clipping, detecting heterozygote variants, and dealing with low-coverage regions and deletions. We introduce CoVpipe2 to address the above challenges and have compared and successfully validated the pipeline against selected publicly available benchmark datasets. CoVpipe2 features high usability, reproducibility, and a modular design that specifically addresses the characteristics of short-read amplicon protocols but can also be used for whole-genome short-read sequencing data. Conclusions CoVpipe2 has seen multiple improvement cycles and is continuously maintained alongside frequently updated primer schemes and new developments in the scientific community. Our pipeline is easy to set up and use and can serve as a blueprint for other pathogens in the future due to its flexibility and modularity, providing a long-term perspective for continuous support. CoVpipe2 is written in Nextflow and is freely accessible from {https://github.com/rki-mf1/CoVpipe2}{github.com/rki-mf1/CoVpipe2} under the GPL3 license. " } { "@context": "http://schema.org", "@type": "BreadcrumbList", "itemListElement": [ { "@type": "ListItem", "position": "1", "item": { "@id": "https://f1000research.com/", "name": "Home" } }, { "@type": "ListItem", "position": "2", "item": { "@id": "https://f1000research.com/browse/articles", "name": "Browse" } }, { "@type": "ListItem", "position": "3", "item": { "@id": "https://f1000research.com/articles/12-1091/22", "name": "Lessons learned: overcoming common challenges in reconstructing the..." } } ] } Home Browse Lessons learned: overcoming common challenges in reconstructing the... ALL Metrics - Views Downloads Get PDF Get XML Cite How to cite this article Lataretu M, Drechsel O, Kmiecinski R et al. Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.12688/f1000research.136683.2 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. Close Copy Citation Details Export Export Citation Sciwheel EndNote Ref. Manager Bibtex ProCite Sente EXPORT Select a format first Track Share ▬ ✚ Software Tool Article Revised Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] Marie Lataretu https://orcid.org/0000-0002-3637-5870 1 , Oliver Drechsel https://orcid.org/0000-0002-7661-2535 1 , René Kmiecinski 1 , Kathrin Trappe https://orcid.org/0000-0002-2887-1575 1 , Martin Hölzer 1 * , Stephan Fuchs 1 * Marie Lataretu https://orcid.org/0000-0002-3637-5870 1 , Oliver Drechsel https://orcid.org/0000-0002-7661-2535 1 , [...] René Kmiecinski 1 , Kathrin Trappe https://orcid.org/0000-0002-2887-1575 1 , Martin Hölzer 1 * , Stephan Fuchs 1 * * Equal contributors PUBLISHED 16 Apr 2024 Author details Author details 1 Genome Competence Center (MF1), Robert Koch Institute, Berlin, 13353, Germany Marie Lataretu Roles: Data Curation, Formal Analysis, Investigation, Methodology, Software, Validation, Visualization, Writing – Original Draft Preparation Oliver Drechsel Roles: Methodology, Software, Writing – Review & Editing René Kmiecinski Roles: Methodology, Software Kathrin Trappe Roles: Software, Writing – Review & Editing Martin Hölzer Roles: Formal Analysis, Investigation, Methodology, Supervision, Validation, Writing – Original Draft Preparation Stephan Fuchs Roles: Conceptualization, Funding Acquisition, Methodology, Project Administration, Supervision, Writing – Review & Editing OPEN PEER REVIEW DETAILS REVIEWER STATUS This article is included in the Bioinformatics gateway. This article is included in the Virus Bioinformatics collection. This article is included in the Coronavirus (COVID-19) collection. Abstract Background Accurate genome sequences form the basis for genomic surveillance programs, the added value of which was impressively demonstrated during the COVID-19 pandemic by tracing transmission chains, discovering new viral lineages and mutations, and assessing them for infectiousness and resistance to available treatments. Amplicon strategies employing Illumina sequencing have become widely established for variant detection and reference-based reconstruction of SARS-CoV-2 genomes, and are routine bioinformatics tasks. Yet, specific challenges arise when analyzing amplicon data, for example, when crucial and even lineage-determining mutations occur near primer sites. Methods We present CoVpipe2, a bioinformatics workflow developed at the Public Health Institute of Germany to reconstruct SARS-CoV-2 genomes based on short-read sequencing data accurately. The decisive factor here is the reliable, accurate, and rapid reconstruction of genomes, considering the specifics of the used sequencing protocol. Besides fundamental tasks like quality control, mapping, variant calling, and consensus generation, we also implemented additional features to ease the detection of mixed samples and recombinants. Results We highlight common pitfalls in primer clipping, detecting heterozygote variants, and dealing with low-coverage regions and deletions. We introduce CoVpipe2 to address the above challenges and have compared and successfully validated the pipeline against selected publicly available benchmark datasets. CoVpipe2 features high usability, reproducibility, and a modular design that specifically addresses the characteristics of short-read amplicon protocols but can also be used for whole-genome short-read sequencing data. Conclusions CoVpipe2 has seen multiple improvement cycles and is continuously maintained alongside frequently updated primer schemes and new developments in the scientific community. Our pipeline is easy to set up and use and can serve as a blueprint for other pathogens in the future due to its flexibility and modularity, providing a long-term perspective for continuous support. CoVpipe2 is written in Nextflow and is freely accessible from {https://github.com/rki-mf1/CoVpipe2}{github.com/rki-mf1/CoVpipe2} under the GPL3 license. READ ALL READ LESS Keywords SARS-CoV-2, genome reconstruction, whole-genome sequencing, short reads, Illumina, amplicons, WGS, Nextflow pipeline, virus bioinformatics Corresponding Author(s) Marie Lataretu ( [email protected] ) Stephan Fuchs ( [email protected] ) Close Corresponding authors: Marie Lataretu, Stephan Fuchs Competing interests: No competing interests were disclosed. Grant information: .L. was supported by the European Centre for Disease Control (grant number ECDC GRANT/2021/008 ECD.12222). This work was further supported by the European Health and Digital Executive Agency (grant number 101113012) and Bundesministerium für Wirtschaft und Klimaschutz, Daten- und KI-gestütztes Frühwarnsystem zur Stabilisierung der deutschen Wirtschaft (grant number 01MK21009H). The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript. Copyright: © 2024 Lataretu M et al . This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. How to cite: Lataretu M, Drechsel O, Kmiecinski R et al. Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.12688/f1000research.136683.2 ) First published: 01 Sep 2023, 12 :1091 ( https://doi.org/10.12688/f1000research.136683.1 ) Latest published: 16 Apr 2024, 12 :1091 ( https://doi.org/10.12688/f1000research.136683.2 ) Revised Amendments from Version 1 In this revision, several enhancements have been made to augment the clarity and comprehensibility of the manuscript while effectively addressing reviewer comments. Textual refinements have been implemented to ensure improved readability throughout the document. An addition comprises the incorporation of additional references and links, notably for resources such as Nextclade, Pangolin, and the precalculated Kraken 2 database on Zenodo. This improves accessibility by providing readers with easily accessible supplementary materials. Also, we appended a list of abbreviations to facilitate smoother comprehension of specialized terminology, addressing a reviewer's comment. Furthermore, the conclusion section has been extended to delineate the potential applications of the methodologies employed in this study to other viruses, thereby broadening the scope and relevance of the research findings. We included a comment elucidating the rationale behind the selection of specific tools, offering insights into the methodological choices made during the research process. The results heading has been removed to streamline the structure and address reviewer feedback, contributing to a more coherent manuscript organization. Lastly, funding sources have been added. In this revision, several enhancements have been made to augment the clarity and comprehensibility of the manuscript while effectively addressing reviewer comments. Textual refinements have been implemented to ensure improved readability throughout the document. An addition comprises the incorporation of additional references and links, notably for resources such as Nextclade, Pangolin, and the precalculated Kraken 2 database on Zenodo. This improves accessibility by providing readers with easily accessible supplementary materials. Also, we appended a list of abbreviations to facilitate smoother comprehension of specialized terminology, addressing a reviewer's comment. Furthermore, the conclusion section has been extended to delineate the potential applications of the methodologies employed in this study to other viruses, thereby broadening the scope and relevance of the research findings. We included a comment elucidating the rationale behind the selection of specific tools, offering insights into the methodological choices made during the research process. The results heading has been removed to streamline the structure and address reviewer feedback, contributing to a more coherent manuscript organization. Lastly, funding sources have been added. See the authors' detailed response to the review by Wolfgang Maier See the authors' detailed response to the review by Sondes Haddad-Boubaker READ REVIEWER RESPONSES List of abbreviations BAM file: Binary Alignment and Map file BED file: Browser Extensible Data file BEDPE file: Browser Extensible Data Paired-End file CCO license: Creative Commons Zero license CDC: Centers for Disease Control and Prevention CorSurV: Coronavirus Surveillance Verordnung (eng., Coronavirus Surveillance Regulation) COVID-19: Coronavirus disease 2019 CSV file: Comma-Separated Values file DESH: Deutscher Elektronischer Sequenzdaten-Hub (eng., German Electronic Sequence Data Hub) EBI: European Bioinformatics Institute EMBL: European Molecular Biology Laboratory ENA: European Nucleotide Archive GFF file: General Feature Format file GISAID: Global Initiative on Sharing All Influenza Data GPL3 license: GNU General Public License HPC: High-Performance Computing HTML: Hypertext Markup Language ID: Identifier IMS-SC2: Integrated Molecular Surveillance for SARS-CoV-2 indel: Insertion/Deletion Variant IUPAC: International Union of Pure and Applied Chemistry JSON file: JavaScript Object Notation file NGS:Next-Generation Sequencing ONT: Oxford Nanopore Technologies PCR: Polymerase Chain Reaction QC: Quality Control RKI: Robert Koch Institute SARS-CoV-2: Severe acute respiratory syndrome coronavirus 2 SRA: Sequence Read Archive VCF file: Variant Call Format file VOC: Variants of Concern VOI: Variants of Interest WSL: Windows Subsystem for Linux Introduction Since the publication of the first genome sequence of the novel SARS-CoV-2 virus in January 2020 – just 12 days after the initial report of the virus – the international GISAID database 1 – 3 now includes more than 15.5 million SARS-CoV-2 whole-genome sequences (accessed May 23, 2023). The genomic data and metadata collected in GISAID and other resources such as EBI’s COVID-19 Data Portal 4 are pivotal for the largest worldwide genomic surveillance effort ever undertaken to track the evolution and spread of the virus causing the COVID-19 disease. Important viral genome regions have been monitored for mutations, for example, in the S (spike) gene and other immunologically relevant loci. The reconstruction of accurate SARS-CoV-2 genomic sequences is paramount to detect and track substitutions, insertions, and deletions correctly; interpret them in terms of vaccine development, test the efficiency of target regions and antibody binding sites, detect outbreaks and transmission chains, and finally inform public health authorities to consider adjustment of containment measures. 5 According to Ewan Birney, director of EMBL-EBI in Cambridge, U.K., “Genome sequencing is routine in the same way the U.S. Navy routinely lands planes on aircraft carriers. Yes, a good, organized crew does this routinely, but it is complex and surprisingly easy to screw up.”. 6 This quote is no less accurate for genome reconstruction, a crucial step in SARS-CoV-2 genomic surveillance. While sequencing efforts were scaling up rapidly around the globe, several pipelines for the reference-based assembly of SARS-CoV-2 genomes were developed in parallel and in an attempt to rapidly generate the necessary genome sequences ( Table 1 ). During the first lockdown in Germany in mid-March 2020, the Bioinformatics unit at the Robert Koch Institute (RKI), Germany’s Public Health institute, also started developing a genome reconstruction pipeline, specifically targeting Illumina amplicon sequencing data and amplicon primer schemes. During the development, the pipeline was extensively tested and has gone through continuous improvement due to adjusted wet lab protocols and primer schemes, to accurately call variants in low-coverage regions and near primer sites, to deal with deletions and low-coverage regions correctly, and to robustly reconstruct high-quality SARS-CoV-2 consensus sequences for downstream analyses and genomic surveillance. Due to these features and tests, CoVpipe1 was successfully used in the past at RKI’s sequencing facility and several labs nationwide. 7 – 10 If not addressed appropriately, genotyping errors can lead to wrong consensus sequences and thus impact downstream analyses such as phylogenetic reconstructions and transmission chain tracking in outbreaks. 11 Table 1. A collection of available software for SARS-CoV-2 genome reconstruction. Here, we mainly focus on open-source pipelines with available source code, specifically targeting the reconstruction of SARS-CoV-2 genomes, also including general pipelines adapted to work with SARS-CoV-2 sequencing data (e.g., V-pipe). Publ. – Publication focusing on the pipeline in a preprint or a peer-reviewed journal, Seq. tech. – focused sequencing technology, Implementation – main software backbone to run the tool (please note that not all pipelines use a workflow management system such as Nextflow 25 or Snakemake 26 ), Dependencies – a list of options to handle necessary software dependencies, Latest release – latest available release version as accessed May 23, 2023. mNGS – sequencing approach not based on amplicons but rather sequencing all available RNA/cDNA (metatranscriptomics/genomics). Tool Publ. Code Seq. tech. Implementation Dependencies Latest release CoVpipe2 – github.com/rki-mf1/covpipe2 Illumina Nextflow Conda, Mamba, Docker, Singularity 0.4.2 (2023-06-09) poreCov 24 github.com/replikation/poreCov Nanopore Nextflow Docker, Singularity 1.8.3 (2023-05-12) viralrecon 27 github.com/nf-core/viralrecon Illumina, Nanopore Nextflow Conda, Docker, Singularity, Podman, Shifter, Charliecloud 2.6.0 (2023-03-23) SIGNAL 28 github.com/jaleezyy/covid-19-signal Illumina Snakemake Conda, Mamba, Docker (a single container with all dependencies) 1.6.2 (2023-05-20) V-pipe 29 github.com/cbg-ethz/V-pipe Illumina Snakemake Conda, Docker (a single container with all dependencies) 2.99.3 (2022-11-02) NCBI SC2VC – github.com/ncbi/sars2variantcalling Illumina, Nanopore, PacBio Snakemake (multiple pipelines, run via wrapper script) Docker (a single container with all dependencies) 3.3.4 (2022-10-28) VirPipe 30 github.com/KijinKims/VirPipe Illumina, Nanopore (mNGS) Python, Nextflow Conda and Docker 1.0.0 (2022-09-24) ViralFlow 31 github.com/dezordi/ViralFlow Illumina Python Conda, Singularity, Docker (one single container/environment) v.0.0.6 (2021-04-18) EDGE COVID-19 32 github.com/LANL-Bioinformatics/EDGE/tree/SARS-CoV2 Illumina, Nanopore Python, Perl, available as web tool Docker (a single container with all dependencies) 2.4.0 (2020-12-03) Galaxy-COVID19 33 galaxyproject.org/projects/covid19/workflows Illumina (amplicon & mNGS), Nanopore Galaxy Access to a Galaxy instance different pipelines & versions HAVoC 34 bitbucket.org/auto_cov_pipeline/havoc Illumina Shell scripts Conda, Mamba no release PipeCoV 35 github.com/alvesrco/pipecov Illumina (amplicon & mNGS) Shell scripts Docker no release While sequencing intensity increased and turnaround times on variant detection decreased in different countries, there are also major disparities between high-, low- and middle-income countries in the SARS-CoV-2 global genomic surveillance efforts. 12 A recent study also found significant differences between bioinformatics approaches that use the same input data but detect different variants in SARS-CoV-2 samples. 13 In addition, each technology and sequencing approach has its own advantages and limitations, challenging a harmonized genomic surveillance of the virus. 14 In Germany, sequencing efforts increased tremendously in January 2021 following the entry into force of a federal directive (Coronavirus Surveillance Verordnung—CorSurV). Subsequently, a large-scale, decentralized genomic sequencing and data collection system (“Deutscher Elektronischer Sequenzdaten-Hub”, DESH) 15 has been established and was running until June 2023, accompanied by a medium-scale integrated molecular surveillance infrastructure (IMS-SC2) which is still continued at the RKI. 8 By May 22, 2023, 1,227,036 whole-genome SARS-CoV-2 sequences that met the quality criteria were transmitted to the RKI via DESH. Due to their low cost, sensitivity, flexibility, specificity, and efficiency, amplicon-based sequencing designs are broadly used for SARS-CoV-2 sequencing and reference-based genome reconstruction. 16 – 20 From the ∼ 1.2 million DESH genomes (publicly available at github.com/robert-koch-institut/SARS-CoV-2-Sequenzdaten_aus_Deutschland ), ∼ 1.1 million (90.91%) were sequenced with Illumina devices, highlighting the importance of the technology for genomic surveillance ( Table 2 ). Illumina technology has a lower share at the international level (78.57%), while Oxford Nanopore Technologies (ONT) increased to 12.73% ( Table 2 ). However, Illumina remains the most widely used approach among the various sequencing technologies, followed by ONT sequencing by a wide margin. A recent benchmark study also showed the advantages of using Illumina MiSeq compared to ONT GridION for SARS-CoV-2 sequencing, resulting in a higher number of consensus genomes classified by Nextclade 21 as good and mediocre. 22 However, these results are, of course, also dependent on the bioinformatics toolchains and could change as ONT becomes more accurate. 23 In addition, although both technologies require the same computational steps for reference-based genome reconstruction (preprocessing, mapping, variant calling, consensus), they need different tools optimized for either short- or long-read data and the associated error profiles to produce high-quality consensus sequences. Therefore, we developed one pipeline, especially for ONT data, 24 and CoVpipe2, specifically targeting SARS-CoV-2 amplicon data derived from short-read Illumina sequencing. CoVpipe2 is a Nextflow re-implementation of CoVpipe1 (written in Snakemake, gitlab.com/RKIBioinformaticsPipelines/ncov_minipipe ) and comes with additional features, simplified installation, full container support, and continuous maintenance. Table 2. Sequencing technologies used for SARS-CoV-2 sequencing of German and international samples. Data based on 1.2 million and 15.5 million whole-genome sequences submitted to the German DESH portal (“Deutscher Elektronischer Sequenzdaten-Hub”) and to the GISAID database, respectively (accessed: May 22, 2023). To the best of our knowledge, we corrected typos such as Nanonopore , Nasnopore , and Illunima in the GISAID metadata and summarized the available terms into the broader categories shown in this table ( e.g. , NextSeq500 ⇒ Illumina ). Entries we could not assign to one of the listed sequencing technologies were added to the Other/Unknown category. Please note that most German DESH sequences are also part of the GISAID data set. # – Number of sequences, % – Percentage of sequences on total data set. Sequencing technology # DESH % DESH # GISAID % GISAID Illumina 1,115,501 90.91 12,240,675 78.57 Nanopore 75,358 6.14 1,984,505 12.73 SMRT PacBio 0 0 573,214 3.67 Ion Torrent 29,450 2.40 349,001 2.24 MGI DNBSEQ 0 0 177,480 1.13 Sanger 0 0 57,820 0.37 Other/Unknown 6,727 0.54 195,833 1.25 Total 1,227,036 100 15,578,528 100 Here we present CoVpipe2, our pipeline engineered over nearly three years of pandemic genome sequencing that accurately reconstructs SARS-CoV-2 consensus sequences from Illumina short-read sequencing data, focusing on challenges associated with amplicon sequencing on a large scale. Besides implementation details, we also highlight pitfalls we discovered and solved during the pipeline development, focusing on variant calling artifacts and how we deal with them in the pipeline. Methods Implementation CoVpipe2 is implemented using the workflow management system Nextflow 25 to achieve high reproducibility and performance on various platforms. The user can choose to use CoVpipe2 with Conda or Mamba support, 36 or containers (Docker, 37 Singularity 38 ) to handle all software dependencies. The Conda/Mamba environments and the container images are preconfigured and have fixed versions of the incorporated tools. The precompiled Docker containers are stored on hub.docker.com/u/rkimf1 . Containers and environments are downloaded and cached automatically when executing CoVpipe2. If needed, the Docker images can also be converted into Singularity images by the pipeline. CovPipe2 includes Nextclade 39 and pangolin 40 for lineage assignment. Both tools rely on their latest code and database versions. To address this, we implemented the --update option inspired by poreCov, 24 which triggers an update to the latest available version from anaconda.org/bioconda/pangolin and anaconda.org/bioconda/nextclade , or hub.docker.com/u/rkimf1 , respectively. --update is disabled by default; the tool versions can also be pinned manually. When CoVpipe2 is running on a high-performance computing (HPC) cluster ( e.g. , SLURM, LSF) or in the cloud ( e.g. , AWS, GCP, Azure), all resources (CPUs, RAM) are pre-configured for all processes but can be customized via a user-specific configuration file. We use complete version control for CoVpipe2, from the workflow itself (releases) to each tool, to guarantee reproducible results. To this end, all Conda/Mamba environments and containers use fixed versions. In addition, each CoVpipe2 release can be invoked and executed individually, and the tool versions used during genome reconstruction and analysis are listed in Nextflow report files. CoVpipe2 is publicly available under a GPL-3.0 license at github.com/rki-mf1/covpipe2 , where details about the implementation and different executions of CoVpipe2 can be found. We use GitHub’s CI for various pipeline tests, particularly a dry-run to check for integrity and an end-to-end test with special attention to the called variants to ensure continuous code quality and robust results. Operation As a minimal setup, Nextflow (minimal version 22.10.1, nextflow.io ) and either Conda, Mamba, Docker, or Singularity need to be installed for CoVpipe2. Nextflow can be used on any POSIX-compatible system, e.g. , Linux, OS X, and on Windows via the Windows Subsystem for Linux (WSL). Nextflow requires Bash 3.2 (or later) and Java 11 (or later, up to 18) to be installed. Initial installation and further updates to the workflow and included tools can be performed with simple commands: # install (or update) the pipeline nextflow pull rki-mf1/CoVpipe2 # check available pipeline versions nextflow info rki-mf1/CoVpipe2 # run a certain release version nextflow run rki-mf1/CoVpipe2 -r v0.4.1 --help # test the installation with local execution and Conda nextflow run rki-mf1/CoVpipe2 -r v0.4.1 -profile local,conda,test --cores 2 --max_cores 4 The pipeline can be executed on various platforms controlled via the Nextflow - profile parameter, which makes it easily scalable, e.g. , for execution on an HPC. Each run profile is created by combining different Executors (local, slurm) and Engines (conda, mamba, docker, singularity); the default execution-engine combination (profile) is -profile local , conda . An overview of the workflow is given in Figure 1 . FASTQ files and a reference genome sequence (FASTA) are the minimum required pipeline inputs. If no reference genome sequence is provided, the SARS-CoV-2 index case reference genome with accession number MN908947.3 (identical to NC_045512.2) is used by default (as well as the corresponding annotation GFF), and then only FASTQ files are required. All files can be provided via file paths or defined via a comma-separated sample sheet (CSV); thus, CoVpipe2 can run in batch mode and analyze multiple samples in one run. Optionally, raw reads are checked for mixed samples with the tool LCS. 41 Next, raw reads are quality-filtered and trimmed using fastp 42 and optionally filtered taxonomically by Kraken 2. 43 We provide an automated download of a precalculated Kraken2 database ( doi.org/10.5281/zenodo.6333909 ) composed of SARS-CoV-2 and human genomes from Zenodo ( zenodo.org ). However, the user is free to use a custom database. The reads are aligned to the reference genome using BWA, 44 and the genome coverage is calculated by BEDTools genomecov. 45 Primers can be optionally clipped after mapping with BAMclipper, 46 which is essential to avoid contaminating primer sequences in amplicon data. To locate the primer sequences, a browser extensible data paired-end ( BEDPE ) file containing all primer coordinates is required as input. If only a BED file is provided, CoVpipe2 can automatically convert it to a BEDPE file. Users can also choose from the provided popular VarSkip ( github.com/nebiolabs/VarSkip ) and ARTIC primer schemes. 47 Next, FreeBayes 48 calls variants (default thresholds: minimum alternate count of 10, minimum alternate fraction of 0.1, and minimum coverage of 20), which are normalized with BCFtools norm. 49 Resulting variants are analyzed and annotated with SnpEff, 50 filtered by QUAL (default 10), INFO/SAP (default disabled), and INFO/MQM (default 40) values, and optionally adjusted for the genotype (default enabled with minimum variant frequency 0.9). Thus, in the case of mixed variants, the predominant variant can be defined as a homozygous genotype if its frequency reaches a certain threshold. With these settings, alternate variants with a more than 90% frequency are set as alternative nucleotides. Furthermore, CoVpipe2 filters out indels below a certain allele frequency, which is, by default, enabled with a minimum allele frequency of 0.6. We create a low coverage mask, without deletions, from the mapping file and the adjusted and filtered VCF file (default coverage cutoff 20), which is then used in the consensus calling with BCFtools consensus. We output an ambiguous consensus sequence with IUPAC characters 51 (RYMKSWHBVDN) and a masked one only containing ACTG and Ns. Then, PRESIDENT ( GitHub ) assesses the quality of each reconstructed genome via pairwise alignment to the SARS-CoV-2 index case (NC_045512.2) using pblat 52 with an identity threshold of 0.9 and N threshold of 0.05 per default. pangolin assigns a lineage and Nextclade annotates the mutations; the Nextclade alignment serves as input for sc2rf (original repository from github.com/lenaschimmel/sc2rf , with updates from github.com/ktmeaton/ncov-recombinant ) for detection of recombinants. If an annotation file is provided, the reference annotation is mapped to the reconstructed genome with Liftoff. 53 Figure 1. Overview of the CoVpipe2 workflow. The illustration shows all input ( ) and output ( ) files as well as optional processing steps and optional input ( ). For each computational step, the used parameters and default values (in brackets [...]) are provided, as well as additional comments ((...)). The arrows connect all steps and are colored to distinguish different data processing steps: green – read (FASTQ) quality control and taxonomy filtering, yellow – reference genome (FASTA) and reference annotation (GFF) for lift-over, blue – mapping files (BAM) and low coverage filter (BED), purple – variant calls (VCF), orange – consensus sequence (FASTA). The icons and diagram components that make up the schematic figure were originally designed by James A. Fellow Yates and nf-core under a CCO license (public domain). All results are summarized via an R Markdown template. The resulting HTML-based report summarizes different quality measures and mapping statistics for each input sample, thus allowing the user to spot low-performing samples even in extensive sequencing runs with many samples. A conditional notification warns the user if samples identified as negative controls (by matching the string ’NK’) show high reference genome coverage (threshold over 0.2). We carefully selected the bioinformatics tools integrated into CoVpipe2 based on internal benchmarks and in-depth manual reviews of sequencing data, mapping results, and called variants. Based on our hands-on experience with SARS-CoV-2 sequencing datasets during the pandemic, this approach ensured that tools that can robustly detect high allele depth and well-covered variants are used for detection. Despite the primary goal of CoVpipe2 to identify high-confidence variants to reconstruct robust consensus genomes for genomic surveillance, the pipeline also provides flexibility for adaptation to different research environments. Common variant calling challenges in amplicon-based genome reconstruction and solutions in CoVpipe2 Here, we highlight some implementation decisions that prevent common pitfalls and thus improve the quality of reconstructed SARS-CoV-2 genomes. Although amplicon-based approaches are widely used, these technologies are associated with flaws and limitations that must be considered to ensure that the genotypes obtained, and thus the resulting genome sequences, are reliable. 13 , 54 While some of these implementations might seem obvious and easy to fix, we frequently observed specific errors in the vast amount of consensus genome sequences sent to us via the DESH system. Accurate primer clipping to avoid dilution and edge effects: Primer clipping is an essential step for amplicon sequencing data because primers are inherent to the reference sequence and can disguise true variants in the sample. 13 However, removing primers before the mapping step can result in unwanted edge effects. 55 For example, deletions located close to the end of amplicons may be soft-clipped by the mapping software and hence can not be called as variants subsequently ( Figure 2 ). Therefore, primer clipping should be performed after mapping to prevent any soft clipping of variants close to the amplicon ends. Figure 2. Exemplary cases that should be considered during genotyping. (A) Primer sequences need clipping to call true variants (red) and not mask them via reference bases (blue). (B) Early primer clipping may result in missed deletions due to algorithmic soft clipping (primer sequence in light gray). (C) Genotyping parameters must be carefully set to reliably call different variant cases and represent them in the final consensus. CoVpipe2 puts primer clipping after the mapping step to prevent B) and implements carefully chosen default parameters for robust genotyping also of mixed variants, C). As an example and worst-case scenario, clipping primer sequences before mapping bears the risk of missing a critical deletion used to define the previous variant of concern (VOC) B.1.1.7, namely deletion HV69/70 in the spike gene (S:H69-, S:V70-). We observed such a misclassification using Paragon CleanPlex amplicon-based sequencing. The kit uses primers similar to the ARTIC protocol V3, where the deletion S:HV69/70 is close to the end of an amplicon. If primer clipping is performed before mapping, the mapping tool might soft clip the amplicon end rather than opening a gap, which is more expensive than masking a few nucleotides ( Figure 2 ). We checked 151,565 B.1.1.7 sequences obtained via DESH for the characteristic S:H69, S:V70 deletions and found that 139,891 (92.3%) included the deletion. The remaining sequences contained the deletion only partially or lacked it completely. It is unlikely that these B.1.1.7 sequences lost this characteristic deletion. Thus we assume that some of the reconstructed Alpha sequences sent to the RKI from different laboratories in Germany do not account for the described effect and thus miss the detection of this deletion. Unfortunately, we don’t know which bioinformatics pipelines the submitting laboratories used and can only assume an error because of missing or incorrect primer clipping. To avoid this problem, we shift primer clipping after the read mapping step in CoVpipe2; otherwise, a vital feature of an emerging virus lineage might be missed if mutations accumulate close to amplicon ends. Genotype adjustment to exclude sporadic variant calls: As an additional feature, we provide the option to adjust the genotype for sites where the vast majority of reads support a variant call, but the variant was called heterozygous. By default, the genotype of these locations is set to homozygous if a particular variant call is supported by 90% of the aligned reads. Deletion-aware masking of low-coverage regions: We ensure that only low-coverage positions that are not deletions are masked. Several pipelines implement a feature to mask low-coverage regions. However, deletions are basically genomic regions with no coverage, and if not appropriately implemented, a pipeline might accidentally mask deletions as low-coverage regions. To prevent this, CoVpipe2 creates a low coverage mask (default minimum coverage 20) from the BAM file with BEDtools genomecov. In the second step, BEDtools subtract removes all deleted sites from the low-coverage mask. Finally, the mask is used in the consensus generation with BCFtools consensus. IUPAC consensus generation with indel filter: CoVpipe2 generates different consensus sequences based on the IUPAC code. First, an explicit consensus is generated where all ambiguous sites and low-covered regions are hard-masked. Second, a consensus where only low-covered regions are hard-masked. The pipeline includes as much information as possible in the unambiguous consensus sequence by adding low-frequency variants with the respective IUPAC symbol. However, no symbol represents “indel or nucleotide”, so indels are always incorporated into the consensus sequence. As a result, low-frequency or heterozygote indels, often false positives, can introduce frameshifts into the consensus sequence. We overcome this with an indel filter based on allele frequency before consensus generation. Thus, indels below a defined threshold (per default 0.6) are not incorporated in the consensus sequences but can still be looked up in the VCF file. Additional features beyond genome reconstruction Over time, we added features beyond genome reconstruction to CoVpipe2 to answer newly emerging questions during the pandemic. The modular design of our implementation makes this seamless. Mixed infections and recombinants: We included LCS for raw reads. LCS was originally developed for the SARS-CoV-2 lineage decomposition of mixed samples, such as wastewater or environmental samples. 41 In amplicon-based SARS-CoV-2 sequencing, we use LCS results to indicate potential new recombinants, and possible mixed infections (co-infections with different SARS-CoV-2 lineages). Further, we added sc2rf to detect potential new recombinants via screening the consensus genome sequences. Like other tools and scripts, sc2rf depends on up-to-date lineage and mutation information, specifically, on a manually curated JSON file. We switched to another repository ( JSON GitHub project ) than the original one that includes the sc2rf scripts because of more frequent updates on the JSON file . Keeping up to date with lineage and clade assignments: We implemented an update feature for pangolin and Nextclade. Both tools, especially pangolin, rely on the latest datasets for the lineage assignment of newly designated Pango lineages. 56 Depending on the selected engine, Conda/Mamba or container execution, CoVpipe2 checks for the latest available version from Anaconda or our DockerHub , respectively. The tool versions can also be pinned manually. Prediction of mutation effects: The called variants are annotated and classified based on predicted effects on annotated genes with SnpEff. SnpEff reports different effects, including synonymous or non-synonymous SNPs, start codon gains or losses, stop codon gains or losses; and classifies them based on their genomic locations. Lastly, CoVpipe2 uses Liftoff to generate an annotation for each sample if a reference annotation is provided. Selection of benchmark datasets and pipeline evaluation We compared the results of CoVpipe2 (v0.4.0) with publicly available benchmark datasets for SARS-CoV-2 surveillance 57 ( GitHub CDC data ). For our study, we selected from the available benchmark datasets all samples that were sequenced with the ARTIC V3 primer set (according to CDC data , release v0.7.2), ending up with in total 54 samples from three datasets: • i) 16 samples from variant of interest (VOI)/VOC lineages (dataset 4) • ii) 33 samples from non-VOI/VOC lineages (dataset 5) • iii) 5 samples from failed QC (dataset 6) We stick to the naming scheme (dataset 4, 5, and 6) as declared in the original study. 57 For all 54 samples, we downloaded the raw reads from ENA with nf-core/fetchngs ( v1.9 ). 58 For 49 samples, we downloaded the available consensus sequences from GISAID 1 – 3 (samples from dataset 4 and 5). We run CoVpipe2 with Singularity, species filtering ( --kraken ) and the latest pangolin and Nextclade versions at that point (containers: rkimf1/pangolin:4.2-1.19--dec5681, rkimf1/nextclade2:2.13.1--ddb9e60 ). We further examined the consensus sequences from dataset 4 and 5, comparing CoVpipe2 and GISAID sequences: We compared lineage information of the reconstructed genomes assigned by pangolin with the indicated lineages from CDC data . Note that the pangolin versions and datasets differ. Also, we run Nextclade (same container version as noted above) to compare the mutation profile, and MAFFT, 59 v7.515 (2023/Jan/15), for comparison on sequence level of the unambiguous IUPAC consensus of CoVpipe2. Reporting The HTML report consists of several sections and summarizes different results. The first table aggregates critical features for each sample: the number of reads, genome QC, the assigned lineage, and recombination potential, see Figure 3 . Furthermore, read properties such as the number of bases and the length of reads (before/after trimming) are summarized. An optional table lists the species filtering results emitted by Kraken 2. 43 The report summarizes reads mapped to the reference genome (number and fraction of input) and the median/standard deviation of fragment sizes, shown as a histogram plot. Genome-wide coverage plots allow users to observe low-coverage regions and potential amplicon drop-outs to optimize primers. We summarize the output for all samples in different tables: i) genome quality from PRESIDENT output, ii) lineage assignments from pangolin output, and iii) variants in amino acid coordinates and detected frameshifts from Nextclade output. Figure 3. Extract of CoVpipe2’s summary report for dataset 4. The standalone HTML report summarizes different quality measures and tool results. The overview table can include a conditional notification for negative controls with high reference genome coverage (not shown in this example). All tables are searchable and sortable. Discussion CovPipe2 reconstructs consensus genomes matching previously reported SARS-CoV-2 lineages Here, we compare the results of CoVpipe2 against a selection of available benchmark datasets 57 and their respective consensus genome sequences available from GISAID. 1 – 3 We discuss the observed differences. No software is perfect, and CoVpipe2 may have problems with certain combinations of amplicon schemes and sequencing designs, leading to specific borderline cases in variant detection, which we also highlight and discuss. In addition, in conflicting cases, the real sequence often stays unknown until further sequencing efforts are performed. This makes continuous development and testing of bioinformatics pipelines all the more critical. Dataset 4, VOI/VOC lineages Lineages Despite the different pangolin tool versions and lineage definitions that changed over time, all pangolin lineages from CoVpipe2 match the corresponding lineages reported at github.com/CDCgov/datasets-sars-cov-2 , see Extended data , Table S1 . Pairwise alignment The sequence identity ranges from 98.44 % to 99.79 % (including Ns) between the corresponding GISAID and CoVpipe2 genome pairs ( Extended data , Table S2 ). Nine out of 16 sequences are identical when mismatches resulting from gaps are not considered. For one sample, the corresponding GISAID and CoVpipe2 sequences contain three ACGT-nucleotide mismatches. Seven out of 16 GISAID consensus sequences do not contain any Ns, possibly indicating that low coverage regions have been not masked. The respective reconstructed genomes from CoVpipe2 contain Ns located in the first or last 150 nucleotides of the genome sequences, thus masking low coverage regions ( Extended data , Table S3 ). Due to tiled PCR amplicons, 5′ and 3′ ends of the genome usually have too little coverage or are not sequenced. The genome ends containing Ns seem to be trimmed in 14 GISAID genomes. Overall, the number of Ns in the GISAID and CoVpipe2 genomes is comparable, with CoVpipe2 genomes tending to have more Ns due to the low coverage filter resulting in Ns at genome ends. In addition, CoVpipe2 does not trim Ns from the genome ends and might be more conservative with its default settings, as positions below 20 X coverage ( --cns_min_cov ) will be masked with ambiguous N bases in the consensus sequence. Mutations Although all genomes reconstructed from CoVpipe2 were assigned to the same lineage as previously reported ( CDC data ), there are minor differences in Nextclade’s mutation profile, Extended data , Table S4 . Three of 16 reconstructed genomes have one or two nucleotide substitutions more than the respective GISAID genome. Dataset 5, non-VOI/VOC lineages Lineages Pangolin lineages from CoVpipe2 exactly match the reported lineages at CDC data in 27 of 33 samples, see Extended data , Table S5 . For five samples, the pangolin lineage based on the genome sequence reconstructed by Covpipe2 is the parent lineage of the expected sub-lineage. For example, the genome sequence of sample SAMN15919634 was assigned to the B.1.1 lineage after CoVpipe2 reconstruction, whereas the corresponding GISAID sequence was assigned to B.1.1.431 ( CDC data ). Since the lineage assignments of Nextclade exactly match for each CoVpipe2-GISAD pair, the different pangolin and pangolin data versions used in CoVpipe2 and Xiaoli and Hagey et al . 57 are most likely the reason for this discrepancy. Different parameter thresholds can also lead to different lineage assignments. For example, the genome sequence of SAMN17571193 was assigned to B.1.1.450 by CoVpipe2 compared to B.1.1.391 in Xiaoli and Hagey et al . 57 The mutation profile differs by two additional SNPs (genome coordinates: C3037T and C3787T) constituting one mutation on amino acid level (ORF1a:D1962E) in the GISAID consensus sequence. Both positions (3037 and 3787) have a read coverage below 20, so they are not eligible for variant calling with default settings in CoVpipe2 and are masked with N. Therefore, this difference in lineage assignment is due to CoVpipe2’s more restrictive approach to integrating the called variants into the consensus. Pairwise alignment The pairwise sequence identity ranges from 95.96 % to 99.80 % (including Ns) between the GISAID and CoVpipe2’s reconstructed genome sequences ( Extended data , Table S6 ). Ignoring gap mismatches, 13 out of 33 sequences are identical. No sample contains ACGT mismatches. 18 out of 33 GISAID consensus sequences do not contain any Ns. Sixteen of the respective reconstructed genomes have Ns, all located in the first or last 150 nucleotides ( Extended data , Table S7 ). Because of tiled PCR amplicons, the 5′ and 3′ ends typically have too little coverage or are not sequenced. The 5′ and 3′ genome ends containing Ns seem trimmed in 30 GISAID genomes. Overall, the number of Ns in the GISAID and CoVpipe2 genomes is comparable, with CoVpipe2 genomes tending to have more Ns. CoVpipe2 does not trim Ns from the genome ends and might be more conservative with its default settings, as positions below 20 ( --cns_min_cov ) will be masked with N in the consensus sequence. Mutations There are minor differences in Nextclade’s mutation profile ( Extended data , Table S8 ). Five of the 33 reconstructed genomes have one or two more nucleotide substitutions compared to the GISAID genome. Dataset 6, samples failing quality control For the five samples from the failed QC dataset by Xiaoli and Hagey et al ., 57 CoVpipe2 correctly labeled four samples with failed genome QC. Three samples contain at least one frameshift, whereas two samples, SAMN17486862 and SAMN17822806, were reconstructed by CoVpipe2 without frameshifts ( Extended data , Table S9 ). The consensus sequence of sample SAMN17486862 passed QC according to CoVpipe2’s genome QC criteria. All the selected samples from this dataset have originally failed QC because of a VADR 60 alert number greater than one. 57 VADR is part of the TheiaCoV (formerly ‘Titan’) 1.4.4 pipeline, 61 which was used to analyze the samples in Xiaoli and Hagey et al . 57 Among other things, VADR considers frameshifts, which do not occur in two genomes reconstructed with CoVpipe2. However, one of these two consensus genomes (SAMN17822806) contains too many Ns to pass CoVpipe2’s genome QC. Limitations The nature of a computational pipeline is that it is a chain of existing individual tools. Especially given the rapid evolution of SARS-CoV-2 during the pandemic, many reference-based tools rely on up-to-date databases and resources to reflect the current situation. For example, LCS depends on a variant marker table and user-defined variant groups. Similarly, sc2rf relies on a list of common variants for each lineage. In addition, Nextclade and pangolin periodically publish up-to-date datasets. Therefore, it is critical for a pipeline, especially in the context of a surveillance tool for rapidly evolving pathogens such as SARS-CoV-2, to allow for regular updates to the underlying data structures. While we have implemented some functionality to update tools such as Nextclade and pangolin automatically, this is not possible for all resources and can only be achieved through the continuous development and maintenance of a pipeline. Furthermore, our default settings may not fit all input data and must be selected carefully. Finally, there must be enough good-quality input reads to reconstruct a genome successfully. In particular, with amplicon sequencing data, some regions might have a lower coverage due to amplicon drop-outs. Thus, the genome as a whole can be reconstructed with acceptable quality. However, some essential mutations can be missing due to low coverage or variant-calling quality. Conclusions Accurate and high-throughput genotyping and genome reconstruction methods are central for monitoring SARS-CoV-2 transmission and evolution. CoVpipe2 provides a fully automated, flexible, modular, and reproducible workflow for reference-based variant calling and genome reconstruction from short-read sequencing data, emphasizing amplicon-based sequencing schemes. Due to the implementation in the Nextflow framework, the setup and automatic installation of the required tools and dependencies is simple and allows the execution on different computing platforms. The comparison with a benchmark dataset showed comparable results where differences could be pinned down to different parameters, filtering thresholds, and tool versions used for lineage assignments. Amplicon-optimized default parameters and the ability to customize critical parameters, combined with comprehensive reporting, ensure the quality of reported SARS-CoV-2 genomes and prevent the inclusion of low-quality sequences in downstream analyses and public repositories. The pipeline is optimized for SARS-CoV-2 amplicon data but can also be used for other viruses and whole-genome sequencing protocols. We used CoVpipe1 and CoVpipe2, which were initially developed for SARS-CoV-2, to reconstruct the genomes of other viruses. By selecting different reference genomes for polio, measles, RSV, and influenza viruses and modifying the analysis processes for the latter two viruses, we were able to demonstrate the workflow’s potential as a universal blueprint for viral genome analysis. This adaptability has been demonstrated by the ability to identify consistent variants and the successful application to different viral characteristics, such as varying genome size and segmentation. However, it should also be noted that for segmented viruses, more customization is required than simply replacing the (non-segmented) reference genome. However, our experience suggests that customization of tools, parameters, and reporting requirements is needed for each virus, pointing to developing a harmonized pipeline for multiple pathogens. CoVpipe2 thus proves to be a robust tool for SARS-CoV-2 and paves the way for broad application in virology research by highlighting its ability to serve as a fundamental framework for building customized pipelines for a wide range of pathogens. CoVpipe2 will therefore form the basis for further genomic surveillance programs at the German Institute of Public Health, which will also extend to other viruses. Data availability Underlying data All SRA accession IDs of raw reads and GISAID IDs of consensus sequences are listed at 57 and github.com/CDCgov/datasets-sars-cov-2 ; and in our Open Science Framework repository osf.io/26hyx ( Extended data : https://doi.org/10.17605/OSF.IO/MJ6EQ ). 62 We used ARTIC V3 samples from dataset 4 (VOI/VOC lineages, 16 samples), dataset 5 (non-VOI/VOC lineages, 33 samples), and dataset 6 (failedQC, 5 samples) from the original study of Xiaoli and Hagey et al . 57 The precalculated Kraken 2 database composed of SARS-CoV-2 and human genomes is available from Zenodo, https://doi.org/10.5281/zenodo.6333909 . 63 Extended data Open Science Framework. Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2, https://doi.org/10.17605/OSF.IO/MJ6EQ . 62 This project contains the following extended data: • Data folder Accession ID. (SRA and GISAID accession ID lists) • Data folder Comparison. (Scripts and results of the benchmark comparison) • Data folder CoVpipe2 results. (Results of CoVpipe2 for each benchmark dataset) • Data folder Extended data. (Supplementary tables S1-S9) Software availability • Software and source code available from: https://github.com/rki-mf1/covpipe2 • Archived source code at time of publication: https://doi.org/10.5281/zenodo.8082695 . 64 • License: GNU General Public License v3.0 (GPL3) Acknowledgements We gratefully acknowledge all data contributors, i.e., the Authors and their originating laboratories responsible for obtaining the specimens, and their submitting laboratories for generating the genetic sequence and metadata and sharing via the GISAID Initiative, on which this research is based. We also thank all German Electronic Sequence Data Hub contributors, the IMS-SC2 laboratory network, and all data providers, i.e., the originating laboratories responsible for obtaining samples and the submitting laboratories where genetic sequence data were generated and shared. We are incredibly grateful to Petra Kurzendörfer, Tanja Pilz, Aleksandar Radonić, and Aaron Houterman for outstanding sequencing support. We thank Marianne Wedde for manually checking and validating consensus sequences and called variants and Thorsten Wolff and Max von Kleist for fruitful discussions. We thank Ben Wulf, Fabian Rost, and Alexander Seitz – early users of the first version of CoVpipe – for valuable feedback and reporting issues. References 1. Shu Y, McCauley J: GISAID: Global initiative on sharing all influenza data – from vision to reality. Eurosurveillance. 2017; 22 (13): 30494. 2. Elbe S, Buckland-Merrett G: Data, disease and diplomacy: Gisaid’s innovative contribution to global health. Global Chall. 2017; 1 (1): 33–46. PubMed Abstract | Publisher Full Text | Free Full Text 3. Khare S, Gurry C, Freitas L, et al. : GISAID Core Curation Team, and Sebastian Maurer-Stroh. Gisaid’s role in pandemic response.2021. 2096-7071. Reference Source 4. Harrison PW, Lopez R, Rahman N, et al. : The COVID-19 Data Portal: accelerating SARS-CoV-2 and COVID-19 research through rapid open access data sharing. Nucleic Acids Res. 2021; 49 (W1): W619–W623. PubMed Abstract | Publisher Full Text | Free Full Text 5. Robishaw JD, Alter SM, Solano JJ, et al. : Genomic surveillance to combat COVID-19: challenges and opportunities. Lancet Microbe. 2021; 2 (9): e481–e484. PubMed Abstract | Publisher Full Text | Free Full Text 6. Julianna LeMieux Genetic Engineering & Biotechnology News: All Aboard the Genome Express: Is a new generation of DNA sequencing technology about to hit the fast track? last accessed December 01, 2022. Reference Source 7. Hufsky F, Lamkiewicz K, Almeida A, et al. : Computational strategies to combat COVID-19: useful tools to accelerate SARS-CoV-2 and coronavirus research. Brief. Bioinform. 2021; 22 (2): 642–663. PubMed Abstract | Publisher Full Text | Free Full Text 8. Djin Ye O, Hölzer M, Paraskevopoulou S, et al. : Advancing precision vaccinology by molecular and genomic surveillance of Severe Acute Respiratory Syndrome Coronavirus 2 in Germany, 2021. Clin. Infect. Dis. 2022; 75 (Supplement_1): S110–S120. Publisher Full Text 9. Baumgarte S, Hartkopf F, Hölzer M, et al. : Investigation of a limited but explosive COVID-19 outbreak in a German secondary school. Viruses. 2022; 14 (1): 87. PubMed Abstract | Publisher Full Text | Free Full Text 10. Loss J, Wurm J, Varnaccia G, et al. : Transmission of sars-cov-2 among children and staff in german daycare centres. Epidemiol. Infect. 2022; 150 : e141. PubMed Abstract | Publisher Full Text | Free Full Text 11. De Maio N, Walker C, Borges R, et al. : Issues with SARS-CoV-2 sequencing data. last accessed November 25, 2022. Reference Source 12. Brito AF, Semenova E, Dudas G, et al. : Global disparities in SARS-CoV-2 genomic surveillance. Nat. Commun. 2022; 13 (1): 1–13. 13. Connor R, Yarmosh DA, Maier W, et al. : Towards increased accuracy and reproducibility in SARS-CoV-2 next generation sequence analysis for public health surveillance. bioRxiv. 2022. 14. Chiara M, D’Erchia AM, Gissi C, et al. : Next generation sequencing of SARS-CoV-2 genomes: challenges, applications and opportunities. Brief. Bioinform. 2021; 22 (2): 616–630. PubMed Abstract | Publisher Full Text | Free Full Text 15. Robert Koch Institute: Deutscher Elektronischer Sequenzdaten-Hub (DESH). last accessed November 25, 2022. Reference Source 16. Grubaugh ND, Gangavarapu K, Quick J, et al. : An amplicon-based sequencing framework for accurately measuring intrahost virus diversity using PrimalSeq and iVar. Genome Biol. 2019; 20 (1): 1–19. Publisher Full Text 17. Resende PC, Motta FC, Roy S, et al. : SARS-CoV-2 genomes recovered by long amplicon tiling multiplex approach using nanopore sequencing and applicable to other sequencing platforms. BioRxiv. 2020. 18. Brinkmann A, Ulm S-L, Uddin S, et al. : Amplicov: Rapid whole-genome sequencing using multiplex PCR amplification and real-time Oxford Nanopore MinION sequencing enables rapid variant identification of SARS-CoV-2. Front. Microbiol. 2021; 12 : 1703. Publisher Full Text 19. Hilaire BGS, Durand NC, Mitra N, et al. : A rapid, low cost, and highly sensitive SARS-CoV-2 diagnostic based on whole genome sequencing. BioRxiv. 2020. 20. Gohl DM, Garbe J, Grady P, et al. : A rapid, cost-effective tailed amplicon method for sequencing SARS-CoV-2. BMC Genomics. 2020; 21 (1): 1–10. Publisher Full Text 21. Hadfield J, Megill C, Bell SM, et al. : Nextstrain: real-time tracking of pathogen evolution. Bioinformatics. 2018; 34 (23): 4121–4123. PubMed Abstract | Publisher Full Text | Free Full Text 22. Tshiabuila D, Giandhari J, Pillay S, et al. : Comparison of SARS-CoV-2 sequencing using the ONT GridION and the Illumina MiSeq. BMC Genomics. 2022; 23 (1): 1–17. Publisher Full Text 23. Luo J, Meng Z, Xingyu X, et al. : Systematic benchmarking of nanopore Q20+ kit in SARS-CoV-2 whole genome sequencing. Front. Microbiol. 2022; 4059. 24. Brandt C, Krautwurst S, Spott R, et al. : poreCov – an easy to use, fast, and robust workflow for SARS-CoV-2 genome reconstruction via nanopore sequencing. Front. Genet. 2021; 1397. 25. Di Tommaso P, Chatzou M, Floden EW, et al. : Nextflow enables reproducible computational workflows. Nat. Biotechnol. 2017; 35 (4): 316–319. PubMed Abstract | Publisher Full Text 26. Köster J, Rahmann S: Snakemake – a scalable bioinformatics workflow engine. Bioinformatics. 2012; 28 (19): 2520–2522. PubMed Abstract | Publisher Full Text 27. Patel H, Monzón S, Varona S, et al. : nf-core/viralrecon: nf-core/viralrecon v2.6.0 - Rhodium Raccoon.March 2023. Publisher Full Text 28. Nasir JA, Kozak RA, Aftanas P, et al. : A comparison of whole genome sequencing of SARS-CoV-2 using amplicon-based sequencing, random hexamers, and bait capture. Viruses. 2020; 12 (8): 895. PubMed Abstract | Publisher Full Text | Free Full Text 29. Posada-Céspedes S, Seifert D, Topolsky I, et al. : V-pipe: a computational pipeline for assessing viral genetic diversity from high-throughput data. Bioinformatics. 2021; 37 (12): 1673–1680. PubMed Abstract | Publisher Full Text | Free Full Text 30. Kim K, Park K, Lee S, et al. : Virpipe: an easy and robust pipeline for detecting customized viral genomes obtained by nanopore sequencing. Bioinformatics. 2023; 39 : btad293. PubMed Abstract | Publisher Full Text | Free Full Text 31. Dezordi FZ, da Silva Neto AM , de Lima Campos T , et al. : Viralflow: a versatile automated workflow for sars-cov-2 genome assembly, lineage assignment, mutations and intrahost variant detection. Viruses. 2022; 14 (2): 217. PubMed Abstract | Publisher Full Text | Free Full Text 32. Lo C-C, Shakya M, Connor R, et al. : EDGE COVID-19: a web platform to generate submission-ready genomes from SARS-CoV-2 sequencing efforts. Bioinformatics. 2022; 38 (10): 2700–2704. PubMed Abstract | Publisher Full Text | Free Full Text 33. Maier W, Bray S, van den Beek M , et al. : Ready-to-use public infrastructure for global SARS-CoV-2 monitoring. Nat. Biotechnol. 2021; 39 (10): 1178–1179. PubMed Abstract | Publisher Full Text | Free Full Text 34. Nguyen PTT, Plyusnin I, Sironen T, et al. : HAVoC, a bioinformatic pipeline for reference-based consensus assembly and lineage assignment for SARS-CoV-2 sequences. BMC Bioinformat. 2021; 22 (1): 1–8. Publisher Full Text 35. Oliveira RRM, Negri TC, Nunes G, et al. : PipeCov: a pipeline for SARS-CoV-2 genome assembly, annotation and variant identification. PeerJ. 2022; 10 : e13300. PubMed Abstract | Publisher Full Text | Free Full Text 36. Grüning B, Dale R, Sjödin A, et al. : Bioconda: sustainable and comprehensive software distribution for the life sciences. Nat. Methods. 2018; 15 (7): 475–476. Publisher Full Text 37. Boettiger C: An introduction to Docker for reproducible research. Oper. Syst. Rev. 2015; 49 (1): 71–79. Publisher Full Text 38. Kurtzer GM, Sochat V, Bauer MW: Singularity: Scientific containers for mobility of compute. PLoS One. 2017; 12 (5): e0177459. PubMed Abstract | Publisher Full Text | Free Full Text 39. Aksamentov I, Roemer C, Hodcroft EB, et al. : Nextclade: clade assignment, mutation calling and quality control for viral genomes. J. Open Source Softw. 2021; 6 (67): 3773. Publisher Full Text 40. O’Toole Á, Scher E, Underwood A, et al. : Assignment of epidemiological lineages in an emerging pandemic using the pangolin tool. Virus Evol. 2021; 7 (2): veab064. PubMed Abstract | Publisher Full Text | Free Full Text 41. Valieris R, Drummond RD, Defelicibus A, et al. : A mixture model for determining SARS-Cov-2 variant composition in pooled samples. Bioinformatics. 2022; 38 (7): 1809–1815. PubMed Abstract | Publisher Full Text 42. Chen S, Zhou Y, Chen Y, et al. : fastp: an ultra-fast all-in-one FASTQ preprocessor. Bioinformatics. 2018; 34 (17): i884–i890. PubMed Abstract | Publisher Full Text | Free Full Text 43. Wood DE, Jennifer L, Langmead B: Improved metagenomic analysis with Kraken 2. Genome Biol. 2019; 20 : 1–13. Publisher Full Text 44. Li H: Aligning sequence reads, clone sequences and assembly contigs with bwa-mem.2013. 45. Quinlan AR: BEDTools: the Swiss-army tool for genome feature analysis. Curr. Protoc. Bioinformat. 2014; 47 (1): 11–12. 46. Chun Hang A, Ho DN, Kwong A, et al. : BAMClipper: removing primers from alignments to minimize false-negative mutations in amplicon next-generation sequencing. Sci. Rep. 2017; 7 (1): 1–7. 47. Tyson JR, James P, Stoddart D, et al. : Improvements to the ARTIC multiplex PCR method for SARS-CoV-2 genome sequencing using nanopore. BioRxiv. 2020. 48. Garrison E, Marth G: Haplotype-based variant detection from short-read sequencing. arXiv preprint arXiv:1207.3907. 2012. 49. Danecek P, Bonfield JK, Liddle J, et al. : Twelve years of SAMtools and BCFtools. Gigascience. 2021; 10 (2): giab008. PubMed Abstract | Publisher Full Text | Free Full Text 50. Cingolani P, Platts A, Coon M, et al. : A program for annotating and predicting the effects of single nucleotide polymorphisms, SnpEff: SNPs in the genome of Drosophila melanogaster strain w1118; iso-2; iso-3. Fly. 2012; 6 (2): 80–92. PubMed Abstract | Publisher Full Text | Free Full Text 51. Cornish-Bowden A: Nomenclature for incompletely specified bases in nucleic acid sequences: recommendations 1984. Nucleic Acids Res. 1985; 13 (9): 3021–3030. PubMed Abstract | Publisher Full Text | Free Full Text 52. Wang M, Kong L: pblat: a multithread blat algorithm speeding up aligning sequences to genomes. BMC Bioinformat. 2019; 20 (1): 1–4. 53. Shumate A, Salzberg SL: Liftoff: accurate mapping of gene annotations. Bioinformatics. 2021; 37 (12): 1639–1643. PubMed Abstract | Publisher Full Text | Free Full Text 54. Kubik S, Marques AC, Xing X, et al. : Recommendations for accurate genotyping of SARS-CoV-2 using amplicon-based sequencing of clinical samples. Clin. Microbiol. Infect. 2021; 27 (7): 1036.e1–1036.e8. Publisher Full Text 55. Satya RV, DiCarlo J: Edge effects in calling variants from targeted amplicon sequencing. BMC Genomics. 2014; 15 (1): 1073–1077. Publisher Full Text 56. Rambaut A, Holmes EC, O’Toole Á, et al. : A dynamic nomenclature proposal for sars-cov-2 lineages to assist genomic epidemiology. Nat. Microbiol. 2020; 5 (11): 1403–1407. PubMed Abstract | Publisher Full Text | Free Full Text 57. Xiaoli L, Hagey JV, Park DJ, et al. : Benchmark datasets for sars-cov-2 surveillance bioinformatics. PeerJ. 2022; 10 : e13821. Publisher Full Text 58. Ewels PA, Peltzer A, Fillinger S, et al. : The nf-core framework for community-curated bioinformatics pipelines. Nat. Biotechnol. 2020; 38 (3): 276–278. PubMed Abstract | Publisher Full Text 59. Katoh K, Standley DM: MAFFT multiple sequence alignment software version 7: improvements in performance and usability. Mol. Biol. Evol. 2013; 30 (4): 772–780. PubMed Abstract | Publisher Full Text | Free Full Text 60. Schäffer AA, Hatcher EL, Yankie L, et al. : Vadr: validation and annotation of virus sequence submissions to genbank. BMC Bioinformat. 2020; 21 : 1–23. 61. Libuit K, RA P III, Ambrosio F, et al. : Public health viral genomics: bioinformatics workflows for genomic characterization, submission preparation, and genomic epidemiology of viral pathogens, especially the sars-cov-2 virus.2022. 62. Lataretu M, Hölzer M: Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2. [Dataset]. 2023, June 27. Publisher Full Text 63. Fuchs S, Drechsel O: Kraken 2 Database (Human, SARS-CoV2) (3.0.0). [Data set]. Zenodo. 2022. Publisher Full Text 64. MarieLataretu: rki-mf1/CoVpipe2: Version v0.4.3 (v0.4.3). Zenodo. 2023. Publisher Full Text Comments on this article Comments (0) Version 2 VERSION 2 PUBLISHED 01 Sep 2023 ADD YOUR COMMENT Comment Author details Author details 1 Genome Competence Center (MF1), Robert Koch Institute, Berlin, 13353, Germany Marie Lataretu Roles: Data Curation, Formal Analysis, Investigation, Methodology, Software, Validation, Visualization, Writing – Original Draft Preparation Oliver Drechsel Roles: Methodology, Software, Writing – Review & Editing René Kmiecinski Roles: Methodology, Software Kathrin Trappe Roles: Software, Writing – Review & Editing Martin Hölzer Roles: Formal Analysis, Investigation, Methodology, Supervision, Validation, Writing – Original Draft Preparation Stephan Fuchs Roles: Conceptualization, Funding Acquisition, Methodology, Project Administration, Supervision, Writing – Review & Editing Competing interests No competing interests were disclosed. Grant information .L. was supported by the European Centre for Disease Control (grant number ECDC GRANT/2021/008 ECD.12222). This work was further supported by the European Health and Digital Executive Agency (grant number 101113012) and Bundesministerium für Wirtschaft und Klimaschutz, Daten- und KI-gestütztes Frühwarnsystem zur Stabilisierung der deutschen Wirtschaft (grant number 01MK21009H). The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript. Article Versions (2) version 2 Revised Published: 16 Apr 2024, 12:1091 https://doi.org/10.12688/f1000research.136683.2 version 1 Published: 01 Sep 2023, 12:1091 https://doi.org/10.12688/f1000research.136683.1 Copyright © 2024 Lataretu M et al . This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. Download Export To Sciwheel Bibtex EndNote ProCite Ref. Manager (RIS) Sente metrics Views Downloads F1000Research - - PubMed Central info_outline Data from PMC are received and updated monthly. - - Citations open_in_new 0 open_in_new 0 open_in_new SEE MORE DETAILS CITE how to cite this article Lataretu M, Drechsel O, Kmiecinski R et al. Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.12688/f1000research.136683.2 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS track receive updates on this article Track an article to receive email alerts on any updates to this article. TRACK THIS ARTICLE Share Open Peer Review Current Reviewer Status: ? Key to Reviewer Statuses VIEW HIDE Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Version 2 VERSION 2 PUBLISHED 16 Apr 2024 Revised Views 0 Cite How to cite this report: Haddad-Boubaker S. Reviewer Report For: Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.5256/f1000research.164888.r266843 ) The direct URL for this report is: https://f1000research.com/articles/12-1091/v2#referee-response-266843 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 06 May 2024 Sondes Haddad-Boubaker , University of Tunis El Manar, Tunis, Tunisia Approved VIEWS 0 https://doi.org/10.5256/f1000research.164888.r266843 The revision made is appropriate thus I approve of the paper ... Continue reading READ ALL The revision made is appropriate thus I approve of the paper in the current form. Thank you for publishing and sharing results. Competing Interests: No competing interests were disclosed. Reviewer Expertise: Virology I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Haddad-Boubaker S. Reviewer Report For: Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.5256/f1000research.164888.r266843 ) The direct URL for this report is: https://f1000research.com/articles/12-1091/v2#referee-response-266843 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Version 1 VERSION 1 PUBLISHED 01 Sep 2023 Views 0 Cite How to cite this report: Haddad-Boubaker S. Reviewer Report For: Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.5256/f1000research.149827.r233668 ) The direct URL for this report is: https://f1000research.com/articles/12-1091/v1#referee-response-233668 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 17 Jan 2024 Sondes Haddad-Boubaker , University of Tunis El Manar, Tunis, Tunisia Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.149827.r233668 This paper presents a comprehensive bioinformatics workflow designed for the reconstruction of SARS-CoV-2 genomes using short-read sequencing data. The workflow description offers a detailed overview of the processes involved; however, there are opportunities for improvement to enhance the overall quality ... Continue reading READ ALL This paper presents a comprehensive bioinformatics workflow designed for the reconstruction of SARS-CoV-2 genomes using short-read sequencing data. The workflow description offers a detailed overview of the processes involved; however, there are opportunities for improvement to enhance the overall quality of the paper. 1-Organization: While the overall structure is clear, introducing numbering at both the heading and sub-heading levels would enhance clarity and facilitate better distinction between various steps in the workflow. 2-Technical Terminology: Given the diverse audience, including virologists and other biologists, it is crucial to ensure accessibility by providing clear definitions for all technical terms and acronyms. This will make the paper more easy to readers who may not be familiar with specific bioinformatics or genomics terminology. 3- References and Sources: Include proper references and sources for the tools and databases mentioned by providing specific URLs for easy access to external resources. 4-Methods and Results: - Introduce a dedicated "Pipeline Evaluation" subsection in both the "Methods" and "Results" sections and change the abstract section accordingly. -Include details on the investigated sequences, with a thorough description of sample types, cycle threshold (ct) values, and sample categories. -Consider expanding the evaluation by incorporating additional samples/sequences, especially failed sequences from samples with varying ct values (especially high ct values). The Investigation of the contribution of the pipeline in obtaining reliable sequences from samples with high ct values may be helpfull for virologist who are dealing with such challenges especially in samples obtained from long-term excretors. - Please Illustrate "common challenges" with a graph or figure for better presentation and understanding. 5- Discussion: - Emphasize and discuss the added value of the pipeline in obtaining reliable sequences from samples with high ct values. Compare the results obtained using this pipeline with those from other existing pipelines to highlight its superiority. - Provide detailed insights into how this pipeline can be applied to study other viruses, Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Partly Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Partly Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Partly Competing Interests: No competing interests were disclosed. Reviewer Expertise: Virology I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Haddad-Boubaker S. Reviewer Report For: Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.5256/f1000research.149827.r233668 ) The direct URL for this report is: https://f1000research.com/articles/12-1091/v1#referee-response-233668 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Author Response 15 May 2024 Martin Hölzer , Genome Competence Center (MF1), Robert Koch Institute, Berlin, Germany 15 May 2024 Author Response (I) - Organization: [1] Reviewer Concern: While the overall structure is clear, introducing numbering at both the heading and sub-heading levels would enhance clarity and facilitate better distinction between various steps ... Continue reading (I) - Organization: [1] Reviewer Concern: While the overall structure is clear, introducing numbering at both the heading and sub-heading levels would enhance clarity and facilitate better distinction between various steps in the workflow. Author Response: Thanks for the suggestion. Numbering (sub)sections helps to structure the text better and distinguish which parts belong together semantically. However, there is little we can do about it, as this is the journal's style. Nevertheless, we will ask the editor/typesetting team if that’s possible. -------------------------------------------------------------------------------------------------------------------- (II) - Technical Terminology: [2] Reviewer Concern: Given the diverse audience, including virologists and other biologists, it is crucial to ensure accessibility by providing clear definitions for all technical terms and acronyms. This will make the paper more easy to readers who may not be familiar with specific bioinformatics or genomics terminology. Author Response: Thanks for the comment. We agreed and added a list of Abbreviations to the manuscript to make it easier for readers to follow the story. COVID-19 - Coronavirus disease 2019 SARS-CoV-2 - Severe acute respiratory syndrome coronavirus 2 GPL3 license - GNU General Public License GISAID - Global Initiative on Sharing All Influenza Data EBI - European Bioinformatics Institute EMBL - European Molecular Biology Laboratory RKI - Robert Koch Institute CorSurV - Coronavirus Surveillance Verordnung (eng., Coronavirus Surveillance Regulation) DESH - Deutscher Elektronischer Sequenzdaten-Hub (eng., German Electronic Sequence Data Hub) IMS-SC2 - Integrated Molecular Surveillance for SARS-CoV-2 ONT - Oxford Nanopore Technologies NGS - Next-Generation Sequencing HPC - High-Performance Computing WSL - Windows Subsystem for Linux CSV file - Comma-Separated Values file GFF file - General Feature Format file BEDPE file - Browser Extensible Data Paired-End file VCF file - Variant Call Format file HTML - Hypertext Markup Language BAM file - Binary Alignment and Map file BED file - Browser Extensible Data file CCO license - Creative Commons Zero license VOC - Variants of Concern VOI - Variants of Interest IUPAC - International Union of Pure and Applied Chemistry indel - Insertion/Deletion Variant JSON file - JavaScript Object Notation file CDC - Centers for Disease Control and Prevention ENA - European Nucleotide Archive QC - Quality Control PCR - Polymerase Chain Reaction ID - Identifier SRA - Sequence Read Archive -------------------------------------------------------------------------------------------------------------------- (III) - References and Sources [3] Reviewer Concern: Include proper references and sources for the tools and databases mentioned by providing specific URLs for easy access to external resources. Author Response: We agree that it’s crucial to acknowledge all tools and resources properly. When there is an original publication for a tool or database, we cite the publication. If not, we cite the code repository or the URL to the resource (such as GitHub or Zenodo). We carefully checked the text again and added citations/URLs if they were missing and necessary. For example, Rev #1 also commented that the specific URL to the custom Kraken 2 database on Zenodo was missing. We added that. -------------------------------------------------------------------------------------------------------------------- (IV) - Methods and Results [4] Reviewer Concern : Introduce a dedicated "Pipeline Evaluation" subsection in both the "Methods" and "Results" sections and change the abstract section accordingly. Author Response: We fully agree that presenting the implementation of a pipeline together with its evaluation with additional analysis can be confusing. So what are the "Methods" and the "Results" parts, then? This is often a problem when simultaneously presenting and evaluating a new software implementation. But we also need to adhere to the style guidelines for journals. We wrote a “Software Tool Articles” and have to follow this structure: https://f1000research.com/for-authors/article-guidelines/software-tool-articles . As you can see in these guidelines, “Software Tool Articles typically contain the following sections: Introduction, Methods, Results (Optional), Use Cases (Optional), Conclusions/Discussion.” In the first version, we skipped the “Use Cases” section to discuss our example data sets for pipeline evaluation directly in the “Results” section. However, thanks to your comment, we believe that also skipping the “Results” section entirely makes our manuscript clearer. We deleted the “Results” section and added the subsections “Selection of benchmark datasets and pipeline evaluation” and “Reporting” at the end of the “Methods”. We changed the subsection “Selection of benchmark datasets” to “Selection of benchmark datasets and pipeline evaluation” to include your suggestion. We think that the structure is now clearer because we first describe in the “Methods” the implementation of the pipeline and how to operate it, according to the journal guidelines for “Software Tool Articles”: The Methods should “Include a subsection on Implementation describing how the tool works and any relevant technical details required for implementation; and a subsection on Operation , which should include the minimal system requirements needed to run the software and an overview of the workflow.” Then, we describe some specific implementation decisions followed by the example data sets for pipeline evaluation and, finally, the report structure we implemented. According to the journal guideline for “Software Tool Articles”: “Abstracts are structured into Background, Methods, Results, and Conclusions”, thus, we can not change sections in the abstract. Although we generally agree that different subheadings would help follow the story, we must also stick to the journal guidelines. -------------------------------------------------------------------------------------------------------------------- [5] Reviewer Concern : Include details on the investigated sequences, with a thorough description of sample types, cycle threshold (ct) values, and sample categories. Author Response: We only use publicly available data sets and reference the original sources (publication, GitHub, and ENA repositories). We suggest that readers should refer to the original sources for further details. However, the description of the benchmark datasets is now also part of the “Methods”. Here, we describe: “We compared the results of CoVpipe2 (v0.4.0) with publicly available benchmark datasets for SARS-CoV-2 surveillance 57 ( GitHub CDC data ).” [57] Xiaoli L, Hagey JV, Park DJ, et al.: Benchmark datasets for sars-cov-2 surveillance bioinformatics. PeerJ. 2022;10:e13821. 10.7717/peerj.13821 We do not want to mirror the details of the benchmark data sets that are described in the original sources. In addition, we can not provide additional information, such as Ct values, because, to the best of our knowledge, this information is not available in the original publication or in the data source (GitHub, ENA). -------------------------------------------------------------------------------------------------------------------- [6] Reviewer Concern : Consider expanding the evaluation by incorporating additional samples/sequences, especially failed sequences from samples with varying ct values (especially high ct values). The Investigation of the contribution of the pipeline in obtaining reliable sequences from samples with high ct values may be helpful for virologist who are dealing with such challenges especially in samples obtained from long-term excretors. Author Response: Thanks for the comment. We agree that investigating challenging samples is especially interesting for users of CoVpipe2. As described above (Q [5]), we selected a publicly available benchmark dataset for SARS-CoV-2 surveillance (Xiaoli et al. 2022) to compare our results directly with previous calculations. In addition, this data set also includes difficult samples that should not withstand automatic quality control (QC) and could mimic high Ct values. CoVpipe2 was developed as a robust and standardized workflow to support genome reconstruction in genomic surveillance programs. Thus, our main goal in developing CoVpipe2 was to provide a robust bioinformatics pipeline for short-read sequencing data that recognizes important mutations with decent allele frequency and automatically identifies and masks ambiguous positions. We implemented parameters (20X coverage to consider a position for variant calling, 90% ACGT nucleotide identity to the reference) to discover low-quality samples that might originate from high Ct values. Running the pipeline on amplicon sequencing data from samples with high Ct can result in "read stacks" with high sequence depth for certain well-amplified amplicons, but it could also lead to low horizontal genome coverage due to low input RNA quantity. CoVpipe2 will report such samples as “failed” in the QC report. Thus, only samples with a decent vertical (sequencing depth) and horizontal genome coverage should be used for downstream genomic surveillance and trustworthy lineage assignment. Nevertheless, CoVpipe2 also reports the full intermediate results, such as BAM files and unfiltered VCF files. Experienced users can investigate all variant calls and their respective allele frequencies - also for QC-failed samples. Thus, it is also possible to investigate mixed variant calls (co-infection, recombinants) and low-frequency variants with the help of CoVpipe2. However, for routine genomic surveillance applications, such challenging samples will be automatically flagged as QC-failed in the pipeline, supporting non-expert users in decision-making and selecting suitable samples for surveillance. Obtaining reliable consensus genomes from high Ct samples is generally difficult. In our experience, it is better to flag such samples with a warning and inform the user that those are of lower quality and probably not suited for further downstream analysis. Thus, CoVpipe2 helps virologists identify such challenging samples so that they can be selected for re-sequencing or exclusion from downstream analysis. -------------------------------------------------------------------------------------------------------------------- [7] Reviewer Concern : Please Illustrate "common challenges" with a graph or figure for better presentation and understanding. Author Response: Here, we present a bioinformatics pipeline with the specific objective of reconstructing robust SARS-CoV-2 consensus genomes from patient samples and short-read data, which is also reflected in the title of our paper. We present CoVpipe2 as a solution to overcome such challenges in reconstructing robust SARS-CoV-2 genomes from short-read (amplicon) data. In Figure 2, we explicitly illustrate common challenges regarding variant calling, which is one of the main obstacles in many reference-based virus bioinformatics pipelines, and where we specifically integrated solutions in CoVpipe2 to overcome such challenges. Besides the dedicated figure for variant calling challenges, we examine other relevant challenges in the context of the CoVpipe2 implementation, such as amplicon drop-outs, in the text. For a general overview, we think that common challenges in the context of amplicon sequencing and virus bioinformatics need to be more broadly addressed in dedicated benchmark studies such as those already available from Beerenwinkel et al. 2012; Murray et al. 2015; Fitzpatrick et al. 2022; Liu et al. 2021. -------------------------------------------------------------------------------------------------------------------- (V) - Discussion: [8] Reviewer Concern : Emphasize and discuss the added value of the pipeline in obtaining reliable sequences from samples with high ct values. Compare the results obtained using this pipeline with those from other existing pipelines to highlight its superiority. Author Response: As described in [6], we developed CoVpipe2 as a robust and modular surveillance pipeline focusing on amplification protocols and short-read data. Thus, for samples with high Ct values and where amplification can not yield enough output, CoVpipe2 will mark them as “failed” in the reporting. High Ct samples will usually result in regions (amplicons) with low coverage. Such regions are then automatically masked by “N” bases in the final consensus. We don't think a bioinformatics pipeline should construct any “reliable” consensus genome sequence when the data is insufficient. Thus, it is more important to identify such low-quality samples and flag them with a user warning. Our filtering and reporting aims to fit the needs of large-scale surveillance programs with detailed QC information and provide a quick overview of sample results, to identify such challenging samples easily. Thus, CoVpipe2 helps virologists to identify such problematic samples so that they can be selected for re-sequencing or excluded from downstream analysis. Regarding the comparison to other pipelines, we implicitly did that by selecting the test data sets. Those come from another independent benchmark study (Xiaoli et al. 2022), and we compare our CoVpipe2 results against those from the original benchmark paper. The original authors wrote in their publication: > The datasets presented here were generated to help public health laboratories build sequencing and bioinformatics capacity, benchmark different workflows and pipelines, and calibrate QC thresholds to ensure sequencing quality. All available pipelines (Tab. 1 in manuscript) excel in various properties. While some strive to have high detection rates for minor variants for research settings, CoVpipe2 was developed to be easily extendable and adjustable to new requirements in surveillance or other viruses [see 9]. Thus, we would like to stick to our decision of utilizing a publicly available and carefully constructed, independent benchmark data set instead of including more samples and pipelines. Our study focuses on presenting the CoVpipe2 implementation and highlighting various implementation decisions in the context of reconstructing robust genome sequences for surveillance tasks. Nevertheless, we agree that another large-scale and up-to-date benchmark study comparing all available pipelines (Tab. 1), including CoVpipe2, would be interesting but is beyond the scope of our Software article. -------------------------------------------------------------------------------------------------------------------- [9] Reviewer Concern : Provide detailed insights into how this pipeline can be applied to study other viruses Author Response: The predecessor of CoVpipe2, the snakemake pipeline CoVpipe1, was used to create adapted pipelines for RSV and Influenza. In this context, we discovered that other viruses might need other tools and parameters to reflect their characteristics (genome size, segmentation, reference selection). Also, the needs for final reporting may differ depending on the virus under investigation, and changes in the pipeline may be necessary. Besides, we successfully used CoVpipe2 on Polio and Measles viruses for genome reconstruction and variant calling from short-read sequencing data. In short, for Polio viruses, we sequenced the same 24 samples with Sanger, Illumina, and Nanopore, and CoVpipe2 was able to identify the same variants compared to the other sequencing technologies and associated bioinformatic steps (unpublished preliminary data). In general, the basic software framework - the generic sub-processes of raw data quality control, read alignment, variant calling, and consensus building - and the bioinformatic challenges for data derived from amplicon sequencing are equally applicable to other viruses. CoVpipe2 can serve as a blueprint for other pathogens, especially other unsegmented viruses, and provide first insights into the variants and consensus sequences. Tools, parameters, and thresholds might need careful adjustments depending on the pathogen. Similarly, the downstream analysis might be pathogen-specific, e.g., require pathogen-specific datasets (such as reference sequences for Influenza from Nextclade). We are currently working on a harmonized multi-pathogen pipeline with different profiles (tools, parameter settings) tailored towards specific viruses. Besides, interested users can already run CoVpipe2 on other (non-segmented) viruses by simply switching to another reference genome as we did before successfully for analyzing Polio virus amplicon data (parameter --ref_genome). We extend the “Conclusion” accordingly: “We used CoVpipe1 and CoVpipe2, which were initially developed for SARS-CoV-2, to reconstruct the genomes of other viruses. By selecting different reference genomes for polio, measles, RSV, and influenza viruses and modifying the analysis processes for the latter two viruses, we were able to demonstrate the workflow's potential as a universal blueprint for viral genome analysis. This adaptability has been demonstrated by the ability to identify consistent variants and the successful application to different viral characteristics, such as varying genome size and segmentation. However, it should also be noted that for segmented viruses, more customization is required than simply replacing the (non-segmented) reference genome. However, our experience suggests that customization of tools, parameters, and reporting requirements is needed for each virus, pointing to developing a harmonized pipeline for multiple pathogens. CoVpipe2 thus proves to be a robust tool for SARS-CoV-2 and paves the way for broad application in virology research by highlighting its ability to serve as a fundamental framework for building customized pipelines for a wide range of pathogens.” (I) - Organization: [1] Reviewer Concern: While the overall structure is clear, introducing numbering at both the heading and sub-heading levels would enhance clarity and facilitate better distinction between various steps in the workflow. Author Response: Thanks for the suggestion. Numbering (sub)sections helps to structure the text better and distinguish which parts belong together semantically. However, there is little we can do about it, as this is the journal's style. Nevertheless, we will ask the editor/typesetting team if that’s possible. -------------------------------------------------------------------------------------------------------------------- (II) - Technical Terminology: [2] Reviewer Concern: Given the diverse audience, including virologists and other biologists, it is crucial to ensure accessibility by providing clear definitions for all technical terms and acronyms. This will make the paper more easy to readers who may not be familiar with specific bioinformatics or genomics terminology. Author Response: Thanks for the comment. We agreed and added a list of Abbreviations to the manuscript to make it easier for readers to follow the story. COVID-19 - Coronavirus disease 2019 SARS-CoV-2 - Severe acute respiratory syndrome coronavirus 2 GPL3 license - GNU General Public License GISAID - Global Initiative on Sharing All Influenza Data EBI - European Bioinformatics Institute EMBL - European Molecular Biology Laboratory RKI - Robert Koch Institute CorSurV - Coronavirus Surveillance Verordnung (eng., Coronavirus Surveillance Regulation) DESH - Deutscher Elektronischer Sequenzdaten-Hub (eng., German Electronic Sequence Data Hub) IMS-SC2 - Integrated Molecular Surveillance for SARS-CoV-2 ONT - Oxford Nanopore Technologies NGS - Next-Generation Sequencing HPC - High-Performance Computing WSL - Windows Subsystem for Linux CSV file - Comma-Separated Values file GFF file - General Feature Format file BEDPE file - Browser Extensible Data Paired-End file VCF file - Variant Call Format file HTML - Hypertext Markup Language BAM file - Binary Alignment and Map file BED file - Browser Extensible Data file CCO license - Creative Commons Zero license VOC - Variants of Concern VOI - Variants of Interest IUPAC - International Union of Pure and Applied Chemistry indel - Insertion/Deletion Variant JSON file - JavaScript Object Notation file CDC - Centers for Disease Control and Prevention ENA - European Nucleotide Archive QC - Quality Control PCR - Polymerase Chain Reaction ID - Identifier SRA - Sequence Read Archive -------------------------------------------------------------------------------------------------------------------- (III) - References and Sources [3] Reviewer Concern: Include proper references and sources for the tools and databases mentioned by providing specific URLs for easy access to external resources. Author Response: We agree that it’s crucial to acknowledge all tools and resources properly. When there is an original publication for a tool or database, we cite the publication. If not, we cite the code repository or the URL to the resource (such as GitHub or Zenodo). We carefully checked the text again and added citations/URLs if they were missing and necessary. For example, Rev #1 also commented that the specific URL to the custom Kraken 2 database on Zenodo was missing. We added that. -------------------------------------------------------------------------------------------------------------------- (IV) - Methods and Results [4] Reviewer Concern : Introduce a dedicated "Pipeline Evaluation" subsection in both the "Methods" and "Results" sections and change the abstract section accordingly. Author Response: We fully agree that presenting the implementation of a pipeline together with its evaluation with additional analysis can be confusing. So what are the "Methods" and the "Results" parts, then? This is often a problem when simultaneously presenting and evaluating a new software implementation. But we also need to adhere to the style guidelines for journals. We wrote a “Software Tool Articles” and have to follow this structure: https://f1000research.com/for-authors/article-guidelines/software-tool-articles . As you can see in these guidelines, “Software Tool Articles typically contain the following sections: Introduction, Methods, Results (Optional), Use Cases (Optional), Conclusions/Discussion.” In the first version, we skipped the “Use Cases” section to discuss our example data sets for pipeline evaluation directly in the “Results” section. However, thanks to your comment, we believe that also skipping the “Results” section entirely makes our manuscript clearer. We deleted the “Results” section and added the subsections “Selection of benchmark datasets and pipeline evaluation” and “Reporting” at the end of the “Methods”. We changed the subsection “Selection of benchmark datasets” to “Selection of benchmark datasets and pipeline evaluation” to include your suggestion. We think that the structure is now clearer because we first describe in the “Methods” the implementation of the pipeline and how to operate it, according to the journal guidelines for “Software Tool Articles”: The Methods should “Include a subsection on Implementation describing how the tool works and any relevant technical details required for implementation; and a subsection on Operation , which should include the minimal system requirements needed to run the software and an overview of the workflow.” Then, we describe some specific implementation decisions followed by the example data sets for pipeline evaluation and, finally, the report structure we implemented. According to the journal guideline for “Software Tool Articles”: “Abstracts are structured into Background, Methods, Results, and Conclusions”, thus, we can not change sections in the abstract. Although we generally agree that different subheadings would help follow the story, we must also stick to the journal guidelines. -------------------------------------------------------------------------------------------------------------------- [5] Reviewer Concern : Include details on the investigated sequences, with a thorough description of sample types, cycle threshold (ct) values, and sample categories. Author Response: We only use publicly available data sets and reference the original sources (publication, GitHub, and ENA repositories). We suggest that readers should refer to the original sources for further details. However, the description of the benchmark datasets is now also part of the “Methods”. Here, we describe: “We compared the results of CoVpipe2 (v0.4.0) with publicly available benchmark datasets for SARS-CoV-2 surveillance 57 ( GitHub CDC data ).” [57] Xiaoli L, Hagey JV, Park DJ, et al.: Benchmark datasets for sars-cov-2 surveillance bioinformatics. PeerJ. 2022;10:e13821. 10.7717/peerj.13821 We do not want to mirror the details of the benchmark data sets that are described in the original sources. In addition, we can not provide additional information, such as Ct values, because, to the best of our knowledge, this information is not available in the original publication or in the data source (GitHub, ENA). -------------------------------------------------------------------------------------------------------------------- [6] Reviewer Concern : Consider expanding the evaluation by incorporating additional samples/sequences, especially failed sequences from samples with varying ct values (especially high ct values). The Investigation of the contribution of the pipeline in obtaining reliable sequences from samples with high ct values may be helpful for virologist who are dealing with such challenges especially in samples obtained from long-term excretors. Author Response: Thanks for the comment. We agree that investigating challenging samples is especially interesting for users of CoVpipe2. As described above (Q [5]), we selected a publicly available benchmark dataset for SARS-CoV-2 surveillance (Xiaoli et al. 2022) to compare our results directly with previous calculations. In addition, this data set also includes difficult samples that should not withstand automatic quality control (QC) and could mimic high Ct values. CoVpipe2 was developed as a robust and standardized workflow to support genome reconstruction in genomic surveillance programs. Thus, our main goal in developing CoVpipe2 was to provide a robust bioinformatics pipeline for short-read sequencing data that recognizes important mutations with decent allele frequency and automatically identifies and masks ambiguous positions. We implemented parameters (20X coverage to consider a position for variant calling, 90% ACGT nucleotide identity to the reference) to discover low-quality samples that might originate from high Ct values. Running the pipeline on amplicon sequencing data from samples with high Ct can result in "read stacks" with high sequence depth for certain well-amplified amplicons, but it could also lead to low horizontal genome coverage due to low input RNA quantity. CoVpipe2 will report such samples as “failed” in the QC report. Thus, only samples with a decent vertical (sequencing depth) and horizontal genome coverage should be used for downstream genomic surveillance and trustworthy lineage assignment. Nevertheless, CoVpipe2 also reports the full intermediate results, such as BAM files and unfiltered VCF files. Experienced users can investigate all variant calls and their respective allele frequencies - also for QC-failed samples. Thus, it is also possible to investigate mixed variant calls (co-infection, recombinants) and low-frequency variants with the help of CoVpipe2. However, for routine genomic surveillance applications, such challenging samples will be automatically flagged as QC-failed in the pipeline, supporting non-expert users in decision-making and selecting suitable samples for surveillance. Obtaining reliable consensus genomes from high Ct samples is generally difficult. In our experience, it is better to flag such samples with a warning and inform the user that those are of lower quality and probably not suited for further downstream analysis. Thus, CoVpipe2 helps virologists identify such challenging samples so that they can be selected for re-sequencing or exclusion from downstream analysis. -------------------------------------------------------------------------------------------------------------------- [7] Reviewer Concern : Please Illustrate "common challenges" with a graph or figure for better presentation and understanding. Author Response: Here, we present a bioinformatics pipeline with the specific objective of reconstructing robust SARS-CoV-2 consensus genomes from patient samples and short-read data, which is also reflected in the title of our paper. We present CoVpipe2 as a solution to overcome such challenges in reconstructing robust SARS-CoV-2 genomes from short-read (amplicon) data. In Figure 2, we explicitly illustrate common challenges regarding variant calling, which is one of the main obstacles in many reference-based virus bioinformatics pipelines, and where we specifically integrated solutions in CoVpipe2 to overcome such challenges. Besides the dedicated figure for variant calling challenges, we examine other relevant challenges in the context of the CoVpipe2 implementation, such as amplicon drop-outs, in the text. For a general overview, we think that common challenges in the context of amplicon sequencing and virus bioinformatics need to be more broadly addressed in dedicated benchmark studies such as those already available from Beerenwinkel et al. 2012; Murray et al. 2015; Fitzpatrick et al. 2022; Liu et al. 2021. -------------------------------------------------------------------------------------------------------------------- (V) - Discussion: [8] Reviewer Concern : Emphasize and discuss the added value of the pipeline in obtaining reliable sequences from samples with high ct values. Compare the results obtained using this pipeline with those from other existing pipelines to highlight its superiority. Author Response: As described in [6], we developed CoVpipe2 as a robust and modular surveillance pipeline focusing on amplification protocols and short-read data. Thus, for samples with high Ct values and where amplification can not yield enough output, CoVpipe2 will mark them as “failed” in the reporting. High Ct samples will usually result in regions (amplicons) with low coverage. Such regions are then automatically masked by “N” bases in the final consensus. We don't think a bioinformatics pipeline should construct any “reliable” consensus genome sequence when the data is insufficient. Thus, it is more important to identify such low-quality samples and flag them with a user warning. Our filtering and reporting aims to fit the needs of large-scale surveillance programs with detailed QC information and provide a quick overview of sample results, to identify such challenging samples easily. Thus, CoVpipe2 helps virologists to identify such problematic samples so that they can be selected for re-sequencing or excluded from downstream analysis. Regarding the comparison to other pipelines, we implicitly did that by selecting the test data sets. Those come from another independent benchmark study (Xiaoli et al. 2022), and we compare our CoVpipe2 results against those from the original benchmark paper. The original authors wrote in their publication: > The datasets presented here were generated to help public health laboratories build sequencing and bioinformatics capacity, benchmark different workflows and pipelines, and calibrate QC thresholds to ensure sequencing quality. All available pipelines (Tab. 1 in manuscript) excel in various properties. While some strive to have high detection rates for minor variants for research settings, CoVpipe2 was developed to be easily extendable and adjustable to new requirements in surveillance or other viruses [see 9]. Thus, we would like to stick to our decision of utilizing a publicly available and carefully constructed, independent benchmark data set instead of including more samples and pipelines. Our study focuses on presenting the CoVpipe2 implementation and highlighting various implementation decisions in the context of reconstructing robust genome sequences for surveillance tasks. Nevertheless, we agree that another large-scale and up-to-date benchmark study comparing all available pipelines (Tab. 1), including CoVpipe2, would be interesting but is beyond the scope of our Software article. -------------------------------------------------------------------------------------------------------------------- [9] Reviewer Concern : Provide detailed insights into how this pipeline can be applied to study other viruses Author Response: The predecessor of CoVpipe2, the snakemake pipeline CoVpipe1, was used to create adapted pipelines for RSV and Influenza. In this context, we discovered that other viruses might need other tools and parameters to reflect their characteristics (genome size, segmentation, reference selection). Also, the needs for final reporting may differ depending on the virus under investigation, and changes in the pipeline may be necessary. Besides, we successfully used CoVpipe2 on Polio and Measles viruses for genome reconstruction and variant calling from short-read sequencing data. In short, for Polio viruses, we sequenced the same 24 samples with Sanger, Illumina, and Nanopore, and CoVpipe2 was able to identify the same variants compared to the other sequencing technologies and associated bioinformatic steps (unpublished preliminary data). In general, the basic software framework - the generic sub-processes of raw data quality control, read alignment, variant calling, and consensus building - and the bioinformatic challenges for data derived from amplicon sequencing are equally applicable to other viruses. CoVpipe2 can serve as a blueprint for other pathogens, especially other unsegmented viruses, and provide first insights into the variants and consensus sequences. Tools, parameters, and thresholds might need careful adjustments depending on the pathogen. Similarly, the downstream analysis might be pathogen-specific, e.g., require pathogen-specific datasets (such as reference sequences for Influenza from Nextclade). We are currently working on a harmonized multi-pathogen pipeline with different profiles (tools, parameter settings) tailored towards specific viruses. Besides, interested users can already run CoVpipe2 on other (non-segmented) viruses by simply switching to another reference genome as we did before successfully for analyzing Polio virus amplicon data (parameter --ref_genome). We extend the “Conclusion” accordingly: “We used CoVpipe1 and CoVpipe2, which were initially developed for SARS-CoV-2, to reconstruct the genomes of other viruses. By selecting different reference genomes for polio, measles, RSV, and influenza viruses and modifying the analysis processes for the latter two viruses, we were able to demonstrate the workflow's potential as a universal blueprint for viral genome analysis. This adaptability has been demonstrated by the ability to identify consistent variants and the successful application to different viral characteristics, such as varying genome size and segmentation. However, it should also be noted that for segmented viruses, more customization is required than simply replacing the (non-segmented) reference genome. However, our experience suggests that customization of tools, parameters, and reporting requirements is needed for each virus, pointing to developing a harmonized pipeline for multiple pathogens. CoVpipe2 thus proves to be a robust tool for SARS-CoV-2 and paves the way for broad application in virology research by highlighting its ability to serve as a fundamental framework for building customized pipelines for a wide range of pathogens.” Competing Interests: No competing interests were disclosed. Close Report a concern Respond or Comment COMMENTS ON THIS REPORT Author Response 15 May 2024 Martin Hölzer , Genome Competence Center (MF1), Robert Koch Institute, Berlin, Germany 15 May 2024 Author Response (I) - Organization: [1] Reviewer Concern: While the overall structure is clear, introducing numbering at both the heading and sub-heading levels would enhance clarity and facilitate better distinction between various steps ... Continue reading (I) - Organization: [1] Reviewer Concern: While the overall structure is clear, introducing numbering at both the heading and sub-heading levels would enhance clarity and facilitate better distinction between various steps in the workflow. Author Response: Thanks for the suggestion. Numbering (sub)sections helps to structure the text better and distinguish which parts belong together semantically. However, there is little we can do about it, as this is the journal's style. Nevertheless, we will ask the editor/typesetting team if that’s possible. -------------------------------------------------------------------------------------------------------------------- (II) - Technical Terminology: [2] Reviewer Concern: Given the diverse audience, including virologists and other biologists, it is crucial to ensure accessibility by providing clear definitions for all technical terms and acronyms. This will make the paper more easy to readers who may not be familiar with specific bioinformatics or genomics terminology. Author Response: Thanks for the comment. We agreed and added a list of Abbreviations to the manuscript to make it easier for readers to follow the story. COVID-19 - Coronavirus disease 2019 SARS-CoV-2 - Severe acute respiratory syndrome coronavirus 2 GPL3 license - GNU General Public License GISAID - Global Initiative on Sharing All Influenza Data EBI - European Bioinformatics Institute EMBL - European Molecular Biology Laboratory RKI - Robert Koch Institute CorSurV - Coronavirus Surveillance Verordnung (eng., Coronavirus Surveillance Regulation) DESH - Deutscher Elektronischer Sequenzdaten-Hub (eng., German Electronic Sequence Data Hub) IMS-SC2 - Integrated Molecular Surveillance for SARS-CoV-2 ONT - Oxford Nanopore Technologies NGS - Next-Generation Sequencing HPC - High-Performance Computing WSL - Windows Subsystem for Linux CSV file - Comma-Separated Values file GFF file - General Feature Format file BEDPE file - Browser Extensible Data Paired-End file VCF file - Variant Call Format file HTML - Hypertext Markup Language BAM file - Binary Alignment and Map file BED file - Browser Extensible Data file CCO license - Creative Commons Zero license VOC - Variants of Concern VOI - Variants of Interest IUPAC - International Union of Pure and Applied Chemistry indel - Insertion/Deletion Variant JSON file - JavaScript Object Notation file CDC - Centers for Disease Control and Prevention ENA - European Nucleotide Archive QC - Quality Control PCR - Polymerase Chain Reaction ID - Identifier SRA - Sequence Read Archive -------------------------------------------------------------------------------------------------------------------- (III) - References and Sources [3] Reviewer Concern: Include proper references and sources for the tools and databases mentioned by providing specific URLs for easy access to external resources. Author Response: We agree that it’s crucial to acknowledge all tools and resources properly. When there is an original publication for a tool or database, we cite the publication. If not, we cite the code repository or the URL to the resource (such as GitHub or Zenodo). We carefully checked the text again and added citations/URLs if they were missing and necessary. For example, Rev #1 also commented that the specific URL to the custom Kraken 2 database on Zenodo was missing. We added that. -------------------------------------------------------------------------------------------------------------------- (IV) - Methods and Results [4] Reviewer Concern : Introduce a dedicated "Pipeline Evaluation" subsection in both the "Methods" and "Results" sections and change the abstract section accordingly. Author Response: We fully agree that presenting the implementation of a pipeline together with its evaluation with additional analysis can be confusing. So what are the "Methods" and the "Results" parts, then? This is often a problem when simultaneously presenting and evaluating a new software implementation. But we also need to adhere to the style guidelines for journals. We wrote a “Software Tool Articles” and have to follow this structure: https://f1000research.com/for-authors/article-guidelines/software-tool-articles . As you can see in these guidelines, “Software Tool Articles typically contain the following sections: Introduction, Methods, Results (Optional), Use Cases (Optional), Conclusions/Discussion.” In the first version, we skipped the “Use Cases” section to discuss our example data sets for pipeline evaluation directly in the “Results” section. However, thanks to your comment, we believe that also skipping the “Results” section entirely makes our manuscript clearer. We deleted the “Results” section and added the subsections “Selection of benchmark datasets and pipeline evaluation” and “Reporting” at the end of the “Methods”. We changed the subsection “Selection of benchmark datasets” to “Selection of benchmark datasets and pipeline evaluation” to include your suggestion. We think that the structure is now clearer because we first describe in the “Methods” the implementation of the pipeline and how to operate it, according to the journal guidelines for “Software Tool Articles”: The Methods should “Include a subsection on Implementation describing how the tool works and any relevant technical details required for implementation; and a subsection on Operation , which should include the minimal system requirements needed to run the software and an overview of the workflow.” Then, we describe some specific implementation decisions followed by the example data sets for pipeline evaluation and, finally, the report structure we implemented. According to the journal guideline for “Software Tool Articles”: “Abstracts are structured into Background, Methods, Results, and Conclusions”, thus, we can not change sections in the abstract. Although we generally agree that different subheadings would help follow the story, we must also stick to the journal guidelines. -------------------------------------------------------------------------------------------------------------------- [5] Reviewer Concern : Include details on the investigated sequences, with a thorough description of sample types, cycle threshold (ct) values, and sample categories. Author Response: We only use publicly available data sets and reference the original sources (publication, GitHub, and ENA repositories). We suggest that readers should refer to the original sources for further details. However, the description of the benchmark datasets is now also part of the “Methods”. Here, we describe: “We compared the results of CoVpipe2 (v0.4.0) with publicly available benchmark datasets for SARS-CoV-2 surveillance 57 ( GitHub CDC data ).” [57] Xiaoli L, Hagey JV, Park DJ, et al.: Benchmark datasets for sars-cov-2 surveillance bioinformatics. PeerJ. 2022;10:e13821. 10.7717/peerj.13821 We do not want to mirror the details of the benchmark data sets that are described in the original sources. In addition, we can not provide additional information, such as Ct values, because, to the best of our knowledge, this information is not available in the original publication or in the data source (GitHub, ENA). -------------------------------------------------------------------------------------------------------------------- [6] Reviewer Concern : Consider expanding the evaluation by incorporating additional samples/sequences, especially failed sequences from samples with varying ct values (especially high ct values). The Investigation of the contribution of the pipeline in obtaining reliable sequences from samples with high ct values may be helpful for virologist who are dealing with such challenges especially in samples obtained from long-term excretors. Author Response: Thanks for the comment. We agree that investigating challenging samples is especially interesting for users of CoVpipe2. As described above (Q [5]), we selected a publicly available benchmark dataset for SARS-CoV-2 surveillance (Xiaoli et al. 2022) to compare our results directly with previous calculations. In addition, this data set also includes difficult samples that should not withstand automatic quality control (QC) and could mimic high Ct values. CoVpipe2 was developed as a robust and standardized workflow to support genome reconstruction in genomic surveillance programs. Thus, our main goal in developing CoVpipe2 was to provide a robust bioinformatics pipeline for short-read sequencing data that recognizes important mutations with decent allele frequency and automatically identifies and masks ambiguous positions. We implemented parameters (20X coverage to consider a position for variant calling, 90% ACGT nucleotide identity to the reference) to discover low-quality samples that might originate from high Ct values. Running the pipeline on amplicon sequencing data from samples with high Ct can result in "read stacks" with high sequence depth for certain well-amplified amplicons, but it could also lead to low horizontal genome coverage due to low input RNA quantity. CoVpipe2 will report such samples as “failed” in the QC report. Thus, only samples with a decent vertical (sequencing depth) and horizontal genome coverage should be used for downstream genomic surveillance and trustworthy lineage assignment. Nevertheless, CoVpipe2 also reports the full intermediate results, such as BAM files and unfiltered VCF files. Experienced users can investigate all variant calls and their respective allele frequencies - also for QC-failed samples. Thus, it is also possible to investigate mixed variant calls (co-infection, recombinants) and low-frequency variants with the help of CoVpipe2. However, for routine genomic surveillance applications, such challenging samples will be automatically flagged as QC-failed in the pipeline, supporting non-expert users in decision-making and selecting suitable samples for surveillance. Obtaining reliable consensus genomes from high Ct samples is generally difficult. In our experience, it is better to flag such samples with a warning and inform the user that those are of lower quality and probably not suited for further downstream analysis. Thus, CoVpipe2 helps virologists identify such challenging samples so that they can be selected for re-sequencing or exclusion from downstream analysis. -------------------------------------------------------------------------------------------------------------------- [7] Reviewer Concern : Please Illustrate "common challenges" with a graph or figure for better presentation and understanding. Author Response: Here, we present a bioinformatics pipeline with the specific objective of reconstructing robust SARS-CoV-2 consensus genomes from patient samples and short-read data, which is also reflected in the title of our paper. We present CoVpipe2 as a solution to overcome such challenges in reconstructing robust SARS-CoV-2 genomes from short-read (amplicon) data. In Figure 2, we explicitly illustrate common challenges regarding variant calling, which is one of the main obstacles in many reference-based virus bioinformatics pipelines, and where we specifically integrated solutions in CoVpipe2 to overcome such challenges. Besides the dedicated figure for variant calling challenges, we examine other relevant challenges in the context of the CoVpipe2 implementation, such as amplicon drop-outs, in the text. For a general overview, we think that common challenges in the context of amplicon sequencing and virus bioinformatics need to be more broadly addressed in dedicated benchmark studies such as those already available from Beerenwinkel et al. 2012; Murray et al. 2015; Fitzpatrick et al. 2022; Liu et al. 2021. -------------------------------------------------------------------------------------------------------------------- (V) - Discussion: [8] Reviewer Concern : Emphasize and discuss the added value of the pipeline in obtaining reliable sequences from samples with high ct values. Compare the results obtained using this pipeline with those from other existing pipelines to highlight its superiority. Author Response: As described in [6], we developed CoVpipe2 as a robust and modular surveillance pipeline focusing on amplification protocols and short-read data. Thus, for samples with high Ct values and where amplification can not yield enough output, CoVpipe2 will mark them as “failed” in the reporting. High Ct samples will usually result in regions (amplicons) with low coverage. Such regions are then automatically masked by “N” bases in the final consensus. We don't think a bioinformatics pipeline should construct any “reliable” consensus genome sequence when the data is insufficient. Thus, it is more important to identify such low-quality samples and flag them with a user warning. Our filtering and reporting aims to fit the needs of large-scale surveillance programs with detailed QC information and provide a quick overview of sample results, to identify such challenging samples easily. Thus, CoVpipe2 helps virologists to identify such problematic samples so that they can be selected for re-sequencing or excluded from downstream analysis. Regarding the comparison to other pipelines, we implicitly did that by selecting the test data sets. Those come from another independent benchmark study (Xiaoli et al. 2022), and we compare our CoVpipe2 results against those from the original benchmark paper. The original authors wrote in their publication: > The datasets presented here were generated to help public health laboratories build sequencing and bioinformatics capacity, benchmark different workflows and pipelines, and calibrate QC thresholds to ensure sequencing quality. All available pipelines (Tab. 1 in manuscript) excel in various properties. While some strive to have high detection rates for minor variants for research settings, CoVpipe2 was developed to be easily extendable and adjustable to new requirements in surveillance or other viruses [see 9]. Thus, we would like to stick to our decision of utilizing a publicly available and carefully constructed, independent benchmark data set instead of including more samples and pipelines. Our study focuses on presenting the CoVpipe2 implementation and highlighting various implementation decisions in the context of reconstructing robust genome sequences for surveillance tasks. Nevertheless, we agree that another large-scale and up-to-date benchmark study comparing all available pipelines (Tab. 1), including CoVpipe2, would be interesting but is beyond the scope of our Software article. -------------------------------------------------------------------------------------------------------------------- [9] Reviewer Concern : Provide detailed insights into how this pipeline can be applied to study other viruses Author Response: The predecessor of CoVpipe2, the snakemake pipeline CoVpipe1, was used to create adapted pipelines for RSV and Influenza. In this context, we discovered that other viruses might need other tools and parameters to reflect their characteristics (genome size, segmentation, reference selection). Also, the needs for final reporting may differ depending on the virus under investigation, and changes in the pipeline may be necessary. Besides, we successfully used CoVpipe2 on Polio and Measles viruses for genome reconstruction and variant calling from short-read sequencing data. In short, for Polio viruses, we sequenced the same 24 samples with Sanger, Illumina, and Nanopore, and CoVpipe2 was able to identify the same variants compared to the other sequencing technologies and associated bioinformatic steps (unpublished preliminary data). In general, the basic software framework - the generic sub-processes of raw data quality control, read alignment, variant calling, and consensus building - and the bioinformatic challenges for data derived from amplicon sequencing are equally applicable to other viruses. CoVpipe2 can serve as a blueprint for other pathogens, especially other unsegmented viruses, and provide first insights into the variants and consensus sequences. Tools, parameters, and thresholds might need careful adjustments depending on the pathogen. Similarly, the downstream analysis might be pathogen-specific, e.g., require pathogen-specific datasets (such as reference sequences for Influenza from Nextclade). We are currently working on a harmonized multi-pathogen pipeline with different profiles (tools, parameter settings) tailored towards specific viruses. Besides, interested users can already run CoVpipe2 on other (non-segmented) viruses by simply switching to another reference genome as we did before successfully for analyzing Polio virus amplicon data (parameter --ref_genome). We extend the “Conclusion” accordingly: “We used CoVpipe1 and CoVpipe2, which were initially developed for SARS-CoV-2, to reconstruct the genomes of other viruses. By selecting different reference genomes for polio, measles, RSV, and influenza viruses and modifying the analysis processes for the latter two viruses, we were able to demonstrate the workflow's potential as a universal blueprint for viral genome analysis. This adaptability has been demonstrated by the ability to identify consistent variants and the successful application to different viral characteristics, such as varying genome size and segmentation. However, it should also be noted that for segmented viruses, more customization is required than simply replacing the (non-segmented) reference genome. However, our experience suggests that customization of tools, parameters, and reporting requirements is needed for each virus, pointing to developing a harmonized pipeline for multiple pathogens. CoVpipe2 thus proves to be a robust tool for SARS-CoV-2 and paves the way for broad application in virology research by highlighting its ability to serve as a fundamental framework for building customized pipelines for a wide range of pathogens.” (I) - Organization: [1] Reviewer Concern: While the overall structure is clear, introducing numbering at both the heading and sub-heading levels would enhance clarity and facilitate better distinction between various steps in the workflow. Author Response: Thanks for the suggestion. Numbering (sub)sections helps to structure the text better and distinguish which parts belong together semantically. However, there is little we can do about it, as this is the journal's style. Nevertheless, we will ask the editor/typesetting team if that’s possible. -------------------------------------------------------------------------------------------------------------------- (II) - Technical Terminology: [2] Reviewer Concern: Given the diverse audience, including virologists and other biologists, it is crucial to ensure accessibility by providing clear definitions for all technical terms and acronyms. This will make the paper more easy to readers who may not be familiar with specific bioinformatics or genomics terminology. Author Response: Thanks for the comment. We agreed and added a list of Abbreviations to the manuscript to make it easier for readers to follow the story. COVID-19 - Coronavirus disease 2019 SARS-CoV-2 - Severe acute respiratory syndrome coronavirus 2 GPL3 license - GNU General Public License GISAID - Global Initiative on Sharing All Influenza Data EBI - European Bioinformatics Institute EMBL - European Molecular Biology Laboratory RKI - Robert Koch Institute CorSurV - Coronavirus Surveillance Verordnung (eng., Coronavirus Surveillance Regulation) DESH - Deutscher Elektronischer Sequenzdaten-Hub (eng., German Electronic Sequence Data Hub) IMS-SC2 - Integrated Molecular Surveillance for SARS-CoV-2 ONT - Oxford Nanopore Technologies NGS - Next-Generation Sequencing HPC - High-Performance Computing WSL - Windows Subsystem for Linux CSV file - Comma-Separated Values file GFF file - General Feature Format file BEDPE file - Browser Extensible Data Paired-End file VCF file - Variant Call Format file HTML - Hypertext Markup Language BAM file - Binary Alignment and Map file BED file - Browser Extensible Data file CCO license - Creative Commons Zero license VOC - Variants of Concern VOI - Variants of Interest IUPAC - International Union of Pure and Applied Chemistry indel - Insertion/Deletion Variant JSON file - JavaScript Object Notation file CDC - Centers for Disease Control and Prevention ENA - European Nucleotide Archive QC - Quality Control PCR - Polymerase Chain Reaction ID - Identifier SRA - Sequence Read Archive -------------------------------------------------------------------------------------------------------------------- (III) - References and Sources [3] Reviewer Concern: Include proper references and sources for the tools and databases mentioned by providing specific URLs for easy access to external resources. Author Response: We agree that it’s crucial to acknowledge all tools and resources properly. When there is an original publication for a tool or database, we cite the publication. If not, we cite the code repository or the URL to the resource (such as GitHub or Zenodo). We carefully checked the text again and added citations/URLs if they were missing and necessary. For example, Rev #1 also commented that the specific URL to the custom Kraken 2 database on Zenodo was missing. We added that. -------------------------------------------------------------------------------------------------------------------- (IV) - Methods and Results [4] Reviewer Concern : Introduce a dedicated "Pipeline Evaluation" subsection in both the "Methods" and "Results" sections and change the abstract section accordingly. Author Response: We fully agree that presenting the implementation of a pipeline together with its evaluation with additional analysis can be confusing. So what are the "Methods" and the "Results" parts, then? This is often a problem when simultaneously presenting and evaluating a new software implementation. But we also need to adhere to the style guidelines for journals. We wrote a “Software Tool Articles” and have to follow this structure: https://f1000research.com/for-authors/article-guidelines/software-tool-articles . As you can see in these guidelines, “Software Tool Articles typically contain the following sections: Introduction, Methods, Results (Optional), Use Cases (Optional), Conclusions/Discussion.” In the first version, we skipped the “Use Cases” section to discuss our example data sets for pipeline evaluation directly in the “Results” section. However, thanks to your comment, we believe that also skipping the “Results” section entirely makes our manuscript clearer. We deleted the “Results” section and added the subsections “Selection of benchmark datasets and pipeline evaluation” and “Reporting” at the end of the “Methods”. We changed the subsection “Selection of benchmark datasets” to “Selection of benchmark datasets and pipeline evaluation” to include your suggestion. We think that the structure is now clearer because we first describe in the “Methods” the implementation of the pipeline and how to operate it, according to the journal guidelines for “Software Tool Articles”: The Methods should “Include a subsection on Implementation describing how the tool works and any relevant technical details required for implementation; and a subsection on Operation , which should include the minimal system requirements needed to run the software and an overview of the workflow.” Then, we describe some specific implementation decisions followed by the example data sets for pipeline evaluation and, finally, the report structure we implemented. According to the journal guideline for “Software Tool Articles”: “Abstracts are structured into Background, Methods, Results, and Conclusions”, thus, we can not change sections in the abstract. Although we generally agree that different subheadings would help follow the story, we must also stick to the journal guidelines. -------------------------------------------------------------------------------------------------------------------- [5] Reviewer Concern : Include details on the investigated sequences, with a thorough description of sample types, cycle threshold (ct) values, and sample categories. Author Response: We only use publicly available data sets and reference the original sources (publication, GitHub, and ENA repositories). We suggest that readers should refer to the original sources for further details. However, the description of the benchmark datasets is now also part of the “Methods”. Here, we describe: “We compared the results of CoVpipe2 (v0.4.0) with publicly available benchmark datasets for SARS-CoV-2 surveillance 57 ( GitHub CDC data ).” [57] Xiaoli L, Hagey JV, Park DJ, et al.: Benchmark datasets for sars-cov-2 surveillance bioinformatics. PeerJ. 2022;10:e13821. 10.7717/peerj.13821 We do not want to mirror the details of the benchmark data sets that are described in the original sources. In addition, we can not provide additional information, such as Ct values, because, to the best of our knowledge, this information is not available in the original publication or in the data source (GitHub, ENA). -------------------------------------------------------------------------------------------------------------------- [6] Reviewer Concern : Consider expanding the evaluation by incorporating additional samples/sequences, especially failed sequences from samples with varying ct values (especially high ct values). The Investigation of the contribution of the pipeline in obtaining reliable sequences from samples with high ct values may be helpful for virologist who are dealing with such challenges especially in samples obtained from long-term excretors. Author Response: Thanks for the comment. We agree that investigating challenging samples is especially interesting for users of CoVpipe2. As described above (Q [5]), we selected a publicly available benchmark dataset for SARS-CoV-2 surveillance (Xiaoli et al. 2022) to compare our results directly with previous calculations. In addition, this data set also includes difficult samples that should not withstand automatic quality control (QC) and could mimic high Ct values. CoVpipe2 was developed as a robust and standardized workflow to support genome reconstruction in genomic surveillance programs. Thus, our main goal in developing CoVpipe2 was to provide a robust bioinformatics pipeline for short-read sequencing data that recognizes important mutations with decent allele frequency and automatically identifies and masks ambiguous positions. We implemented parameters (20X coverage to consider a position for variant calling, 90% ACGT nucleotide identity to the reference) to discover low-quality samples that might originate from high Ct values. Running the pipeline on amplicon sequencing data from samples with high Ct can result in "read stacks" with high sequence depth for certain well-amplified amplicons, but it could also lead to low horizontal genome coverage due to low input RNA quantity. CoVpipe2 will report such samples as “failed” in the QC report. Thus, only samples with a decent vertical (sequencing depth) and horizontal genome coverage should be used for downstream genomic surveillance and trustworthy lineage assignment. Nevertheless, CoVpipe2 also reports the full intermediate results, such as BAM files and unfiltered VCF files. Experienced users can investigate all variant calls and their respective allele frequencies - also for QC-failed samples. Thus, it is also possible to investigate mixed variant calls (co-infection, recombinants) and low-frequency variants with the help of CoVpipe2. However, for routine genomic surveillance applications, such challenging samples will be automatically flagged as QC-failed in the pipeline, supporting non-expert users in decision-making and selecting suitable samples for surveillance. Obtaining reliable consensus genomes from high Ct samples is generally difficult. In our experience, it is better to flag such samples with a warning and inform the user that those are of lower quality and probably not suited for further downstream analysis. Thus, CoVpipe2 helps virologists identify such challenging samples so that they can be selected for re-sequencing or exclusion from downstream analysis. -------------------------------------------------------------------------------------------------------------------- [7] Reviewer Concern : Please Illustrate "common challenges" with a graph or figure for better presentation and understanding. Author Response: Here, we present a bioinformatics pipeline with the specific objective of reconstructing robust SARS-CoV-2 consensus genomes from patient samples and short-read data, which is also reflected in the title of our paper. We present CoVpipe2 as a solution to overcome such challenges in reconstructing robust SARS-CoV-2 genomes from short-read (amplicon) data. In Figure 2, we explicitly illustrate common challenges regarding variant calling, which is one of the main obstacles in many reference-based virus bioinformatics pipelines, and where we specifically integrated solutions in CoVpipe2 to overcome such challenges. Besides the dedicated figure for variant calling challenges, we examine other relevant challenges in the context of the CoVpipe2 implementation, such as amplicon drop-outs, in the text. For a general overview, we think that common challenges in the context of amplicon sequencing and virus bioinformatics need to be more broadly addressed in dedicated benchmark studies such as those already available from Beerenwinkel et al. 2012; Murray et al. 2015; Fitzpatrick et al. 2022; Liu et al. 2021. -------------------------------------------------------------------------------------------------------------------- (V) - Discussion: [8] Reviewer Concern : Emphasize and discuss the added value of the pipeline in obtaining reliable sequences from samples with high ct values. Compare the results obtained using this pipeline with those from other existing pipelines to highlight its superiority. Author Response: As described in [6], we developed CoVpipe2 as a robust and modular surveillance pipeline focusing on amplification protocols and short-read data. Thus, for samples with high Ct values and where amplification can not yield enough output, CoVpipe2 will mark them as “failed” in the reporting. High Ct samples will usually result in regions (amplicons) with low coverage. Such regions are then automatically masked by “N” bases in the final consensus. We don't think a bioinformatics pipeline should construct any “reliable” consensus genome sequence when the data is insufficient. Thus, it is more important to identify such low-quality samples and flag them with a user warning. Our filtering and reporting aims to fit the needs of large-scale surveillance programs with detailed QC information and provide a quick overview of sample results, to identify such challenging samples easily. Thus, CoVpipe2 helps virologists to identify such problematic samples so that they can be selected for re-sequencing or excluded from downstream analysis. Regarding the comparison to other pipelines, we implicitly did that by selecting the test data sets. Those come from another independent benchmark study (Xiaoli et al. 2022), and we compare our CoVpipe2 results against those from the original benchmark paper. The original authors wrote in their publication: > The datasets presented here were generated to help public health laboratories build sequencing and bioinformatics capacity, benchmark different workflows and pipelines, and calibrate QC thresholds to ensure sequencing quality. All available pipelines (Tab. 1 in manuscript) excel in various properties. While some strive to have high detection rates for minor variants for research settings, CoVpipe2 was developed to be easily extendable and adjustable to new requirements in surveillance or other viruses [see 9]. Thus, we would like to stick to our decision of utilizing a publicly available and carefully constructed, independent benchmark data set instead of including more samples and pipelines. Our study focuses on presenting the CoVpipe2 implementation and highlighting various implementation decisions in the context of reconstructing robust genome sequences for surveillance tasks. Nevertheless, we agree that another large-scale and up-to-date benchmark study comparing all available pipelines (Tab. 1), including CoVpipe2, would be interesting but is beyond the scope of our Software article. -------------------------------------------------------------------------------------------------------------------- [9] Reviewer Concern : Provide detailed insights into how this pipeline can be applied to study other viruses Author Response: The predecessor of CoVpipe2, the snakemake pipeline CoVpipe1, was used to create adapted pipelines for RSV and Influenza. In this context, we discovered that other viruses might need other tools and parameters to reflect their characteristics (genome size, segmentation, reference selection). Also, the needs for final reporting may differ depending on the virus under investigation, and changes in the pipeline may be necessary. Besides, we successfully used CoVpipe2 on Polio and Measles viruses for genome reconstruction and variant calling from short-read sequencing data. In short, for Polio viruses, we sequenced the same 24 samples with Sanger, Illumina, and Nanopore, and CoVpipe2 was able to identify the same variants compared to the other sequencing technologies and associated bioinformatic steps (unpublished preliminary data). In general, the basic software framework - the generic sub-processes of raw data quality control, read alignment, variant calling, and consensus building - and the bioinformatic challenges for data derived from amplicon sequencing are equally applicable to other viruses. CoVpipe2 can serve as a blueprint for other pathogens, especially other unsegmented viruses, and provide first insights into the variants and consensus sequences. Tools, parameters, and thresholds might need careful adjustments depending on the pathogen. Similarly, the downstream analysis might be pathogen-specific, e.g., require pathogen-specific datasets (such as reference sequences for Influenza from Nextclade). We are currently working on a harmonized multi-pathogen pipeline with different profiles (tools, parameter settings) tailored towards specific viruses. Besides, interested users can already run CoVpipe2 on other (non-segmented) viruses by simply switching to another reference genome as we did before successfully for analyzing Polio virus amplicon data (parameter --ref_genome). We extend the “Conclusion” accordingly: “We used CoVpipe1 and CoVpipe2, which were initially developed for SARS-CoV-2, to reconstruct the genomes of other viruses. By selecting different reference genomes for polio, measles, RSV, and influenza viruses and modifying the analysis processes for the latter two viruses, we were able to demonstrate the workflow's potential as a universal blueprint for viral genome analysis. This adaptability has been demonstrated by the ability to identify consistent variants and the successful application to different viral characteristics, such as varying genome size and segmentation. However, it should also be noted that for segmented viruses, more customization is required than simply replacing the (non-segmented) reference genome. However, our experience suggests that customization of tools, parameters, and reporting requirements is needed for each virus, pointing to developing a harmonized pipeline for multiple pathogens. CoVpipe2 thus proves to be a robust tool for SARS-CoV-2 and paves the way for broad application in virology research by highlighting its ability to serve as a fundamental framework for building customized pipelines for a wide range of pathogens.” Competing Interests: No competing interests were disclosed. Close Report a concern COMMENT ON THIS REPORT Views 0 Cite How to cite this report: Maier W. Reviewer Report For: Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.5256/f1000research.149827.r203328 ) The direct URL for this report is: https://f1000research.com/articles/12-1091/v1#referee-response-203328 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 20 Sep 2023 Wolfgang Maier , Bioinformatics Group, Department of Computer Science, Albert-Ludwigs-Universitat Freiburg, Freiburg, Baden-Württemberg, Germany Approved VIEWS 0 https://doi.org/10.5256/f1000research.149827.r203328 The manuscript by Lataretu et al. describes CoVpipe2, a bioinformatics pipeline for constructing viral consensus sequences from SARS-CoV-2 short sequenced reads obtained using the Illumina platform. It discusses the current state of the pipeline, various design decisions that have led ... Continue reading READ ALL The manuscript by Lataretu et al. describes CoVpipe2, a bioinformatics pipeline for constructing viral consensus sequences from SARS-CoV-2 short sequenced reads obtained using the Illumina platform. It discusses the current state of the pipeline, various design decisions that have led to that state, and typical analysis pitfalls that the authors hope to overcome with their pipeline design. The pipeline itself is implemented as a Nextflow pipeline and comes under a free and open-source license, which means that its exact steps and parameters can be explored down to any desired level of detail. Still the authors provide a very helpful overview in the manuscript text and in Figure 1 through both of which the reader can gain a good understanding of the pipeline layout and its components. The authors point out the existence of multiple alternative analysis pipelines with similar scope as CoVpipe2 and list many of them in Table 1. I would expect most of these alternative pipelines to produce consensus genomes of similar quality as CoVpipe2, and most of the steps that CoVpipe2 is composed of are relatively standard in the field. From this perspective, the manuscript could be said to lack novelty. Beyond the core steps shared in similar form with many other pipelines, there are, however, some smart extra steps built into CoVpipe2 that are innovative ideas, like screening steps for mixed-infection and recombinant samples and consensus genome annotation with Liftoff. More importantly, however, the authors are not just advertising yet another pipeline for SARS-CoV-2 genome analysis, but their manuscript is the kind of documentation that you wish every such pipeline came with: it explains not only individual analysis steps, but also their purpose, special analysis tweaks found to be necessary, and provides links to all relevant resources. In summary, the manuscript describes a robust and mature resource for reproducible data analysis and does an excellent job at that. I enjoyed reading it and even though I've spent a considerable amount of time on developing similar pipelines I still picked up a few new ideas from it. I have no concerns regarding publication of this valuable manuscript, just one comment that the authors may wish to address in the manuscript directly or in a separate reply: For some steps in CoVpipe2 it seems there would have been several tools to choose from, and I'm wondering whether the authors of the pipeline have evaluated alternatives. In particular, I'd be interested in learning why freebayes was chosen as the variant caller and why primer trimming is done with BAMclipper, as I don't think these two tools are used by many other comparable pipelines. If the authors had specific reasons to prefer these tools over alternatives, it might add to the value of the manuscript if these were added to the text. Beyond that, I have found a small number of inconsistencies and typos that I think should be fixed before publication: 1. the authors have done a very careful citation job, in general, but I think the following resources/specifications also deserve links/citations: Zenodo The "precalculated Kraken2 database" deposited at Zenodo The BEDPE format ( https://bedtools.readthedocs.io/en/latest/content/general-usage.html#bedpe-format ) Instead of providing a general anaconda.org link (as done twice in the Methods section), it would be more helpful to provide direct links to the latest versions of pangolin and nextclade ( https://anaconda.org/bioconda/pangolin and https://anaconda.org/bioconda/nextclade ), which also includes channel information. 2. In the introduction " While sequencing intensity and turnaround times on variant detection increased in different countries " is probably intended to mean increasing sequencing intensity, but *decreasing* turnaround times? 3. In the introduction, when DESH genomes statistics are given, the numbers should be 1.2 million and 1.1 million (period instead of comma), and in the discussion of Dataset 5 -> Lineages " for each CoVpip2-GISAD pair " has a typo in the pipeline name. Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Yes Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Yes Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes Competing Interests: No competing interests were disclosed. Reviewer Expertise: Bioinformatics, pathogen genomics, genetics I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Maier W. Reviewer Report For: Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.5256/f1000research.149827.r203328 ) The direct URL for this report is: https://f1000research.com/articles/12-1091/v1#referee-response-203328 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Author Response 15 May 2024 Martin Hölzer , Genome Competence Center (MF1), Robert Koch Institute, Berlin, Germany 15 May 2024 Author Response [1] Reviewer Question: For some steps in CoVpipe2 it seems there would have been several tools to choose from, and I'm wondering whether the authors of the pipeline have evaluated ... Continue reading [1] Reviewer Question: For some steps in CoVpipe2 it seems there would have been several tools to choose from, and I'm wondering whether the authors of the pipeline have evaluated alternatives. In particular, I'd be interested in learning why freebayes was chosen as the variant caller and why primer trimming is done with BAMclipper, as I don't think these two tools are used by many other comparable pipelines. If the authors had specific reasons to prefer these tools over alternatives, it might add to the value of the manuscript if these were added to the text. Author Response: Thanks for the question. You are absolutely right; as with almost any bioinformatics pipeline, there are different options when choosing tools for steps such as quality control, mapping, and variant calling. Our final selection of tools is based on our experience in analyzing sequencing data throughout the SARS-CoV-2 pandemic. Especially in the early days, we performed various internal benchmarks on SARS-CoV-2 Illumina data and manually investigated mapping results and variant calls together with colleagues from our expert unit for respiratory viruses. Thus, our main objective in CoVpipe2 was to reliably detect variants with high allelic depth and good read support. Low-frequency variants were not the primary focus of the pipeline, as the tool is intended to reconstruct robust consensus genomes from patient samples that can be used for genomic surveillance. However, if a user wants to use CoVpipe2 for different research questions, the implementation allows full customization of the necessary parameters (allele frequency, genotype adjustment, …) By screening the literature and examining other pipelines and community standards, we carefully selected the tools that performed best in our internal benchmarks for SARS-CoV-2 short-read data. Regarding variant calling, we first tested LoFreq (Wilm et al. 2012). Although very sensitive, LoFreq lacks a strong genotyping module that was crucial for our downstream processing of the called variants. Furthermore, the output files were hard to process (non-standard VCF formats). We implemented GATK as a second choice, which is a standard tool for eukaryotic genomic variant calling (McKenna et al . (2010), Van der Auwera & O'Connor (2020)) but was also shown to perform well on non-human targets (Lefouili et al. 2022). Performance and output standards were excellent, but it turned out that GATK misses a low amount of viral genomic variants in some samples, although multisample calling was employed. Single false negative variants, which we identified via a comprehensive investigation of the BAM files, were deemed to be too important to stick with the tool. Finally, we chose freebayes (Garrison et al . 2012), which excelled with high performance, high precision, and output files that were straightforward to process in downstream steps of the pipeline. In addition to the variant callers, there is indeed a large selection of quality processing tools. We opted for fastp rather than Trimmomatic because the processing speed is much faster, and the output quality is at least as good. It was shown, that “data filtered by Trimmomatic, SOAPNuke, Cutadapt and fastp were detected with 7174, 7040, 6942 and 6708 false positive variants respectively” (Chen et al . 2018), highlighting that fastp preprocessing can even improve the specificity of downstream analysis. In addition, fastp is equipped with a broader portfolio of parameter options. Another crucial step in read processing is primer clipping, which may promote mapping artifacts and dilution of variant calls if done before read alignment. If InDels occur close to the end of amplicons, a gap open penalty is more expensive than a few mismatches in mapping. For example, this caused trouble with the Spike DEL69/70 in several amplicon kits and made it necessary to use the artificial primer as a mapping anchor. Hence we replaced cutadapt, which clips adapters before mapping, with BAMclipper, which removes adapters after mapping. In our opinion, primer clipping should generally be performed after mapping in reference-based analysis. We, therefore, also discuss late primer clipping more prominently in connection with CoVpipe2. Furthermore, we fully agree that continuous benchmarking is a necessary process to adapt pipelines to changing wet lab procedures, adapted priming schemes, and pathogen evolution. If we find problems, we will also adapt CoVpipe2 accordingly and release new stable versions for reproducible research. We added the following text to the manuscript: “We carefully selected the bioinformatics tools integrated into CoVpipe2 based on internal benchmarks and in-depth manual reviews of sequencing data, mapping results, and called variants. Based on our hands-on experience with SARS-CoV-2 sequencing datasets during the pandemic, this approach ensured that tools that can robustly detect high allele depth and well-covered variants are used for detection. Despite the primary goal of CoVpipe2 to identify high-confidence variants to reconstruct robust consensus genomes for genomic surveillance, the pipeline also provides flexibility for adaptation to different research environments.” -------------------------------------------------------------------------------------------------------------------------- [2] Reviewer Question: the authors have done a very careful citation job, in general, but I think the following resources/specifications also deserve links/citations]: Author Response: Thanks for the comment. We fully agree and added more precise references for various sources such as Zenodo, the custom Kraken 2 database, BEDPE format, and Nextcalde/pangolin on anaconda.org. -------------------------------------------------------------------------------------------------------------------------- [3] Reviewer Question: In the introduction "While sequencing intensity and turnaround times on variant detection increased in different countries" is probably intended to mean increasing sequencing intensity, but *decreasing* turnaround times? Author Response: Yes, you are right; thanks for catching this. We changed the text accordingly: “While sequencing intensity increased and turnaround times on variant detection decreased in different countries, there are also major disparities between high-, low- and middle-income countries in the SARS-CoV-2 global genomic surveillance efforts.” -------------------------------------------------------------------------------------------------------------------------- [4] Reviewer Question: In the introduction, when DESH genomes statistics are given, the numbers should be 1.2 million and 1.1 million (period instead of comma), and in the discussion of Dataset 5 -> Lineages "for each CoVpip2-GISAD pair" has a typo in the pipeline name. Author Response: Thanks, we corrected that. [1] Reviewer Question: For some steps in CoVpipe2 it seems there would have been several tools to choose from, and I'm wondering whether the authors of the pipeline have evaluated alternatives. In particular, I'd be interested in learning why freebayes was chosen as the variant caller and why primer trimming is done with BAMclipper, as I don't think these two tools are used by many other comparable pipelines. If the authors had specific reasons to prefer these tools over alternatives, it might add to the value of the manuscript if these were added to the text. Author Response: Thanks for the question. You are absolutely right; as with almost any bioinformatics pipeline, there are different options when choosing tools for steps such as quality control, mapping, and variant calling. Our final selection of tools is based on our experience in analyzing sequencing data throughout the SARS-CoV-2 pandemic. Especially in the early days, we performed various internal benchmarks on SARS-CoV-2 Illumina data and manually investigated mapping results and variant calls together with colleagues from our expert unit for respiratory viruses. Thus, our main objective in CoVpipe2 was to reliably detect variants with high allelic depth and good read support. Low-frequency variants were not the primary focus of the pipeline, as the tool is intended to reconstruct robust consensus genomes from patient samples that can be used for genomic surveillance. However, if a user wants to use CoVpipe2 for different research questions, the implementation allows full customization of the necessary parameters (allele frequency, genotype adjustment, …) By screening the literature and examining other pipelines and community standards, we carefully selected the tools that performed best in our internal benchmarks for SARS-CoV-2 short-read data. Regarding variant calling, we first tested LoFreq (Wilm et al. 2012). Although very sensitive, LoFreq lacks a strong genotyping module that was crucial for our downstream processing of the called variants. Furthermore, the output files were hard to process (non-standard VCF formats). We implemented GATK as a second choice, which is a standard tool for eukaryotic genomic variant calling (McKenna et al . (2010), Van der Auwera & O'Connor (2020)) but was also shown to perform well on non-human targets (Lefouili et al. 2022). Performance and output standards were excellent, but it turned out that GATK misses a low amount of viral genomic variants in some samples, although multisample calling was employed. Single false negative variants, which we identified via a comprehensive investigation of the BAM files, were deemed to be too important to stick with the tool. Finally, we chose freebayes (Garrison et al . 2012), which excelled with high performance, high precision, and output files that were straightforward to process in downstream steps of the pipeline. In addition to the variant callers, there is indeed a large selection of quality processing tools. We opted for fastp rather than Trimmomatic because the processing speed is much faster, and the output quality is at least as good. It was shown, that “data filtered by Trimmomatic, SOAPNuke, Cutadapt and fastp were detected with 7174, 7040, 6942 and 6708 false positive variants respectively” (Chen et al . 2018), highlighting that fastp preprocessing can even improve the specificity of downstream analysis. In addition, fastp is equipped with a broader portfolio of parameter options. Another crucial step in read processing is primer clipping, which may promote mapping artifacts and dilution of variant calls if done before read alignment. If InDels occur close to the end of amplicons, a gap open penalty is more expensive than a few mismatches in mapping. For example, this caused trouble with the Spike DEL69/70 in several amplicon kits and made it necessary to use the artificial primer as a mapping anchor. Hence we replaced cutadapt, which clips adapters before mapping, with BAMclipper, which removes adapters after mapping. In our opinion, primer clipping should generally be performed after mapping in reference-based analysis. We, therefore, also discuss late primer clipping more prominently in connection with CoVpipe2. Furthermore, we fully agree that continuous benchmarking is a necessary process to adapt pipelines to changing wet lab procedures, adapted priming schemes, and pathogen evolution. If we find problems, we will also adapt CoVpipe2 accordingly and release new stable versions for reproducible research. We added the following text to the manuscript: “We carefully selected the bioinformatics tools integrated into CoVpipe2 based on internal benchmarks and in-depth manual reviews of sequencing data, mapping results, and called variants. Based on our hands-on experience with SARS-CoV-2 sequencing datasets during the pandemic, this approach ensured that tools that can robustly detect high allele depth and well-covered variants are used for detection. Despite the primary goal of CoVpipe2 to identify high-confidence variants to reconstruct robust consensus genomes for genomic surveillance, the pipeline also provides flexibility for adaptation to different research environments.” -------------------------------------------------------------------------------------------------------------------------- [2] Reviewer Question: the authors have done a very careful citation job, in general, but I think the following resources/specifications also deserve links/citations]: Author Response: Thanks for the comment. We fully agree and added more precise references for various sources such as Zenodo, the custom Kraken 2 database, BEDPE format, and Nextcalde/pangolin on anaconda.org. -------------------------------------------------------------------------------------------------------------------------- [3] Reviewer Question: In the introduction "While sequencing intensity and turnaround times on variant detection increased in different countries" is probably intended to mean increasing sequencing intensity, but *decreasing* turnaround times? Author Response: Yes, you are right; thanks for catching this. We changed the text accordingly: “While sequencing intensity increased and turnaround times on variant detection decreased in different countries, there are also major disparities between high-, low- and middle-income countries in the SARS-CoV-2 global genomic surveillance efforts.” -------------------------------------------------------------------------------------------------------------------------- [4] Reviewer Question: In the introduction, when DESH genomes statistics are given, the numbers should be 1.2 million and 1.1 million (period instead of comma), and in the discussion of Dataset 5 -> Lineages "for each CoVpip2-GISAD pair" has a typo in the pipeline name. Author Response: Thanks, we corrected that. Competing Interests: No competing interests were disclosed. Close Report a concern Respond or Comment COMMENTS ON THIS REPORT Author Response 15 May 2024 Martin Hölzer , Genome Competence Center (MF1), Robert Koch Institute, Berlin, Germany 15 May 2024 Author Response [1] Reviewer Question: For some steps in CoVpipe2 it seems there would have been several tools to choose from, and I'm wondering whether the authors of the pipeline have evaluated ... Continue reading [1] Reviewer Question: For some steps in CoVpipe2 it seems there would have been several tools to choose from, and I'm wondering whether the authors of the pipeline have evaluated alternatives. In particular, I'd be interested in learning why freebayes was chosen as the variant caller and why primer trimming is done with BAMclipper, as I don't think these two tools are used by many other comparable pipelines. If the authors had specific reasons to prefer these tools over alternatives, it might add to the value of the manuscript if these were added to the text. Author Response: Thanks for the question. You are absolutely right; as with almost any bioinformatics pipeline, there are different options when choosing tools for steps such as quality control, mapping, and variant calling. Our final selection of tools is based on our experience in analyzing sequencing data throughout the SARS-CoV-2 pandemic. Especially in the early days, we performed various internal benchmarks on SARS-CoV-2 Illumina data and manually investigated mapping results and variant calls together with colleagues from our expert unit for respiratory viruses. Thus, our main objective in CoVpipe2 was to reliably detect variants with high allelic depth and good read support. Low-frequency variants were not the primary focus of the pipeline, as the tool is intended to reconstruct robust consensus genomes from patient samples that can be used for genomic surveillance. However, if a user wants to use CoVpipe2 for different research questions, the implementation allows full customization of the necessary parameters (allele frequency, genotype adjustment, …) By screening the literature and examining other pipelines and community standards, we carefully selected the tools that performed best in our internal benchmarks for SARS-CoV-2 short-read data. Regarding variant calling, we first tested LoFreq (Wilm et al. 2012). Although very sensitive, LoFreq lacks a strong genotyping module that was crucial for our downstream processing of the called variants. Furthermore, the output files were hard to process (non-standard VCF formats). We implemented GATK as a second choice, which is a standard tool for eukaryotic genomic variant calling (McKenna et al . (2010), Van der Auwera & O'Connor (2020)) but was also shown to perform well on non-human targets (Lefouili et al. 2022). Performance and output standards were excellent, but it turned out that GATK misses a low amount of viral genomic variants in some samples, although multisample calling was employed. Single false negative variants, which we identified via a comprehensive investigation of the BAM files, were deemed to be too important to stick with the tool. Finally, we chose freebayes (Garrison et al . 2012), which excelled with high performance, high precision, and output files that were straightforward to process in downstream steps of the pipeline. In addition to the variant callers, there is indeed a large selection of quality processing tools. We opted for fastp rather than Trimmomatic because the processing speed is much faster, and the output quality is at least as good. It was shown, that “data filtered by Trimmomatic, SOAPNuke, Cutadapt and fastp were detected with 7174, 7040, 6942 and 6708 false positive variants respectively” (Chen et al . 2018), highlighting that fastp preprocessing can even improve the specificity of downstream analysis. In addition, fastp is equipped with a broader portfolio of parameter options. Another crucial step in read processing is primer clipping, which may promote mapping artifacts and dilution of variant calls if done before read alignment. If InDels occur close to the end of amplicons, a gap open penalty is more expensive than a few mismatches in mapping. For example, this caused trouble with the Spike DEL69/70 in several amplicon kits and made it necessary to use the artificial primer as a mapping anchor. Hence we replaced cutadapt, which clips adapters before mapping, with BAMclipper, which removes adapters after mapping. In our opinion, primer clipping should generally be performed after mapping in reference-based analysis. We, therefore, also discuss late primer clipping more prominently in connection with CoVpipe2. Furthermore, we fully agree that continuous benchmarking is a necessary process to adapt pipelines to changing wet lab procedures, adapted priming schemes, and pathogen evolution. If we find problems, we will also adapt CoVpipe2 accordingly and release new stable versions for reproducible research. We added the following text to the manuscript: “We carefully selected the bioinformatics tools integrated into CoVpipe2 based on internal benchmarks and in-depth manual reviews of sequencing data, mapping results, and called variants. Based on our hands-on experience with SARS-CoV-2 sequencing datasets during the pandemic, this approach ensured that tools that can robustly detect high allele depth and well-covered variants are used for detection. Despite the primary goal of CoVpipe2 to identify high-confidence variants to reconstruct robust consensus genomes for genomic surveillance, the pipeline also provides flexibility for adaptation to different research environments.” -------------------------------------------------------------------------------------------------------------------------- [2] Reviewer Question: the authors have done a very careful citation job, in general, but I think the following resources/specifications also deserve links/citations]: Author Response: Thanks for the comment. We fully agree and added more precise references for various sources such as Zenodo, the custom Kraken 2 database, BEDPE format, and Nextcalde/pangolin on anaconda.org. -------------------------------------------------------------------------------------------------------------------------- [3] Reviewer Question: In the introduction "While sequencing intensity and turnaround times on variant detection increased in different countries" is probably intended to mean increasing sequencing intensity, but *decreasing* turnaround times? Author Response: Yes, you are right; thanks for catching this. We changed the text accordingly: “While sequencing intensity increased and turnaround times on variant detection decreased in different countries, there are also major disparities between high-, low- and middle-income countries in the SARS-CoV-2 global genomic surveillance efforts.” -------------------------------------------------------------------------------------------------------------------------- [4] Reviewer Question: In the introduction, when DESH genomes statistics are given, the numbers should be 1.2 million and 1.1 million (period instead of comma), and in the discussion of Dataset 5 -> Lineages "for each CoVpip2-GISAD pair" has a typo in the pipeline name. Author Response: Thanks, we corrected that. [1] Reviewer Question: For some steps in CoVpipe2 it seems there would have been several tools to choose from, and I'm wondering whether the authors of the pipeline have evaluated alternatives. In particular, I'd be interested in learning why freebayes was chosen as the variant caller and why primer trimming is done with BAMclipper, as I don't think these two tools are used by many other comparable pipelines. If the authors had specific reasons to prefer these tools over alternatives, it might add to the value of the manuscript if these were added to the text. Author Response: Thanks for the question. You are absolutely right; as with almost any bioinformatics pipeline, there are different options when choosing tools for steps such as quality control, mapping, and variant calling. Our final selection of tools is based on our experience in analyzing sequencing data throughout the SARS-CoV-2 pandemic. Especially in the early days, we performed various internal benchmarks on SARS-CoV-2 Illumina data and manually investigated mapping results and variant calls together with colleagues from our expert unit for respiratory viruses. Thus, our main objective in CoVpipe2 was to reliably detect variants with high allelic depth and good read support. Low-frequency variants were not the primary focus of the pipeline, as the tool is intended to reconstruct robust consensus genomes from patient samples that can be used for genomic surveillance. However, if a user wants to use CoVpipe2 for different research questions, the implementation allows full customization of the necessary parameters (allele frequency, genotype adjustment, …) By screening the literature and examining other pipelines and community standards, we carefully selected the tools that performed best in our internal benchmarks for SARS-CoV-2 short-read data. Regarding variant calling, we first tested LoFreq (Wilm et al. 2012). Although very sensitive, LoFreq lacks a strong genotyping module that was crucial for our downstream processing of the called variants. Furthermore, the output files were hard to process (non-standard VCF formats). We implemented GATK as a second choice, which is a standard tool for eukaryotic genomic variant calling (McKenna et al . (2010), Van der Auwera & O'Connor (2020)) but was also shown to perform well on non-human targets (Lefouili et al. 2022). Performance and output standards were excellent, but it turned out that GATK misses a low amount of viral genomic variants in some samples, although multisample calling was employed. Single false negative variants, which we identified via a comprehensive investigation of the BAM files, were deemed to be too important to stick with the tool. Finally, we chose freebayes (Garrison et al . 2012), which excelled with high performance, high precision, and output files that were straightforward to process in downstream steps of the pipeline. In addition to the variant callers, there is indeed a large selection of quality processing tools. We opted for fastp rather than Trimmomatic because the processing speed is much faster, and the output quality is at least as good. It was shown, that “data filtered by Trimmomatic, SOAPNuke, Cutadapt and fastp were detected with 7174, 7040, 6942 and 6708 false positive variants respectively” (Chen et al . 2018), highlighting that fastp preprocessing can even improve the specificity of downstream analysis. In addition, fastp is equipped with a broader portfolio of parameter options. Another crucial step in read processing is primer clipping, which may promote mapping artifacts and dilution of variant calls if done before read alignment. If InDels occur close to the end of amplicons, a gap open penalty is more expensive than a few mismatches in mapping. For example, this caused trouble with the Spike DEL69/70 in several amplicon kits and made it necessary to use the artificial primer as a mapping anchor. Hence we replaced cutadapt, which clips adapters before mapping, with BAMclipper, which removes adapters after mapping. In our opinion, primer clipping should generally be performed after mapping in reference-based analysis. We, therefore, also discuss late primer clipping more prominently in connection with CoVpipe2. Furthermore, we fully agree that continuous benchmarking is a necessary process to adapt pipelines to changing wet lab procedures, adapted priming schemes, and pathogen evolution. If we find problems, we will also adapt CoVpipe2 accordingly and release new stable versions for reproducible research. We added the following text to the manuscript: “We carefully selected the bioinformatics tools integrated into CoVpipe2 based on internal benchmarks and in-depth manual reviews of sequencing data, mapping results, and called variants. Based on our hands-on experience with SARS-CoV-2 sequencing datasets during the pandemic, this approach ensured that tools that can robustly detect high allele depth and well-covered variants are used for detection. Despite the primary goal of CoVpipe2 to identify high-confidence variants to reconstruct robust consensus genomes for genomic surveillance, the pipeline also provides flexibility for adaptation to different research environments.” -------------------------------------------------------------------------------------------------------------------------- [2] Reviewer Question: the authors have done a very careful citation job, in general, but I think the following resources/specifications also deserve links/citations]: Author Response: Thanks for the comment. We fully agree and added more precise references for various sources such as Zenodo, the custom Kraken 2 database, BEDPE format, and Nextcalde/pangolin on anaconda.org. -------------------------------------------------------------------------------------------------------------------------- [3] Reviewer Question: In the introduction "While sequencing intensity and turnaround times on variant detection increased in different countries" is probably intended to mean increasing sequencing intensity, but *decreasing* turnaround times? Author Response: Yes, you are right; thanks for catching this. We changed the text accordingly: “While sequencing intensity increased and turnaround times on variant detection decreased in different countries, there are also major disparities between high-, low- and middle-income countries in the SARS-CoV-2 global genomic surveillance efforts.” -------------------------------------------------------------------------------------------------------------------------- [4] Reviewer Question: In the introduction, when DESH genomes statistics are given, the numbers should be 1.2 million and 1.1 million (period instead of comma), and in the discussion of Dataset 5 -> Lineages "for each CoVpip2-GISAD pair" has a typo in the pipeline name. Author Response: Thanks, we corrected that. Competing Interests: No competing interests were disclosed. Close Report a concern COMMENT ON THIS REPORT Comments on this article Comments (0) Version 2 VERSION 2 PUBLISHED 01 Sep 2023 ADD YOUR COMMENT Comment keyboard_arrow_left keyboard_arrow_right Open Peer Review Reviewer Status info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Reviewer Reports Invited Reviewers 1 2 Version 2 (revision) 16 Apr 24 read Version 1 01 Sep 23 read read Wolfgang Maier , Albert-Ludwigs-Universitat Freiburg, Freiburg, Germany Sondes Haddad-Boubaker , University of Tunis El Manar, Tunis, Tunisia Comments on this article All Comments (0) Add a comment Sign up for content alerts Sign Up You are now signed up to receive this alert Browse by related subjects keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2024 Haddad-Boubaker S. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 06 May 2024 | for Version 2 Sondes Haddad-Boubaker , University of Tunis El Manar, Tunis, Tunisia 0 Views copyright © 2024 Haddad-Boubaker S. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions The revision made is appropriate thus I approve of the paper in the current form. Thank you for publishing and sharing results. Competing Interests No competing interests were disclosed. Reviewer Expertise Virology I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. reply Respond to this report Responses (0) Haddad-Boubaker S. Peer Review Report For: Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.5256/f1000research.164888.r266843) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/12-1091/v2#referee-response-266843 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2024 Haddad-Boubaker S. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 17 Jan 2024 | for Version 1 Sondes Haddad-Boubaker , University of Tunis El Manar, Tunis, Tunisia 0 Views copyright © 2024 Haddad-Boubaker S. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (1) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions This paper presents a comprehensive bioinformatics workflow designed for the reconstruction of SARS-CoV-2 genomes using short-read sequencing data. The workflow description offers a detailed overview of the processes involved; however, there are opportunities for improvement to enhance the overall quality of the paper. 1-Organization: While the overall structure is clear, introducing numbering at both the heading and sub-heading levels would enhance clarity and facilitate better distinction between various steps in the workflow. 2-Technical Terminology: Given the diverse audience, including virologists and other biologists, it is crucial to ensure accessibility by providing clear definitions for all technical terms and acronyms. This will make the paper more easy to readers who may not be familiar with specific bioinformatics or genomics terminology. 3- References and Sources: Include proper references and sources for the tools and databases mentioned by providing specific URLs for easy access to external resources. 4-Methods and Results: - Introduce a dedicated "Pipeline Evaluation" subsection in both the "Methods" and "Results" sections and change the abstract section accordingly. -Include details on the investigated sequences, with a thorough description of sample types, cycle threshold (ct) values, and sample categories. -Consider expanding the evaluation by incorporating additional samples/sequences, especially failed sequences from samples with varying ct values (especially high ct values). The Investigation of the contribution of the pipeline in obtaining reliable sequences from samples with high ct values may be helpfull for virologist who are dealing with such challenges especially in samples obtained from long-term excretors. - Please Illustrate "common challenges" with a graph or figure for better presentation and understanding. 5- Discussion: - Emphasize and discuss the added value of the pipeline in obtaining reliable sequences from samples with high ct values. Compare the results obtained using this pipeline with those from other existing pipelines to highlight its superiority. - Provide detailed insights into how this pipeline can be applied to study other viruses, Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Partly Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Partly Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Partly Competing Interests No competing interests were disclosed. Reviewer Expertise Virology I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (1) Author Response 15 May 2024 Martin Hölzer, Genome Competence Center (MF1), Robert Koch Institute, Berlin, Germany (I) - Organization: [1] Reviewer Concern: While the overall structure is clear, introducing numbering at both the heading and sub-heading levels would enhance clarity and facilitate better distinction between various steps in the workflow. Author Response: Thanks for the suggestion. Numbering (sub)sections helps to structure the text better and distinguish which parts belong together semantically. However, there is little we can do about it, as this is the journal's style. Nevertheless, we will ask the editor/typesetting team if that’s possible. -------------------------------------------------------------------------------------------------------------------- (II) - Technical Terminology: [2] Reviewer Concern: Given the diverse audience, including virologists and other biologists, it is crucial to ensure accessibility by providing clear definitions for all technical terms and acronyms. This will make the paper more easy to readers who may not be familiar with specific bioinformatics or genomics terminology. Author Response: Thanks for the comment. We agreed and added a list of Abbreviations to the manuscript to make it easier for readers to follow the story. COVID-19 - Coronavirus disease 2019 SARS-CoV-2 - Severe acute respiratory syndrome coronavirus 2 GPL3 license - GNU General Public License GISAID - Global Initiative on Sharing All Influenza Data EBI - European Bioinformatics Institute EMBL - European Molecular Biology Laboratory RKI - Robert Koch Institute CorSurV - Coronavirus Surveillance Verordnung (eng., Coronavirus Surveillance Regulation) DESH - Deutscher Elektronischer Sequenzdaten-Hub (eng., German Electronic Sequence Data Hub) IMS-SC2 - Integrated Molecular Surveillance for SARS-CoV-2 ONT - Oxford Nanopore Technologies NGS - Next-Generation Sequencing HPC - High-Performance Computing WSL - Windows Subsystem for Linux CSV file - Comma-Separated Values file GFF file - General Feature Format file BEDPE file - Browser Extensible Data Paired-End file VCF file - Variant Call Format file HTML - Hypertext Markup Language BAM file - Binary Alignment and Map file BED file - Browser Extensible Data file CCO license - Creative Commons Zero license VOC - Variants of Concern VOI - Variants of Interest IUPAC - International Union of Pure and Applied Chemistry indel - Insertion/Deletion Variant JSON file - JavaScript Object Notation file CDC - Centers for Disease Control and Prevention ENA - European Nucleotide Archive QC - Quality Control PCR - Polymerase Chain Reaction ID - Identifier SRA - Sequence Read Archive -------------------------------------------------------------------------------------------------------------------- (III) - References and Sources [3] Reviewer Concern: Include proper references and sources for the tools and databases mentioned by providing specific URLs for easy access to external resources. Author Response: We agree that it’s crucial to acknowledge all tools and resources properly. When there is an original publication for a tool or database, we cite the publication. If not, we cite the code repository or the URL to the resource (such as GitHub or Zenodo). We carefully checked the text again and added citations/URLs if they were missing and necessary. For example, Rev #1 also commented that the specific URL to the custom Kraken 2 database on Zenodo was missing. We added that. -------------------------------------------------------------------------------------------------------------------- (IV) - Methods and Results [4] Reviewer Concern : Introduce a dedicated "Pipeline Evaluation" subsection in both the "Methods" and "Results" sections and change the abstract section accordingly. Author Response: We fully agree that presenting the implementation of a pipeline together with its evaluation with additional analysis can be confusing. So what are the "Methods" and the "Results" parts, then? This is often a problem when simultaneously presenting and evaluating a new software implementation. But we also need to adhere to the style guidelines for journals. We wrote a “Software Tool Articles” and have to follow this structure: https://f1000research.com/for-authors/article-guidelines/software-tool-articles . As you can see in these guidelines, “Software Tool Articles typically contain the following sections: Introduction, Methods, Results (Optional), Use Cases (Optional), Conclusions/Discussion.” In the first version, we skipped the “Use Cases” section to discuss our example data sets for pipeline evaluation directly in the “Results” section. However, thanks to your comment, we believe that also skipping the “Results” section entirely makes our manuscript clearer. We deleted the “Results” section and added the subsections “Selection of benchmark datasets and pipeline evaluation” and “Reporting” at the end of the “Methods”. We changed the subsection “Selection of benchmark datasets” to “Selection of benchmark datasets and pipeline evaluation” to include your suggestion. We think that the structure is now clearer because we first describe in the “Methods” the implementation of the pipeline and how to operate it, according to the journal guidelines for “Software Tool Articles”: The Methods should “Include a subsection on Implementation describing how the tool works and any relevant technical details required for implementation; and a subsection on Operation , which should include the minimal system requirements needed to run the software and an overview of the workflow.” Then, we describe some specific implementation decisions followed by the example data sets for pipeline evaluation and, finally, the report structure we implemented. According to the journal guideline for “Software Tool Articles”: “Abstracts are structured into Background, Methods, Results, and Conclusions”, thus, we can not change sections in the abstract. Although we generally agree that different subheadings would help follow the story, we must also stick to the journal guidelines. -------------------------------------------------------------------------------------------------------------------- [5] Reviewer Concern : Include details on the investigated sequences, with a thorough description of sample types, cycle threshold (ct) values, and sample categories. Author Response: We only use publicly available data sets and reference the original sources (publication, GitHub, and ENA repositories). We suggest that readers should refer to the original sources for further details. However, the description of the benchmark datasets is now also part of the “Methods”. Here, we describe: “We compared the results of CoVpipe2 (v0.4.0) with publicly available benchmark datasets for SARS-CoV-2 surveillance 57 ( GitHub CDC data ).” [57] Xiaoli L, Hagey JV, Park DJ, et al.: Benchmark datasets for sars-cov-2 surveillance bioinformatics. PeerJ. 2022;10:e13821. 10.7717/peerj.13821 We do not want to mirror the details of the benchmark data sets that are described in the original sources. In addition, we can not provide additional information, such as Ct values, because, to the best of our knowledge, this information is not available in the original publication or in the data source (GitHub, ENA). -------------------------------------------------------------------------------------------------------------------- [6] Reviewer Concern : Consider expanding the evaluation by incorporating additional samples/sequences, especially failed sequences from samples with varying ct values (especially high ct values). The Investigation of the contribution of the pipeline in obtaining reliable sequences from samples with high ct values may be helpful for virologist who are dealing with such challenges especially in samples obtained from long-term excretors. Author Response: Thanks for the comment. We agree that investigating challenging samples is especially interesting for users of CoVpipe2. As described above (Q [5]), we selected a publicly available benchmark dataset for SARS-CoV-2 surveillance (Xiaoli et al. 2022) to compare our results directly with previous calculations. In addition, this data set also includes difficult samples that should not withstand automatic quality control (QC) and could mimic high Ct values. CoVpipe2 was developed as a robust and standardized workflow to support genome reconstruction in genomic surveillance programs. Thus, our main goal in developing CoVpipe2 was to provide a robust bioinformatics pipeline for short-read sequencing data that recognizes important mutations with decent allele frequency and automatically identifies and masks ambiguous positions. We implemented parameters (20X coverage to consider a position for variant calling, 90% ACGT nucleotide identity to the reference) to discover low-quality samples that might originate from high Ct values. Running the pipeline on amplicon sequencing data from samples with high Ct can result in "read stacks" with high sequence depth for certain well-amplified amplicons, but it could also lead to low horizontal genome coverage due to low input RNA quantity. CoVpipe2 will report such samples as “failed” in the QC report. Thus, only samples with a decent vertical (sequencing depth) and horizontal genome coverage should be used for downstream genomic surveillance and trustworthy lineage assignment. Nevertheless, CoVpipe2 also reports the full intermediate results, such as BAM files and unfiltered VCF files. Experienced users can investigate all variant calls and their respective allele frequencies - also for QC-failed samples. Thus, it is also possible to investigate mixed variant calls (co-infection, recombinants) and low-frequency variants with the help of CoVpipe2. However, for routine genomic surveillance applications, such challenging samples will be automatically flagged as QC-failed in the pipeline, supporting non-expert users in decision-making and selecting suitable samples for surveillance. Obtaining reliable consensus genomes from high Ct samples is generally difficult. In our experience, it is better to flag such samples with a warning and inform the user that those are of lower quality and probably not suited for further downstream analysis. Thus, CoVpipe2 helps virologists identify such challenging samples so that they can be selected for re-sequencing or exclusion from downstream analysis. -------------------------------------------------------------------------------------------------------------------- [7] Reviewer Concern : Please Illustrate "common challenges" with a graph or figure for better presentation and understanding. Author Response: Here, we present a bioinformatics pipeline with the specific objective of reconstructing robust SARS-CoV-2 consensus genomes from patient samples and short-read data, which is also reflected in the title of our paper. We present CoVpipe2 as a solution to overcome such challenges in reconstructing robust SARS-CoV-2 genomes from short-read (amplicon) data. In Figure 2, we explicitly illustrate common challenges regarding variant calling, which is one of the main obstacles in many reference-based virus bioinformatics pipelines, and where we specifically integrated solutions in CoVpipe2 to overcome such challenges. Besides the dedicated figure for variant calling challenges, we examine other relevant challenges in the context of the CoVpipe2 implementation, such as amplicon drop-outs, in the text. For a general overview, we think that common challenges in the context of amplicon sequencing and virus bioinformatics need to be more broadly addressed in dedicated benchmark studies such as those already available from Beerenwinkel et al. 2012; Murray et al. 2015; Fitzpatrick et al. 2022; Liu et al. 2021. -------------------------------------------------------------------------------------------------------------------- (V) - Discussion: [8] Reviewer Concern : Emphasize and discuss the added value of the pipeline in obtaining reliable sequences from samples with high ct values. Compare the results obtained using this pipeline with those from other existing pipelines to highlight its superiority. Author Response: As described in [6], we developed CoVpipe2 as a robust and modular surveillance pipeline focusing on amplification protocols and short-read data. Thus, for samples with high Ct values and where amplification can not yield enough output, CoVpipe2 will mark them as “failed” in the reporting. High Ct samples will usually result in regions (amplicons) with low coverage. Such regions are then automatically masked by “N” bases in the final consensus. We don't think a bioinformatics pipeline should construct any “reliable” consensus genome sequence when the data is insufficient. Thus, it is more important to identify such low-quality samples and flag them with a user warning. Our filtering and reporting aims to fit the needs of large-scale surveillance programs with detailed QC information and provide a quick overview of sample results, to identify such challenging samples easily. Thus, CoVpipe2 helps virologists to identify such problematic samples so that they can be selected for re-sequencing or excluded from downstream analysis. Regarding the comparison to other pipelines, we implicitly did that by selecting the test data sets. Those come from another independent benchmark study (Xiaoli et al. 2022), and we compare our CoVpipe2 results against those from the original benchmark paper. The original authors wrote in their publication: > The datasets presented here were generated to help public health laboratories build sequencing and bioinformatics capacity, benchmark different workflows and pipelines, and calibrate QC thresholds to ensure sequencing quality. All available pipelines (Tab. 1 in manuscript) excel in various properties. While some strive to have high detection rates for minor variants for research settings, CoVpipe2 was developed to be easily extendable and adjustable to new requirements in surveillance or other viruses [see 9]. Thus, we would like to stick to our decision of utilizing a publicly available and carefully constructed, independent benchmark data set instead of including more samples and pipelines. Our study focuses on presenting the CoVpipe2 implementation and highlighting various implementation decisions in the context of reconstructing robust genome sequences for surveillance tasks. Nevertheless, we agree that another large-scale and up-to-date benchmark study comparing all available pipelines (Tab. 1), including CoVpipe2, would be interesting but is beyond the scope of our Software article. -------------------------------------------------------------------------------------------------------------------- [9] Reviewer Concern : Provide detailed insights into how this pipeline can be applied to study other viruses Author Response: The predecessor of CoVpipe2, the snakemake pipeline CoVpipe1, was used to create adapted pipelines for RSV and Influenza. In this context, we discovered that other viruses might need other tools and parameters to reflect their characteristics (genome size, segmentation, reference selection). Also, the needs for final reporting may differ depending on the virus under investigation, and changes in the pipeline may be necessary. Besides, we successfully used CoVpipe2 on Polio and Measles viruses for genome reconstruction and variant calling from short-read sequencing data. In short, for Polio viruses, we sequenced the same 24 samples with Sanger, Illumina, and Nanopore, and CoVpipe2 was able to identify the same variants compared to the other sequencing technologies and associated bioinformatic steps (unpublished preliminary data). In general, the basic software framework - the generic sub-processes of raw data quality control, read alignment, variant calling, and consensus building - and the bioinformatic challenges for data derived from amplicon sequencing are equally applicable to other viruses. CoVpipe2 can serve as a blueprint for other pathogens, especially other unsegmented viruses, and provide first insights into the variants and consensus sequences. Tools, parameters, and thresholds might need careful adjustments depending on the pathogen. Similarly, the downstream analysis might be pathogen-specific, e.g., require pathogen-specific datasets (such as reference sequences for Influenza from Nextclade). We are currently working on a harmonized multi-pathogen pipeline with different profiles (tools, parameter settings) tailored towards specific viruses. Besides, interested users can already run CoVpipe2 on other (non-segmented) viruses by simply switching to another reference genome as we did before successfully for analyzing Polio virus amplicon data (parameter --ref_genome). We extend the “Conclusion” accordingly: “We used CoVpipe1 and CoVpipe2, which were initially developed for SARS-CoV-2, to reconstruct the genomes of other viruses. By selecting different reference genomes for polio, measles, RSV, and influenza viruses and modifying the analysis processes for the latter two viruses, we were able to demonstrate the workflow's potential as a universal blueprint for viral genome analysis. This adaptability has been demonstrated by the ability to identify consistent variants and the successful application to different viral characteristics, such as varying genome size and segmentation. However, it should also be noted that for segmented viruses, more customization is required than simply replacing the (non-segmented) reference genome. However, our experience suggests that customization of tools, parameters, and reporting requirements is needed for each virus, pointing to developing a harmonized pipeline for multiple pathogens. CoVpipe2 thus proves to be a robust tool for SARS-CoV-2 and paves the way for broad application in virology research by highlighting its ability to serve as a fundamental framework for building customized pipelines for a wide range of pathogens.” View more View less Competing Interests No competing interests were disclosed. reply Respond Report a concern Haddad-Boubaker S. Peer Review Report For: Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.5256/f1000research.149827.r233668) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/12-1091/v1#referee-response-233668 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2023 Maier W. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 20 Sep 2023 | for Version 1 Wolfgang Maier , Bioinformatics Group, Department of Computer Science, Albert-Ludwigs-Universitat Freiburg, Freiburg, Baden-Württemberg, Germany 0 Views copyright © 2023 Maier W. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (1) Approved info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions The manuscript by Lataretu et al. describes CoVpipe2, a bioinformatics pipeline for constructing viral consensus sequences from SARS-CoV-2 short sequenced reads obtained using the Illumina platform. It discusses the current state of the pipeline, various design decisions that have led to that state, and typical analysis pitfalls that the authors hope to overcome with their pipeline design. The pipeline itself is implemented as a Nextflow pipeline and comes under a free and open-source license, which means that its exact steps and parameters can be explored down to any desired level of detail. Still the authors provide a very helpful overview in the manuscript text and in Figure 1 through both of which the reader can gain a good understanding of the pipeline layout and its components. The authors point out the existence of multiple alternative analysis pipelines with similar scope as CoVpipe2 and list many of them in Table 1. I would expect most of these alternative pipelines to produce consensus genomes of similar quality as CoVpipe2, and most of the steps that CoVpipe2 is composed of are relatively standard in the field. From this perspective, the manuscript could be said to lack novelty. Beyond the core steps shared in similar form with many other pipelines, there are, however, some smart extra steps built into CoVpipe2 that are innovative ideas, like screening steps for mixed-infection and recombinant samples and consensus genome annotation with Liftoff. More importantly, however, the authors are not just advertising yet another pipeline for SARS-CoV-2 genome analysis, but their manuscript is the kind of documentation that you wish every such pipeline came with: it explains not only individual analysis steps, but also their purpose, special analysis tweaks found to be necessary, and provides links to all relevant resources. In summary, the manuscript describes a robust and mature resource for reproducible data analysis and does an excellent job at that. I enjoyed reading it and even though I've spent a considerable amount of time on developing similar pipelines I still picked up a few new ideas from it. I have no concerns regarding publication of this valuable manuscript, just one comment that the authors may wish to address in the manuscript directly or in a separate reply: For some steps in CoVpipe2 it seems there would have been several tools to choose from, and I'm wondering whether the authors of the pipeline have evaluated alternatives. In particular, I'd be interested in learning why freebayes was chosen as the variant caller and why primer trimming is done with BAMclipper, as I don't think these two tools are used by many other comparable pipelines. If the authors had specific reasons to prefer these tools over alternatives, it might add to the value of the manuscript if these were added to the text. Beyond that, I have found a small number of inconsistencies and typos that I think should be fixed before publication: 1. the authors have done a very careful citation job, in general, but I think the following resources/specifications also deserve links/citations: Zenodo The "precalculated Kraken2 database" deposited at Zenodo The BEDPE format ( https://bedtools.readthedocs.io/en/latest/content/general-usage.html#bedpe-format ) Instead of providing a general anaconda.org link (as done twice in the Methods section), it would be more helpful to provide direct links to the latest versions of pangolin and nextclade ( https://anaconda.org/bioconda/pangolin and https://anaconda.org/bioconda/nextclade ), which also includes channel information. 2. In the introduction " While sequencing intensity and turnaround times on variant detection increased in different countries " is probably intended to mean increasing sequencing intensity, but *decreasing* turnaround times? 3. In the introduction, when DESH genomes statistics are given, the numbers should be 1.2 million and 1.1 million (period instead of comma), and in the discussion of Dataset 5 -> Lineages " for each CoVpip2-GISAD pair " has a typo in the pipeline name. Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Yes Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Yes Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes Competing Interests No competing interests were disclosed. Reviewer Expertise Bioinformatics, pathogen genomics, genetics I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. reply Respond to this report Responses (1) Author Response 15 May 2024 Martin Hölzer, Genome Competence Center (MF1), Robert Koch Institute, Berlin, Germany [1] Reviewer Question: For some steps in CoVpipe2 it seems there would have been several tools to choose from, and I'm wondering whether the authors of the pipeline have evaluated alternatives. In particular, I'd be interested in learning why freebayes was chosen as the variant caller and why primer trimming is done with BAMclipper, as I don't think these two tools are used by many other comparable pipelines. If the authors had specific reasons to prefer these tools over alternatives, it might add to the value of the manuscript if these were added to the text. Author Response: Thanks for the question. You are absolutely right; as with almost any bioinformatics pipeline, there are different options when choosing tools for steps such as quality control, mapping, and variant calling. Our final selection of tools is based on our experience in analyzing sequencing data throughout the SARS-CoV-2 pandemic. Especially in the early days, we performed various internal benchmarks on SARS-CoV-2 Illumina data and manually investigated mapping results and variant calls together with colleagues from our expert unit for respiratory viruses. Thus, our main objective in CoVpipe2 was to reliably detect variants with high allelic depth and good read support. Low-frequency variants were not the primary focus of the pipeline, as the tool is intended to reconstruct robust consensus genomes from patient samples that can be used for genomic surveillance. However, if a user wants to use CoVpipe2 for different research questions, the implementation allows full customization of the necessary parameters (allele frequency, genotype adjustment, …) By screening the literature and examining other pipelines and community standards, we carefully selected the tools that performed best in our internal benchmarks for SARS-CoV-2 short-read data. Regarding variant calling, we first tested LoFreq (Wilm et al. 2012). Although very sensitive, LoFreq lacks a strong genotyping module that was crucial for our downstream processing of the called variants. Furthermore, the output files were hard to process (non-standard VCF formats). We implemented GATK as a second choice, which is a standard tool for eukaryotic genomic variant calling (McKenna et al . (2010), Van der Auwera & O'Connor (2020)) but was also shown to perform well on non-human targets (Lefouili et al. 2022). Performance and output standards were excellent, but it turned out that GATK misses a low amount of viral genomic variants in some samples, although multisample calling was employed. Single false negative variants, which we identified via a comprehensive investigation of the BAM files, were deemed to be too important to stick with the tool. Finally, we chose freebayes (Garrison et al . 2012), which excelled with high performance, high precision, and output files that were straightforward to process in downstream steps of the pipeline. In addition to the variant callers, there is indeed a large selection of quality processing tools. We opted for fastp rather than Trimmomatic because the processing speed is much faster, and the output quality is at least as good. It was shown, that “data filtered by Trimmomatic, SOAPNuke, Cutadapt and fastp were detected with 7174, 7040, 6942 and 6708 false positive variants respectively” (Chen et al . 2018), highlighting that fastp preprocessing can even improve the specificity of downstream analysis. In addition, fastp is equipped with a broader portfolio of parameter options. Another crucial step in read processing is primer clipping, which may promote mapping artifacts and dilution of variant calls if done before read alignment. If InDels occur close to the end of amplicons, a gap open penalty is more expensive than a few mismatches in mapping. For example, this caused trouble with the Spike DEL69/70 in several amplicon kits and made it necessary to use the artificial primer as a mapping anchor. Hence we replaced cutadapt, which clips adapters before mapping, with BAMclipper, which removes adapters after mapping. In our opinion, primer clipping should generally be performed after mapping in reference-based analysis. We, therefore, also discuss late primer clipping more prominently in connection with CoVpipe2. Furthermore, we fully agree that continuous benchmarking is a necessary process to adapt pipelines to changing wet lab procedures, adapted priming schemes, and pathogen evolution. If we find problems, we will also adapt CoVpipe2 accordingly and release new stable versions for reproducible research. We added the following text to the manuscript: “We carefully selected the bioinformatics tools integrated into CoVpipe2 based on internal benchmarks and in-depth manual reviews of sequencing data, mapping results, and called variants. Based on our hands-on experience with SARS-CoV-2 sequencing datasets during the pandemic, this approach ensured that tools that can robustly detect high allele depth and well-covered variants are used for detection. Despite the primary goal of CoVpipe2 to identify high-confidence variants to reconstruct robust consensus genomes for genomic surveillance, the pipeline also provides flexibility for adaptation to different research environments.” -------------------------------------------------------------------------------------------------------------------------- [2] Reviewer Question: the authors have done a very careful citation job, in general, but I think the following resources/specifications also deserve links/citations]: Author Response: Thanks for the comment. We fully agree and added more precise references for various sources such as Zenodo, the custom Kraken 2 database, BEDPE format, and Nextcalde/pangolin on anaconda.org. -------------------------------------------------------------------------------------------------------------------------- [3] Reviewer Question: In the introduction "While sequencing intensity and turnaround times on variant detection increased in different countries" is probably intended to mean increasing sequencing intensity, but *decreasing* turnaround times? Author Response: Yes, you are right; thanks for catching this. We changed the text accordingly: “While sequencing intensity increased and turnaround times on variant detection decreased in different countries, there are also major disparities between high-, low- and middle-income countries in the SARS-CoV-2 global genomic surveillance efforts.” -------------------------------------------------------------------------------------------------------------------------- [4] Reviewer Question: In the introduction, when DESH genomes statistics are given, the numbers should be 1.2 million and 1.1 million (period instead of comma), and in the discussion of Dataset 5 -> Lineages "for each CoVpip2-GISAD pair" has a typo in the pipeline name. Author Response: Thanks, we corrected that. View more View less Competing Interests No competing interests were disclosed. reply Respond Report a concern Maier W. Peer Review Report For: Lessons learned: overcoming common challenges in reconstructing the SARS-CoV-2 genome from short-read sequencing data via CoVpipe2 [version 2; peer review: 2 approved] . F1000Research 2024, 12 :1091 ( https://doi.org/10.5256/f1000research.149827.r203328) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/12-1091/v1#referee-response-203328 Alongside their report, reviewers assign a status to the article: Approved - the paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations - A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved - fundamental flaws in the paper seriously undermine the findings and conclusions Adjust parameters to alter display View on desktop for interactive features Includes Interactive Elements View on desktop for interactive features Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Stay Updated Sign up for content alerts and receive a weekly or monthly email with all newly published articles Register with F1000Research Already registered? Sign in Not now, thanks close PLEASE NOTE If you are an AUTHOR of this article, please check that you signed in with the account associated with this article otherwise we cannot automatically identify your role as an author and your comment will be labelled as a “User Comment”. If you are a REVIEWER of this article, please check that you have signed in with the account associated with this article and then go to your account to submit your report, please do not post your review here. If you do not have access to your original account, please contact us . All commenters must hold a formal affiliation as per our Policies . The information that you give us will be displayed next to your comment. User comments must be in English, comprehensible and relevant to the article under discussion. We reserve the right to remove any comments that we consider to be inappropriate, offensive or otherwise in breach of the User Comment Terms and Conditions . Commenters must not use a comment for personal attacks. When criticisms of the article are based on unpublished data, the data should be made available. I accept the User Comment Terms and Conditions Please confirm that you accept the User Comment Terms and Conditions. Affiliation ✕ refresh Please enter your institution. Note: To add your institution or organisation, start typing the name and then select the correct name from the list. Where applicable, the name will appear in both the original language and in English. Do not paste in the name. If the name does not appear in the drop-down list, we will display the information you have entered. ✕ refresh Country/Region * USA UK Canada China France Germany Afghanistan Aland Islands Albania Algeria American Samoa Andorra Angola Anguilla Antarctica Antigua and Barbuda Argentina Armenia Aruba Australia Austria Azerbaijan Bahamas Bahrain Bangladesh Barbados Belarus Belgium Belize Benin Bermuda Bhutan Bolivia Bosnia and Herzegovina Botswana Bouvet Island Brazil British Indian Ocean Territory British Virgin Islands Brunei Bulgaria Burkina Faso Burundi Cambodia Cameroon Canada Cape Verde Cayman Islands Central African Republic Chad Chile China Christmas Island Cocos (Keeling) Islands Colombia Comoros Congo Cook Islands Costa Rica Cote d'Ivoire Croatia Cuba Cyprus Czech Republic Democratic Republic of the Congo Denmark Djibouti Dominica Dominican Republic Ecuador Egypt El Salvador Equatorial Guinea Eritrea Estonia Ethiopia Falkland Islands Faroe Islands Federated States of Micronesia Fiji Finland France French Guiana French Polynesia French Southern Territories Gabon Georgia Germany Ghana Gibraltar Greece Greenland Grenada Guadeloupe Guam Guatemala Guernsey Guinea Guinea-Bissau Guyana Haiti Heard Island and Mcdonald Islands Holy See (Vatican City State) Honduras Hong Kong Hungary Iceland India Indonesia Iran Iraq Ireland Israel Italy Jamaica Japan Jersey Jordan Kazakhstan Kenya Kiribati Kosovo (Serbia and Montenegro) Kuwait Kyrgyzstan Lao People's Democratic Republic Latvia Lebanon Lesotho Liberia Libya Liechtenstein Lithuania Luxembourg Macao Madagascar Malawi Malaysia Maldives Mali Malta Marshall Islands Martinique Mauritania Mauritius Mayotte Mexico Minor Outlying Islands of the United States Moldova Monaco Mongolia Montenegro Montserrat Morocco Mozambique Myanmar Namibia Nauru Nepal Netherlands Antilles New Caledonia New Zealand Nicaragua Niger Nigeria Niue Norfolk Island North Korea North Macedonia Northern Mariana Islands Norway Oman Pakistan Palau Palestinian Territory Panama Papua New Guinea Paraguay Peru Philippines Pitcairn Poland Portugal Puerto Rico Qatar Reunion Romania Russian Federation Rwanda Saint Helena Saint Kitts and Nevis Saint Lucia Saint Pierre and Miquelon Saint Vincent and the Grenadines Samoa San Marino Sao Tome and Principe Saudi Arabia Senegal Serbia Seychelles Sierra Leone Singapore Slovakia Slovenia Solomon Islands Somalia South Africa South Georgia and the South Sandwich Is South Korea South Sudan Spain Sri Lanka Sudan Suriname Svalbard and Jan Mayen Swaziland Sweden Switzerland Syria Taiwan Tajikistan Tanzania Thailand The Gambia The Netherlands Timor-Leste Togo Tokelau Tonga Trinidad and Tobago Tunisia Turkey Turkmenistan Turks and Caicos Islands Tuvalu UK USA Uganda Ukraine United Arab Emirates United States Virgin Islands Uruguay Uzbekistan Vanuatu Venezuela Vietnam Wallis and Futuna West Bank and Gaza Strip Western Sahara Yemen Zambia Zimbabwe Please select your country/region. You must enter a comment. Competing Interests Please disclose any competing interests that might be construed to influence your judgment of the article's or peer review report's validity or importance. Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Please state your competing interests The comment has been saved. An error has occurred. Please try again. Cancel Post var lTitle = "Lessons learned: overcoming common challenges...".replace("'", ''); var linkedInUrl = "http://www.linkedin.com/shareArticle?url=https://f1000research.com/articles/12-1091/v2" + "&title=" + encodeURIComponent(lTitle) + "&summary=" + encodeURIComponent('Read the article by '); var deliciousUrl = "https://del.icio.us/post?url=https://f1000research.com/articles/12-1091/v2&title=" + encodeURIComponent(lTitle); var redditUrl = "http://reddit.com/submit?url=https://f1000research.com/articles/12-1091/v2" + "&title=" + encodeURIComponent(lTitle); linkedInUrl += encodeURIComponent('Lataretu M et al.'); var offsetTop = /chrome/i.test( navigator.userAgent ) ? 4 : -10; var addthis_config = { ui_offset_top: offsetTop, services_compact : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_expanded : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_custom : [ { name: "LinkedIn", url: linkedInUrl, icon:"/img/icon/at_linkedin.svg" }, { name: "Mendeley", url: "http://www.mendeley.com/import/?url=https://f1000research.com/articles/12-1091/v2/mendeley", icon:"/img/icon/at_mendeley.svg" }, { name: "Reddit", url: redditUrl, icon:"/img/icon/at_reddit.svg" }, ] }; var addthis_share = { url: "https://f1000research.com/articles/12-1091", templates : { twitter : "Lessons learned: overcoming common challenges in reconstructing.... Lataretu M et al., published by " + "@F1000Research" + ", https://f1000research.com/articles/12-1091/v2" } }; if (typeof(addthis) != "undefined"){ addthis.addEventListener('addthis.ready', checkCount); addthis.addEventListener('addthis.menu.share', checkCount); } $(".f1r-shares-twitter").attr("href", "https://twitter.com/intent/tweet?text=" + addthis_share.templates.twitter); $(".f1r-shares-facebook").attr("href", "https://www.facebook.com/sharer/sharer.php?u=" + addthis_share.url); $(".f1r-shares-linkedin").attr("href", addthis_config.services_custom[0].url); $(".f1r-shares-reddit").attr("href", addthis_config.services_custom[2].url); $(".f1r-shares-mendelay").attr("href", addthis_config.services_custom[1].url); function checkCount(){ setTimeout(function(){ $(".addthis_button_expanded").each(function(){ var count = $(this).text(); if (count !== "" && count != "0") $(this).removeClass("is-hidden"); else $(this).addClass("is-hidden"); }); }, 1000); } close How to cite this report {{reportCitation}} Cancel Copy Citation Details $(function(){R.ui.buttonDropdowns('.dropdown-for-downloads');}); $(function(){R.ui.toolbarDropdowns('.toolbar-dropdown-for-downloads');}); $.get("/articles/acj/136683/164888") new F1000.Clipboard(); new F1000.ThesaurusTermsDisplay("articles", "article", "164888"); $(document).ready(function() { $( "#frame1" ).on('load', function() { var mydiv = $(this).contents().find("div"); var h = mydiv.height(); console.log(h) }); var tooltipLivingFigure = jQuery(".interactive-living-figure-label .icon-more-info"), titleLivingFigure = tooltipLivingFigure.attr("title"); tooltipLivingFigure.simpletip({ fixed: true, position: ["-115", "30"], baseClass: 'small-tooltip', content:titleLivingFigure + " " }); tooltipLivingFigure.removeAttr("title"); $("body").on("click", ".cite-living-figure", function(e) { e.preventDefault(); var ref = $(this).attr("data-ref"); $(this).closest(".living-figure-list-container").find("#" + ref).fadeIn(200); }); $("body").on("click", ".close-cite-living-figure", function(e) { e.preventDefault(); $(this).closest(".popup-window-wrapper").fadeOut(200); }); $(document).on("mouseup", function(e) { var metricsContainer = $(".article-metrics-popover-wrapper"); if (!metricsContainer.is(e.target) && metricsContainer.has(e.target).length === 0) { $(".article-metrics-close-button").click(); } }); var articleId = $('#articleId').val(); if($("#main-article-count-box").attachArticleMetrics) { $("#main-article-count-box").attachArticleMetrics(articleId, { articleMetricsView: true }); } }); var figshareWidget = $(".new_figshare_widget"); if (figshareWidget.length > 0) { window.figshare.load("f1000", function(Widget) { // Select a tag/tags defined in your page. In this tag we will place the widget. _.map(figshareWidget, function(el){ var widget = new Widget({ articleId: $(el).attr("figshare_articleId") //height:300 // this is the height of the viewer part. [Default: 550] }); widget.initialize(); // initialize the widget widget.mount(el); // mount it in a tag that's on your page // this will save the widget on the global scope for later use from // your JS scripts. This line is optional. //window.widget = widget; }); }); } close Error Close Add Reset F1000.MICROSERVICES.AFFILIATION = ''; $(document).ready(function () { $('.js-affiliations-form').each((index, form) => { new AffiliationForm({ formId: form.id, institutionErrorSelector: '.comment-enter-institution', departmentErrorSelector: '.comment-enter-department', placeSelector: '.js-add-comment-place', stateSelector: '.js-add-comment-state', zipCodeSelector: '.js-add-comment-zipcode', countrySelector: '.js-add-comment-country', countryErrorSelector: '.comment-enter-country', }); }); }); $(document).ready(function () { var reportIds = { "203331": 0, "233667": 0, "203330": 0, "233666": 0, "203329": 0, "233665": 0, "203328": 37, "233671": 0, "233670": 0, "233669": 0, "233668": 26, "233674": 0, "233673": 0, "233672": 0, "229327": 0, "229326": 0, "229325": 0, "229331": 0, "229330": 0, "229329": 0, "229328": 0, "229334": 0, "229333": 0, "229332": 0, "266842": 0, "266843": 10, }; $(".referee-response-container,.js-referee-report").each(function(index, el) { var reportId = $(el).attr("data-reportid"), reportCount = reportIds[reportId] || 0; $(el).find(".comments-count-container,.js-referee-report-views").html(reportCount); }); var uuidInput = $("#article_uuid"), oldUUId = uuidInput.val(), newUUId = "c181f349-5e49-40f1-8b73-5684aa12e913"; uuidInput.val(newUUId); $("a[href*='article_uuid=']").each(function(index, el) { var newHref = $(el).attr("href").replace(oldUUId, newUUId); $(el).attr("href", newHref); }); }); An innovative open access publishing platform offering rapid publication and open peer review, whilst supporting data deposition and sharing. Browse Gateways Collections How it Works Contact For Developers Cookie Notice Privacy Notice RSS Submit Your Research Follow us © 2012-2026 F1000 Research Ltd. ISSN 2046-1402 | Legal | Partner of Research4Life • CrossRef • ORCID • FAIRSharing R.templateTests.simpleTemplate = R.template(' $text $text $text $text $text '); R.templateTests.runTests(); var F1000platform = new F1000.Platform({ name: "f1000research", displayName: "F1000Research", hostName: "f1000research.com", id: "1", editorialEmail: "
[email protected]", infoEmail: "
[email protected]", usePmcStats: true }); $(function(){R.ui.dropdowns('.dropdown-for-authors, .dropdown-for-about, .dropdown-for-myresearch');}); // $(function(){R.ui.dropdowns('.dropdown-for-referees');}); $(document).ready(function () { if ($(".cookie-warning").is(":visible")) { $(".sticky").css("margin-bottom", "35px"); $(".devices").addClass("devices-and-cookie-warning"); } $(".cookie-warning .close-button").click(function (e) { $(".devices").removeClass("devices-and-cookie-warning"); $(".sticky").css("margin-bottom", "0"); }); $("#tweeter-feed .tweet-message").each(function (i, message) { var self = $(message); self.html(linkify(self.html())); }); $(".partner").on("mouseenter mouseleave", function() { $(this).find(".gray-scale, .colour").toggleClass("is-hidden"); }); }); Sign In Remember me Forgotten your password? Sign In Cancel Email or password not correct. Please try again Please wait... $(function(){ // Note: All the setup needs to run against a name attribute and *not* the id due the clonish // nature of facebox... $("a[id=googleSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("GOOGLE"); $("form[id=oAuthForm]").submit(); }); $("a[id=facebookSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("FACEBOOK"); $("form[id=oAuthForm]").submit(); }); $("a[id=orcidSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("ORCID"); $("form[id=oAuthForm]").submit(); }); }); If you've forgotten your password, please enter your email address below and we'll send you instructions on how to reset your password. The email address should be the one you originally registered with F1000. Email address not valid, please try again You registered with F1000 via Google, so we cannot reset your password. To sign in, please click here . If you still need help with your Google account password, please click here . You registered with F1000 via Facebook, so we cannot reset your password. To sign in, please click here . If you still need help with your Facebook account password, please click here . Code not correct, please try again Reset password Cancel Email us for further assistance. Server error, please try again. If your email address is registered with us, we will email you instructions to reset your password. If you think you should have received this email but it has not arrived, please check your spam filters and/or contact for further assistance. Please wait... Register $(document).ready(function () { signIn.createSignInAsRow($("#sign-in-form-gfb-popup")); $(".target-field").each(function () { var uris = $(this).val().split("/"); if (uris.pop() === "login") { $(this).val(uris.toString().replace(",","/")); } }); });
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.