bcbioRNASeq: R package for bcbio RNA-seq analysis

preprint OA: closed CC-BY-4.0

Abstract

RNA-seq analysis involves multiple steps, from processing raw sequencing data to identifying, organizing, annotating, and reporting differentially expressed genes. bcbio is an open source, community-maintained framework providing automated and scalable RNA-seq methods for identifying gene abundance counts. We have developed bcbioRNASeq, a Bioconductor package that provides ready-to-render templates, objects and wrapper functions to post-process bcbio RNA sequencing output data. bcbioRNASeq helps automate the generation of high-level RNA-seq reports, facilitating the quality control analyses, identification of differentially expressed genes and functional enrichment analyses.
Full text 188,533 characters · extracted from preprint-html · click to expand
bcbioRNASeq: R package for bcbio RNA-seq analysis | F1000Research "use strict";function _typeof(t){return(_typeof="function"==typeof Symbol&&"symbol"==typeof Symbol.iterator?function(t){return typeof t}:function(t){return t&&"function"==typeof Symbol&&t.constructor===Symbol&&t!==Symbol.prototype?"symbol":typeof t})(t)}!function(){var t=function(){var t,e,o=[],n=window,r=n;for(;r;){try{if(r.frames.__tcfapiLocator){t=r;break}}catch(t){}if(r===n.top)break;r=r.parent}t||(!function t(){var e=n.document,o=!!n.frames.__tcfapiLocator;if(!o)if(e.body){var r=e.createElement("iframe");r.style.cssText="display:none",r.name="__tcfapiLocator",e.body.appendChild(r)}else setTimeout(t,5);return!o}(),n.__tcfapi=function(){for(var t=arguments.length,n=new Array(t),r=0;r 3&&2===parseInt(n[1],10)&&"boolean"==typeof n[3]&&(e=n[3],"function"==typeof n[2]&&n[2]("set",!0)):"ping"===n[0]?"function"==typeof n[2]&&n[2]({gdprApplies:e,cmpLoaded:!1,cmpStatus:"stub"}):o.push(n)},n.addEventListener("message",(function(t){var e="string"==typeof t.data,o={};if(e)try{o=JSON.parse(t.data)}catch(t){}else o=t.data;var n="object"===_typeof(o)&&null!==o?o.__tcfapiCall:null;n&&window.__tcfapi(n.command,n.version,(function(o,r){var a={__tcfapiReturn:{returnValue:o,success:r,callId:n.callId}};t&&t.source&&t.source.postMessage&&t.source.postMessage(e?JSON.stringify(a):a,"*")}),n.parameter)}),!1))};"undefined"!=typeof module?module.exports=t:t()}(); dataLayer = dataLayer || []; // Standard GTM initialization - Google Consent Mode handles consent automatically (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start': new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0], j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src= 'https://www.googletagmanager.com/gtm.js?id='+i+dl+ '>m_auth=hzk0Vc3qFsQYhCrIoHz68A>m_preview=env-1>m_cookies_win=x';f.parentNode.insertBefore(j,f); })(window,document,'script','dataLayer','GTM-MWFK8L5J'); ;window.NREUM||(NREUM={});NREUM.init={distributed_tracing:{enabled:true},privacy:{cookies_enabled:true},ajax:{deny_list:["bam.nr-data.net"]}}; ;NREUM.loader_config={accountID:"438030",trustKey:"438030",agentID:"772317073",licenseKey:"97f8f67f26",applicationID:"772317073"} ;NREUM.info={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net",licenseKey:"97f8f67f26",applicationID:"772317073",sa:1} ;/*! For license information please see nr-loader-spa-1.236.0.min.js.LICENSE.txt */ (()=>{"use strict";var e,t,r={5763:(e,t,r)=>{r.d(t,{P_:()=>l,Mt:()=>g,C5:()=>s,DL:()=>v,OP:()=>T,lF:()=>D,Yu:()=>y,Dg:()=>h,CX:()=>c,GE:()=>b,sU:()=>_});var n=r(8632),i=r(9567);const o={beacon:n.ce.beacon,errorBeacon:n.ce.errorBeacon,licenseKey:void 0,applicationID:void 0,sa:void 0,queueTime:void 0,applicationTime:void 0,ttGuid:void 0,user:void 0,account:void 0,product:void 0,extra:void 0,jsAttributes:{},userAttributes:void 0,atts:void 0,transactionName:void 0,tNamePlain:void 0},a={};function s(e){if(!e)throw new Error("All info objects require an agent identifier!");if(!a[e])throw new Error("Info for ".concat(e," was never set"));return a[e]}function c(e,t){if(!e)throw new Error("All info objects require an agent identifier!");a[e]=(0,i.D)(t,o),(0,n.Qy)(e,a[e],"info")}var u=r(7056);const d=()=>{const e={blockSelector:"[data-nr-block]",maskInputOptions:{password:!0}};return{allow_bfcache:!0,privacy:{cookies_enabled:!0},ajax:{deny_list:void 0,enabled:!0,harvestTimeSeconds:10},distributed_tracing:{enabled:void 0,exclude_newrelic_header:void 0,cors_use_newrelic_header:void 0,cors_use_tracecontext_headers:void 0,allowed_origins:void 0},session:{domain:void 0,expiresMs:u.oD,inactiveMs:u.Hb},ssl:void 0,obfuscate:void 0,jserrors:{enabled:!0,harvestTimeSeconds:10},metrics:{enabled:!0},page_action:{enabled:!0,harvestTimeSeconds:30},page_view_event:{enabled:!0},page_view_timing:{enabled:!0,harvestTimeSeconds:30,long_task:!1},session_trace:{enabled:!0,harvestTimeSeconds:10},harvest:{tooManyRequestsDelay:60},session_replay:{enabled:!1,harvestTimeSeconds:60,sampleRate:.1,errorSampleRate:.1,maskTextSelector:"*",maskAllInputs:!0,get blockClass(){return"nr-block"},get ignoreClass(){return"nr-ignore"},get maskTextClass(){return"nr-mask"},get blockSelector(){return e.blockSelector},set blockSelector(t){e.blockSelector+=",".concat(t)},get maskInputOptions(){return e.maskInputOptions},set maskInputOptions(t){e.maskInputOptions={...t,password:!0}}},spa:{enabled:!0,harvestTimeSeconds:10}}},f={};function l(e){if(!e)throw new Error("All configuration objects require an agent identifier!");if(!f[e])throw new Error("Configuration for ".concat(e," was never set"));return f[e]}function h(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");f[e]=(0,i.D)(t,d()),(0,n.Qy)(e,f[e],"config")}function g(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");var r=l(e);if(r){for(var n=t.split("."),i=0;i {r.d(t,{D:()=>i});var n=r(50);function i(e,t){try{if(!e||"object"!=typeof e)return(0,n.Z)("Setting a Configurable requires an object as input");if(!t||"object"!=typeof t)return(0,n.Z)("Setting a Configurable requires a model to set its initial properties");const r=Object.create(Object.getPrototypeOf(t),Object.getOwnPropertyDescriptors(t)),o=0===Object.keys(r).length?e:r;for(let a in o)if(void 0!==e[a])try{"object"==typeof e[a]&&"object"==typeof t[a]?r[a]=i(e[a],t[a]):r[a]=e[a]}catch(e){(0,n.Z)("An error occurred while setting a property of a Configurable",e)}return r}catch(e){(0,n.Z)("An error occured while setting a Configurable",e)}}},6818:(e,t,r)=>{r.d(t,{Re:()=>i,gF:()=>o,q4:()=>n});const n="1.236.0",i="PROD",o="CDN"},385:(e,t,r)=>{r.d(t,{FN:()=>a,IF:()=>u,Nk:()=>f,Tt:()=>s,_A:()=>o,il:()=>n,pL:()=>c,v6:()=>i,w1:()=>d});const n="undefined"!=typeof window&&!!window.document,i="undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self.navigator instanceof WorkerNavigator||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis.navigator instanceof WorkerNavigator),o=n?window:"undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis),a=""+o?.location,s=/iPad|iPhone|iPod/.test(navigator.userAgent),c=s&&"undefined"==typeof SharedWorker,u=(()=>{const e=navigator.userAgent.match(/Firefox[/\s](\d+\.\d+)/);return Array.isArray(e)&&e.length>=2?+e[1]:0})(),d=Boolean(n&&window.document.documentMode),f=!!navigator.sendBeacon},1117:(e,t,r)=>{r.d(t,{w:()=>o});var n=r(50);const i={agentIdentifier:"",ee:void 0};class o{constructor(e){try{if("object"!=typeof e)return(0,n.Z)("shared context requires an object as input");this.sharedContext={},Object.assign(this.sharedContext,i),Object.entries(e).forEach((e=>{let[t,r]=e;Object.keys(i).includes(t)&&(this.sharedContext[t]=r)}))}catch(e){(0,n.Z)("An error occured while setting SharedContext",e)}}}},8e3:(e,t,r)=>{r.d(t,{L:()=>d,R:()=>c});var n=r(2177),i=r(1284),o=r(4322),a=r(3325);const s={};function c(e,t){const r={staged:!1,priority:a.p[t]||0};u(e),s[e].get(t)||s[e].set(t,r)}function u(e){e&&(s[e]||(s[e]=new Map))}function d(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:"",t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:"feature";if(u(e),!e||!s[e].get(t))return a(t);s[e].get(t).staged=!0;const r=[...s[e]];function a(t){const r=e?n.ee.get(e):n.ee,a=o.X.handlers;if(r.backlog&&a){var s=r.backlog[t],c=a[t];if(c){for(var u=0;s&&u {let[t,r]=e;return r.staged}))&&(r.sort(((e,t)=>e[1].priority-t[1].priority)),r.forEach((e=>{let[t]=e;a(t)})))}function f(e,t){var r=e[1];(0,i.D)(t[r],(function(t,r){var n=e[0];if(r[0]===n){var i=r[1],o=e[3],a=e[2];i.apply(o,a)}}))}},2177:(e,t,r)=>{r.d(t,{c:()=>f,ee:()=>u});var n=r(8632),i=r(2210),o=r(1284),a=r(5763),s="nr@context";let c=(0,n.fP)();var u;function d(){}function f(e){return(0,i.X)(e,s,l)}function l(){return new d}function h(){u.aborted=!0,u.backlog={}}c.ee?u=c.ee:(u=function e(t,r){var n={},c={},f={},g=!1;try{g=16===r.length&&(0,a.OP)(r).isolatedBacklog}catch(e){}var p={on:b,addEventListener:b,removeEventListener:y,emit:v,get:x,listeners:w,context:m,buffer:A,abort:h,aborted:!1,isBuffering:E,debugId:r,backlog:g?{}:t&&"object"==typeof t.backlog?t.backlog:{}};return p;function m(e){return e&&e instanceof d?e:e?(0,i.X)(e,s,l):l()}function v(e,r,n,i,o){if(!1!==o&&(o=!0),!u.aborted||i){t&&o&&t.emit(e,r,n);for(var a=m(n),s=w(e),d=s.length,f=0;fn,p:()=>i});var n=r(2177).ee.get("handle");function i(e,t,r,i,o){o?(o.buffer([e],i),o.emit(e,t,r)):(n.buffer([e],i),n.emit(e,t,r))}},4322:(e,t,r)=>{r.d(t,{X:()=>o});var n=r(5546);o.on=a;var i=o.handlers={};function o(e,t,r,o){a(o||n.E,i,e,t,r)}function a(e,t,r,i,o){o||(o="feature"),e||(e=n.E);var a=t[o]=t[o]||{};(a[r]=a[r]||[]).push([e,i])}},3239:(e,t,r)=>{r.d(t,{bP:()=>s,iz:()=>c,m$:()=>a});var n=r(385);let i=!1,o=!1;try{const e={get passive(){return i=!0,!1},get signal(){return o=!0,!1}};n._A.addEventListener("test",null,e),n._A.removeEventListener("test",null,e)}catch(e){}function a(e,t){return i||o?{capture:!!e,passive:i,signal:t}:!!e}function s(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;window.addEventListener(e,t,a(r,n))}function c(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;document.addEventListener(e,t,a(r,n))}},4402:(e,t,r)=>{r.d(t,{Ht:()=>u,M:()=>c,Rl:()=>a,ky:()=>s});var n=r(385);const i="xxxxxxxx-xxxx-4xxx-yxxx-xxxxxxxxxxxx";function o(e,t){return e?15&e[t]:16*Math.random()|0}function a(){const e=n._A?.crypto||n._A?.msCrypto;let t,r=0;return e&&e.getRandomValues&&(t=e.getRandomValues(new Uint8Array(31))),i.split("").map((e=>"x"===e?o(t,++r).toString(16):"y"===e?(3&o()|8).toString(16):e)).join("")}function s(e){const t=n._A?.crypto||n._A?.msCrypto;let r,i=0;t&&t.getRandomValues&&(r=t.getRandomValues(new Uint8Array(31)));const a=[];for(var s=0;s {r.d(t,{Bq:()=>n,Hb:()=>o,oD:()=>i});const n="NRBA",i=144e5,o=18e5},7894:(e,t,r)=>{function n(){return Math.round(performance.now())}r.d(t,{z:()=>n})},7243:(e,t,r)=>{r.d(t,{e:()=>o});var n=r(385),i={};function o(e){if(e in i)return i[e];if(0===(e||"").indexOf("data:"))return{protocol:"data"};let t;var r=n._A?.location,o={};if(n.il)t=document.createElement("a"),t.href=e;else try{t=new URL(e,r.href)}catch(e){return o}o.port=t.port;var a=t.href.split("://");!o.port&&a[1]&&(o.port=a[1].split("/")[0].split("@").pop().split(":")[1]),o.port&&"0"!==o.port||(o.port="https"===a[0]?"443":"80"),o.hostname=t.hostname||r.hostname,o.pathname=t.pathname,o.protocol=a[0],"/"!==o.pathname.charAt(0)&&(o.pathname="/"+o.pathname);var s=!t.protocol||":"===t.protocol||t.protocol===r.protocol,c=t.hostname===r.hostname&&t.port===r.port;return o.sameOrigin=s&&(!t.hostname||c),"/"===o.pathname&&(i[e]=o),o}},50:(e,t,r)=>{function n(e,t){"function"==typeof console.warn&&(console.warn("New Relic: ".concat(e)),t&&console.warn(t))}r.d(t,{Z:()=>n})},2587:(e,t,r)=>{r.d(t,{N:()=>c,T:()=>u});var n=r(2177),i=r(5546),o=r(8e3),a=r(3325);const s={stn:[a.D.sessionTrace],err:[a.D.jserrors,a.D.metrics],ins:[a.D.pageAction],spa:[a.D.spa],sr:[a.D.sessionReplay,a.D.sessionTrace]};function c(e,t){const r=n.ee.get(t);e&&"object"==typeof e&&(Object.entries(e).forEach((e=>{let[t,n]=e;void 0===u[t]&&(s[t]?s[t].forEach((e=>{n?(0,i.p)("feat-"+t,[],void 0,e,r):(0,i.p)("block-"+t,[],void 0,e,r),(0,i.p)("rumresp-"+t,[Boolean(n)],void 0,e,r)})):n&&(0,i.p)("feat-"+t,[],void 0,void 0,r),u[t]=Boolean(n))})),Object.keys(s).forEach((e=>{void 0===u[e]&&(s[e]?.forEach((t=>(0,i.p)("rumresp-"+e,[!1],void 0,t,r))),u[e]=!1)})),(0,o.L)(t,a.D.pageViewEvent))}const u={}},2210:(e,t,r)=>{r.d(t,{X:()=>i});var n=Object.prototype.hasOwnProperty;function i(e,t,r){if(n.call(e,t))return e[t];var i=r();if(Object.defineProperty&&Object.keys)try{return Object.defineProperty(e,t,{value:i,writable:!0,enumerable:!1}),i}catch(e){}return e[t]=i,i}},1284:(e,t,r)=>{r.d(t,{D:()=>n});const n=(e,t)=>Object.entries(e||{}).map((e=>{let[r,n]=e;return t(r,n)}))},4351:(e,t,r)=>{r.d(t,{P:()=>o});var n=r(2177);const i=()=>{const e=new WeakSet;return(t,r)=>{if("object"==typeof r&&null!==r){if(e.has(r))return;e.add(r)}return r}};function o(e){try{return JSON.stringify(e,i())}catch(e){try{n.ee.emit("internal-error",[e])}catch(e){}}}},3960:(e,t,r)=>{r.d(t,{K:()=>a,b:()=>o});var n=r(3239);function i(){return"undefined"==typeof document||"complete"===document.readyState}function o(e,t){if(i())return e();(0,n.bP)("load",e,t)}function a(e){if(i())return e();(0,n.iz)("DOMContentLoaded",e)}},8632:(e,t,r)=>{r.d(t,{EZ:()=>u,Qy:()=>c,ce:()=>o,fP:()=>a,gG:()=>d,mF:()=>s});var n=r(7894),i=r(385);const o={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net"};function a(){return i._A.NREUM||(i._A.NREUM={}),void 0===i._A.newrelic&&(i._A.newrelic=i._A.NREUM),i._A.NREUM}function s(){let e=a();return e.o||(e.o={ST:i._A.setTimeout,SI:i._A.setImmediate,CT:i._A.clearTimeout,XHR:i._A.XMLHttpRequest,REQ:i._A.Request,EV:i._A.Event,PR:i._A.Promise,MO:i._A.MutationObserver,FETCH:i._A.fetch}),e}function c(e,t,r){let i=a();const o=i.initializedAgents||{},s=o[e]||{};return Object.keys(s).length||(s.initializedAt={ms:(0,n.z)(),date:new Date}),i.initializedAgents={...o,[e]:{...s,[r]:t}},i}function u(e,t){a()[e]=t}function d(){return function(){let e=a();const t=e.info||{};e.info={beacon:o.beacon,errorBeacon:o.errorBeacon,...t}}(),function(){let e=a();const t=e.init||{};e.init={...t}}(),s(),function(){let e=a();const t=e.loader_config||{};e.loader_config={...t}}(),a()}},7956:(e,t,r)=>{r.d(t,{N:()=>i});var n=r(3239);function i(e){let t=arguments.length>1&&void 0!==arguments[1]&&arguments[1],r=arguments.length>2?arguments[2]:void 0,i=arguments.length>3?arguments[3]:void 0;return void(0,n.iz)("visibilitychange",(function(){if(t)return void("hidden"==document.visibilityState&&e());e(document.visibilityState)}),r,i)}},1214:(e,t,r)=>{r.d(t,{em:()=>v,u5:()=>N,QU:()=>S,_L:()=>I,Gm:()=>L,Lg:()=>M,gy:()=>U,BV:()=>Q,Kf:()=>ee});var n=r(2177);const i="nr@original";var o=Object.prototype.hasOwnProperty,a=!1;function s(e,t){return e||(e=n.ee),r.inPlace=function(e,t,n,i,o){n||(n="");var a,s,c,u="-"===n.charAt(0);for(c=0;c 2?n-2:0),o=2;o {r(A[T],e,w),r(E[T],e,w)})),r(l._A,"fetch",y),t.on(y+"end",(function(e,r){var n=this;if(r){var i=r.headers.get("content-length");null!==i&&(n.rxSize=i),t.emit(y+"done",[null,r],n)}else t.emit(y+"done",[e],n)})),t}const O={},j=["pushState","replaceState"];function S(e){const t=function(e){return(e||n.ee).get("history")}(e);return!l.il||O[t.debugId]++||(O[t.debugId]=1,s(t).inPlace(window.history,j,"-")),t}var P=r(3239);const C={},R=["appendChild","insertBefore","replaceChild"];function I(e){const t=function(e){return(e||n.ee).get("jsonp")}(e);if(!l.il||C[t.debugId])return t;C[t.debugId]=!0;var r=s(t),i=/[?&](?:callback|cb)=([^&#]+)/,o=/(.*)\.([^.]+)/,a=/^(\w+)(\.|$)(.*)$/;function c(e,t){var r=e.match(a),n=r[1],i=r[3];return i?c(i,t[n]):t[n]}return r.inPlace(Node.prototype,R,"dom-"),t.on("dom-start",(function(e){!function(e){if(!e||"string"!=typeof e.nodeName||"script"!==e.nodeName.toLowerCase())return;if("function"!=typeof e.addEventListener)return;var n=(a=e.src,s=a.match(i),s?s[1]:null);var a,s;if(!n)return;var u=function(e){var t=e.match(o);if(t&&t.length>=3)return{key:t[2],parent:c(t[1],window)};return{key:e,parent:window}}(n);if("function"!=typeof u.parent[u.key])return;var d={};function f(){t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}function l(){t.emit("jsonp-error",[],d),t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}r.inPlace(u.parent,[u.key],"cb-",d),e.addEventListener("load",f,(0,P.m$)(!1)),e.addEventListener("error",l,(0,P.m$)(!1)),t.emit("new-jsonp",[e.src],d)}(e[0])})),t}var k=r(5763);const H={};function L(e){const t=function(e){return(e||n.ee).get("mutation")}(e);if(!l.il||H[t.debugId])return t;H[t.debugId]=!0;var r=s(t),i=k.Yu.MO;return i&&(window.MutationObserver=function(e){return this instanceof i?new i(r(e,"fn-")):i.apply(this,arguments)},MutationObserver.prototype=i.prototype),t}const z={};function M(e){const t=function(e){return(e||n.ee).get("promise")}(e);if(z[t.debugId])return t;z[t.debugId]=!0;var r=n.c,o=s(t),a=k.Yu.PR;return a&&function(){function e(r){var n=t.context(),i=o(r,"executor-",n,null,!1);const s=Reflect.construct(a,[i],e);return t.context(s).getCtx=function(){return n},s}l._A.Promise=e,Object.defineProperty(e,"name",{value:"Promise"}),e.toString=function(){return a.toString()},Object.setPrototypeOf(e,a),["all","race"].forEach((function(r){const n=a[r];e[r]=function(e){let i=!1;[...e||[]].forEach((e=>{this.resolve(e).then(a("all"===r),a(!1))}));const o=n.apply(this,arguments);return o;function a(e){return function(){t.emit("propagate",[null,!i],o,!1,!1),i=i||!e}}}})),["resolve","reject"].forEach((function(r){const n=a[r];e[r]=function(e){const r=n.apply(this,arguments);return e!==r&&t.emit("propagate",[e,!0],r,!1,!1),r}})),e.prototype=a.prototype;const n=a.prototype.then;a.prototype.then=function(){var e=this,i=r(e);i.promise=e;for(var a=arguments.length,s=new Array(a),c=0;c e())),t};function m(e,t){i.inPlace(t,["onreadystatechange"],"fn-",E)}function b(){var e=this,t=r.context(e);e.readyState>3&&!t.resolved&&(t.resolved=!0,r.emit("xhr-resolved",[],e)),i.inPlace(e,f,"fn-",E)}if(function(e,t){for(var r in e)t[r]=e[r]}(o,p),p.prototype=o.prototype,i.inPlace(p.prototype,J,"-xhr-",E),r.on("send-xhr-start",(function(e,t){m(e,t),function(e){h.push(e),a&&(y?y.then(A):u?u(A):(w=-w,x.data=w))}(t)})),r.on("open-xhr-start",m),a){var y=c&&c.resolve();if(!u&&!c){var w=1,x=document.createTextNode(w);new a(A).observe(x,{characterData:!0})}}else t.on("fn-end",(function(e){e[0]&&e[0].type===d||A()}));function A(){for(var e=0;e {r.d(t,{t:()=>n});const n=r(3325).D.ajax},6660:(e,t,r)=>{r.d(t,{A:()=>i,t:()=>n});const n=r(3325).D.jserrors,i="nr@seenError"},3081:(e,t,r)=>{r.d(t,{gF:()=>o,mY:()=>i,t9:()=>n,vz:()=>s,xS:()=>a});const n=r(3325).D.metrics,i="sm",o="cm",a="storeSupportabilityMetrics",s="storeEventMetrics"},4649:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageAction},7633:(e,t,r)=>{r.d(t,{Dz:()=>i,OJ:()=>a,qw:()=>o,t9:()=>n});const n=r(3325).D.pageViewEvent,i="firstbyte",o="domcontent",a="windowload"},9251:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageViewTiming},3614:(e,t,r)=>{r.d(t,{BST_RESOURCE:()=>i,END:()=>s,FEATURE_NAME:()=>n,FN_END:()=>u,FN_START:()=>c,PUSH_STATE:()=>d,RESOURCE:()=>o,START:()=>a});const n=r(3325).D.sessionTrace,i="bstResource",o="resource",a="-start",s="-end",c="fn"+a,u="fn"+s,d="pushState"},7836:(e,t,r)=>{r.d(t,{BODY:()=>A,CB_END:()=>E,CB_START:()=>u,END:()=>x,FEATURE_NAME:()=>i,FETCH:()=>_,FETCH_BODY:()=>v,FETCH_DONE:()=>m,FETCH_START:()=>p,FN_END:()=>c,FN_START:()=>s,INTERACTION:()=>l,INTERACTION_API:()=>d,INTERACTION_EVENTS:()=>o,JSONP_END:()=>b,JSONP_NODE:()=>g,JS_TIME:()=>T,MAX_TIMER_BUDGET:()=>a,REMAINING:()=>f,SPA_NODE:()=>h,START:()=>w,originalSetTimeout:()=>y});var n=r(5763);const i=r(3325).D.spa,o=["click","submit","keypress","keydown","keyup","change"],a=999,s="fn-start",c="fn-end",u="cb-start",d="api-ixn-",f="remaining",l="interaction",h="spaNode",g="jsonpNode",p="fetch-start",m="fetch-done",v="fetch-body-",b="jsonp-end",y=n.Yu.ST,w="-start",x="-end",A="-body",E="cb"+x,T="jsTime",_="fetch"},5938:(e,t,r)=>{r.d(t,{W:()=>o});var n=r(5763),i=r(2177);class o{constructor(e,t,r){this.agentIdentifier=e,this.aggregator=t,this.ee=i.ee.get(e,(0,n.OP)(this.agentIdentifier).isolatedBacklog),this.featureName=r,this.blocked=!1}}},9144:(e,t,r)=>{r.d(t,{j:()=>m});var n=r(3325),i=r(5763),o=r(5546),a=r(2177),s=r(7894),c=r(8e3),u=r(3960),d=r(385),f=r(50),l=r(3081),h=r(8632);function g(){const e=(0,h.gG)();["setErrorHandler","finished","addToTrace","inlineHit","addRelease","addPageAction","setCurrentRouteName","setPageViewName","setCustomAttribute","interaction","noticeError","setUserId"].forEach((t=>{e[t]=function(){for(var r=arguments.length,n=new Array(r),i=0;i 1?r-1:0),i=1;i {e.exposed&&e.api[t]&&o.push(e.api[t](...n))})),o.length>1?o:o[0]}(t,...n)}}))}var p=r(2587);function m(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:{},m=arguments.length>2?arguments[2]:void 0,v=arguments.length>3?arguments[3]:void 0,{init:b,info:y,loader_config:w,runtime:x={loaderType:m},exposed:A=!0}=t;const E=(0,h.gG)();y||(b=E.init,y=E.info,w=E.loader_config),(0,i.Dg)(e,b||{}),(0,i.GE)(e,w||{}),(0,i.sU)(e,x),y.jsAttributes??={},d.v6&&(y.jsAttributes.isWorker=!0),(0,i.CX)(e,y),g();const T=function(e,t){t||(0,c.R)(e,"api");const h={};var g=a.ee.get(e),p=g.get("tracer"),m="api-",v=m+"ixn-";function b(t,r,n,o){const a=(0,i.C5)(e);return null===r?delete a.jsAttributes[t]:(0,i.CX)(e,{...a,jsAttributes:{...a.jsAttributes,[t]:r}}),x(m,n,!0,o||null===r?"session":void 0)(t,r)}function y(){}["setErrorHandler","finished","addToTrace","inlineHit","addRelease"].forEach((e=>h[e]=x(m,e,!0,"api"))),h.addPageAction=x(m,"addPageAction",!0,n.D.pageAction),h.setCurrentRouteName=x(m,"routeName",!0,n.D.spa),h.setPageViewName=function(t,r){if("string"==typeof t)return"/"!==t.charAt(0)&&(t="/"+t),(0,i.OP)(e).customTransaction=(r||"http://custom.transaction")+t,x(m,"setPageViewName",!0)()},h.setCustomAttribute=function(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2];if("string"==typeof e){if(["string","number"].includes(typeof t)||null===t)return b(e,t,"setCustomAttribute",r);(0,f.Z)("Failed to execute setCustomAttribute.\nNon-null value must be a string or number type, but a type of was provided."))}else(0,f.Z)("Failed to execute setCustomAttribute.\nName must be a string type, but a type of was provided."))},h.setUserId=function(e){if("string"==typeof e||null===e)return b("enduser.id",e,"setUserId",!0);(0,f.Z)("Failed to execute setUserId.\nNon-null value must be a string type, but a type of was provided."))},h.interaction=function(){return(new y).get()};var w=y.prototype={createTracer:function(e,t){var r={},i=this,a="function"==typeof t;return(0,o.p)(v+"tracer",[(0,s.z)(),e,r],i,n.D.spa,g),function(){if(p.emit((a?"":"no-")+"fn-start",[(0,s.z)(),i,a],r),a)try{return t.apply(this,arguments)}catch(e){throw p.emit("fn-err",[arguments,this,"string"==typeof e?new Error(e):e],r),e}finally{p.emit("fn-end",[(0,s.z)()],r)}}}};function x(e,t,r,i){return function(){return(0,o.p)(l.xS,["API/"+t+"/called"],void 0,n.D.metrics,g),i&&(0,o.p)(e+t,[(0,s.z)(),...arguments],r?null:this,i,g),r?void 0:this}}function A(){r.e(439).then(r.bind(r,7438)).then((t=>{let{setAPI:r}=t;r(e),(0,c.L)(e,"api")})).catch((()=>(0,f.Z)("Downloading runtime APIs failed...")))}return["actionText","setName","setAttribute","save","ignore","onEnd","getContext","end","get"].forEach((e=>{w[e]=x(v,e,void 0,n.D.spa)})),h.noticeError=function(e,t){"string"==typeof e&&(e=new Error(e)),(0,o.p)(l.xS,["API/noticeError/called"],void 0,n.D.metrics,g),(0,o.p)("err",[e,(0,s.z)(),!1,t],void 0,n.D.jserrors,g)},d.il?(0,u.b)((()=>A()),!0):A(),h}(e,v);return(0,h.Qy)(e,T,"api"),(0,h.Qy)(e,A,"exposed"),(0,h.EZ)("activatedFeatures",p.T),T}},3325:(e,t,r)=>{r.d(t,{D:()=>n,p:()=>i});const n={ajax:"ajax",jserrors:"jserrors",metrics:"metrics",pageAction:"page_action",pageViewEvent:"page_view_event",pageViewTiming:"page_view_timing",sessionReplay:"session_replay",sessionTrace:"session_trace",spa:"spa"},i={[n.pageViewEvent]:1,[n.pageViewTiming]:2,[n.metrics]:3,[n.jserrors]:4,[n.ajax]:5,[n.sessionTrace]:6,[n.pageAction]:7,[n.spa]:8,[n.sessionReplay]:9}}},n={};function i(e){var t=n[e];if(void 0!==t)return t.exports;var o=n[e]={exports:{}};return r[e](o,o.exports,i),o.exports}i.m=r,i.d=(e,t)=>{for(var r in t)i.o(t,r)&&!i.o(e,r)&&Object.defineProperty(e,r,{enumerable:!0,get:t[r]})},i.f={},i.e=e=>Promise.all(Object.keys(i.f).reduce(((t,r)=>(i.f[r](e,t),t)),[])),i.u=e=>(({78:"page_action-aggregate",147:"metrics-aggregate",242:"session-manager",317:"jserrors-aggregate",348:"page_view_timing-aggregate",412:"lazy-feature-loader",439:"async-api",538:"recorder",590:"session_replay-aggregate",675:"compressor",733:"session_trace-aggregate",786:"page_view_event-aggregate",873:"spa-aggregate",898:"ajax-aggregate"}[e]||e)+"."+{78:"ac76d497",147:"3dc53903",148:"1a20d5fe",242:"2a64278a",317:"49e41428",348:"bd6de33a",412:"2f55ce66",439:"30bd804e",538:"1b18459f",590:"cf0efb30",675:"ae9f91a8",733:"83105561",786:"06482edd",860:"03a8b7a5",873:"e6b09d52",898:"998ef92b"}[e]+"-1.236.0.min.js"),i.o=(e,t)=>Object.prototype.hasOwnProperty.call(e,t),e={},t="NRBA:",i.l=(r,n,o,a)=>{if(e[r])e[r].push(n);else{var s,c;if(void 0!==o)for(var u=document.getElementsByTagName("script"),d=0;d {s.onerror=s.onload=null,clearTimeout(h);var i=e[r];if(delete e[r],s.parentNode&&s.parentNode.removeChild(s),i&&i.forEach((e=>e(n))),t)return t(n)},h=setTimeout(l.bind(null,void 0,{type:"timeout",target:s}),12e4);s.onerror=l.bind(null,s.onerror),s.onload=l.bind(null,s.onload),c&&document.head.appendChild(s)}},i.r=e=>{"undefined"!=typeof Symbol&&Symbol.toStringTag&&Object.defineProperty(e,Symbol.toStringTag,{value:"Module"}),Object.defineProperty(e,"__esModule",{value:!0})},i.j=364,i.p="https://js-agent.newrelic.com/",(()=>{var e={364:0,953:0};i.f.j=(t,r)=>{var n=i.o(e,t)?e[t]:void 0;if(0!==n)if(n)r.push(n[2]);else{var o=new Promise(((r,i)=>n=e[t]=[r,i]));r.push(n[2]=o);var a=i.p+i.u(t),s=new Error;i.l(a,(r=>{if(i.o(e,t)&&(0!==(n=e[t])&&(e[t]=void 0),n)){var o=r&&("load"===r.type?"missing":r.type),a=r&&r.target&&r.target.src;s.message="Loading chunk "+t+" failed.\n("+o+": "+a+")",s.name="ChunkLoadError",s.type=o,s.request=a,n[1](s)}}),"chunk-"+t,t)}};var t=(t,r)=>{var n,o,[a,s,c]=r,u=0;if(a.some((t=>0!==e[t]))){for(n in s)i.o(s,n)&&(i.m[n]=s[n]);if(c)c(i)}for(t&&t(r);u {i.r(o);var e=i(3325),t=i(5763);const r=Object.values(e.D);function n(e){const n={};return r.forEach((r=>{n[r]=function(e,r){return!1!==(0,t.Mt)(r,"".concat(e,".enabled"))}(r,e)})),n}var a=i(9144);var s=i(5546),c=i(385),u=i(8e3),d=i(5938),f=i(3960),l=i(50);class h extends d.W{constructor(e,t,r){let n=!(arguments.length>3&&void 0!==arguments[3])||arguments[3];super(e,t,r),this.auto=n,this.abortHandler,this.featAggregate,this.onAggregateImported,n&&(0,u.R)(e,r)}importAggregator(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:{};if(this.featAggregate||!this.auto)return;const r=c.il&&!0===(0,t.Mt)(this.agentIdentifier,"privacy.cookies_enabled");let n;this.onAggregateImported=new Promise((e=>{n=e}));const o=async()=>{let t;try{if(r){const{setupAgentSession:e}=await Promise.all([i.e(860),i.e(242)]).then(i.bind(i,3228));t=e(this.agentIdentifier)}}catch(e){(0,l.Z)("A problem occurred when starting up session manager. This page will not start or extend any session.",e)}try{if(!this.shouldImportAgg(this.featureName,t))return void(0,u.L)(this.agentIdentifier,this.featureName);const{lazyFeatureLoader:r}=await i.e(412).then(i.bind(i,8582)),{Aggregate:o}=await r(this.featureName,"aggregate");this.featAggregate=new o(this.agentIdentifier,this.aggregator,e),n(!0)}catch(e){(0,l.Z)("Downloading and initializing ".concat(this.featureName," failed..."),e),this.abortHandler?.(),n(!1)}};c.il?(0,f.b)((()=>o()),!0):o()}shouldImportAgg(r,n){return r!==e.D.sessionReplay||!1!==(0,t.Mt)(this.agentIdentifier,"session_trace.enabled")&&(!!n?.isNew||!!n?.state.sessionReplay)}}var g=i(7633),p=i(7894);class m extends h{static featureName=g.t9;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];if(super(r,n,g.t9,i),("undefined"==typeof PerformanceNavigationTiming||c.Tt)&&"undefined"!=typeof PerformanceTiming){const n=(0,t.OP)(r);n[g.Dz]=Math.max(Date.now()-n.offset,0),(0,f.K)((()=>n[g.qw]=Math.max((0,p.z)()-n[g.Dz],0))),(0,f.b)((()=>{const t=(0,p.z)();n[g.OJ]=Math.max(t-n[g.Dz],0),(0,s.p)("timing",["load",t],void 0,e.D.pageViewTiming,this.ee)}))}this.importAggregator()}}var v=i(1117),b=i(1284);class y extends v.w{constructor(e){super(e),this.aggregatedData={}}store(e,t,r,n,i){var o=this.getBucket(e,t,r,i);return o.metrics=function(e,t){t||(t={count:0});return t.count+=1,(0,b.D)(e,(function(e,r){t[e]=w(r,t[e])})),t}(n,o.metrics),o}merge(e,t,r,n,i){var o=this.getBucket(e,t,n,i);if(o.metrics){var a=o.metrics;a.count+=r.count,(0,b.D)(r,(function(e,t){if("count"!==e){var n=a[e],i=r[e];i&&!i.c?a[e]=w(i.t,n):a[e]=function(e,t){if(!t)return e;t.c||(t=x(t.t));return t.min=Math.min(e.min,t.min),t.max=Math.max(e.max,t.max),t.t+=e.t,t.sos+=e.sos,t.c+=e.c,t}(i,a[e])}}))}else o.metrics=r}storeMetric(e,t,r,n){var i=this.getBucket(e,t,r);return i.stats=w(n,i.stats),i}getBucket(e,t,r,n){this.aggregatedData[e]||(this.aggregatedData[e]={});var i=this.aggregatedData[e][t];return i||(i=this.aggregatedData[e][t]={params:r||{}},n&&(i.custom=n)),i}get(e,t){return t?this.aggregatedData[e]&&this.aggregatedData[e][t]:this.aggregatedData[e]}take(e){for(var t={},r="",n=!1,i=0;i t.max&&(t.max=e),e 2&&void 0!==arguments[2])||arguments[2];super(e,r,j.t,n),c.il&&((0,t.OP)(e).initHidden=Boolean("hidden"===document.visibilityState),(0,N.N)((()=>(0,s.p)("docHidden",[(0,p.z)()],void 0,j.t,this.ee)),!0),(0,O.bP)("pagehide",(()=>(0,s.p)("winPagehide",[(0,p.z)()],void 0,j.t,this.ee))),this.importAggregator())}}var P=i(3081);class C extends h{static featureName=P.t9;constructor(e,t){let r=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(e,t,P.t9,r),this.importAggregator()}}var R,I=i(2210),k=i(1214),H=i(2177),L={};try{R=localStorage.getItem("__nr_flags").split(","),console&&"function"==typeof console.log&&(L.console=!0,-1!==R.indexOf("dev")&&(L.dev=!0),-1!==R.indexOf("nr_dev")&&(L.nrDev=!0))}catch(e){}function z(e){try{L.console&&z(e)}catch(e){}}L.nrDev&&H.ee.on("internal-error",(function(e){z(e.stack)})),L.dev&&H.ee.on("fn-err",(function(e,t,r){z(r.stack)})),L.dev&&(z("NR AGENT IN DEVELOPMENT MODE"),z("flags: "+(0,b.D)(L,(function(e,t){return e})).join(", ")));var M=i(6660);class B extends h{static featureName=M.t;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(r,n,M.t,i),this.skipNext=0;try{this.removeOnAbort=new AbortController}catch(e){}const o=this;o.ee.on("fn-start",(function(e,t,r){o.abortHandler&&(o.skipNext+=1)})),o.ee.on("fn-err",(function(t,r,n){o.abortHandler&&!n[M.A]&&((0,I.X)(n,M.A,(function(){return!0})),this.thrown=!0,(0,s.p)("err",[n,(0,p.z)()],void 0,e.D.jserrors,o.ee))})),o.ee.on("fn-end",(function(){o.abortHandler&&!this.thrown&&o.skipNext>0&&(o.skipNext-=1)})),o.ee.on("internal-error",(function(t){(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,o.ee)})),this.origOnerror=c._A.onerror,c._A.onerror=this.onerrorHandler.bind(this),c._A.addEventListener("unhandledrejection",(t=>{const r=function(e){let t="Unhandled Promise Rejection: ";if(e instanceof Error)try{return e.message=t+e.message,e}catch(t){return e}if(void 0===e)return new Error(t);try{return new Error(t+(0,D.P)(e))}catch(e){return new Error(t)}}(t.reason);(0,s.p)("err",[r,(0,p.z)(),!1,{unhandledPromiseRejection:1}],void 0,e.D.jserrors,this.ee)}),(0,O.m$)(!1,this.removeOnAbort?.signal)),(0,k.gy)(this.ee),(0,k.BV)(this.ee),(0,k.em)(this.ee),(0,t.OP)(r).xhrWrappable&&(0,k.Kf)(this.ee),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}onerrorHandler(t,r,n,i,o){"function"==typeof this.origOnerror&&this.origOnerror(...arguments);try{this.skipNext?this.skipNext-=1:(0,s.p)("err",[o||new F(t,r,n),(0,p.z)()],void 0,e.D.jserrors,this.ee)}catch(t){try{(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,this.ee)}catch(e){}}return!1}}function F(e,t,r){this.message=e||"Uncaught error with no additional information",this.sourceURL=t,this.line=r}let U=1;const q="nr@id";function G(e){const t=typeof e;return!e||"object"!==t&&"function"!==t?-1:e===c._A?0:(0,I.X)(e,q,(function(){return U++}))}function V(e){if("string"==typeof e&&e.length)return e.length;if("object"==typeof e){if("undefined"!=typeof ArrayBuffer&&e instanceof ArrayBuffer&&e.byteLength)return e.byteLength;if("undefined"!=typeof Blob&&e instanceof Blob&&e.size)return e.size;if(!("undefined"!=typeof FormData&&e instanceof FormData))try{return(0,D.P)(e).length}catch(e){return}}}var X=i(7243);class W{constructor(e){this.agentIdentifier=e,this.generateTracePayload=this.generateTracePayload.bind(this),this.shouldGenerateTrace=this.shouldGenerateTrace.bind(this)}generateTracePayload(e){if(!this.shouldGenerateTrace(e))return null;var r=(0,t.DL)(this.agentIdentifier);if(!r)return null;var n=(r.accountID||"").toString()||null,i=(r.agentID||"").toString()||null,o=(r.trustKey||"").toString()||null;if(!n||!i)return null;var a=(0,_.M)(),s=(0,_.Ht)(),c=Date.now(),u={spanId:a,traceId:s,timestamp:c};return(e.sameOrigin||this.isAllowedOrigin(e)&&this.useTraceContextHeadersForCors())&&(u.traceContextParentHeader=this.generateTraceContextParentHeader(a,s),u.traceContextStateHeader=this.generateTraceContextStateHeader(a,c,n,i,o)),(e.sameOrigin&&!this.excludeNewrelicHeader()||!e.sameOrigin&&this.isAllowedOrigin(e)&&this.useNewrelicHeaderForCors())&&(u.newrelicHeader=this.generateTraceHeader(a,s,c,n,i,o)),u}generateTraceContextParentHeader(e,t){return"00-"+t+"-"+e+"-01"}generateTraceContextStateHeader(e,t,r,n,i){return i+"@nr=0-1-"+r+"-"+n+"-"+e+"----"+t}generateTraceHeader(e,t,r,n,i,o){if(!("function"==typeof c._A?.btoa))return null;var a={v:[0,1],d:{ty:"Browser",ac:n,ap:i,id:e,tr:t,ti:r}};return o&&n!==o&&(a.d.tk=o),btoa((0,D.P)(a))}shouldGenerateTrace(e){return this.isDtEnabled()&&this.isAllowedOrigin(e)}isAllowedOrigin(e){var r=!1,n={};if((0,t.Mt)(this.agentIdentifier,"distributed_tracing")&&(n=(0,t.P_)(this.agentIdentifier).distributed_tracing),e.sameOrigin)r=!0;else if(n.allowed_origins instanceof Array)for(var i=0;i 2&&void 0!==arguments[2])||arguments[2];super(r,n,Z.t,i),(0,t.OP)(r).xhrWrappable&&(this.dt=new W(r),this.handler=(e,t,r,n)=>(0,s.p)(e,t,r,n,this.ee),(0,k.u5)(this.ee),(0,k.Kf)(this.ee),function(r,n,i,o){function a(e){var t=this;t.totalCbs=0,t.called=0,t.cbTime=0,t.end=E,t.ended=!1,t.xhrGuids={},t.lastSize=null,t.loadCaptureCalled=!1,t.params=this.params||{},t.metrics=this.metrics||{},e.addEventListener("load",(function(r){_(t,e)}),(0,O.m$)(!1)),c.IF||e.addEventListener("progress",(function(e){t.lastSize=e.loaded}),(0,O.m$)(!1))}function s(e){this.params={method:e[0]},T(this,e[1]),this.metrics={}}function u(e,n){var i=(0,t.DL)(r);i.xpid&&this.sameOrigin&&n.setRequestHeader("X-NewRelic-ID",i.xpid);var a=o.generateTracePayload(this.parsedOrigin);if(a){var s=!1;a.newrelicHeader&&(n.setRequestHeader("newrelic",a.newrelicHeader),s=!0),a.traceContextParentHeader&&(n.setRequestHeader("traceparent",a.traceContextParentHeader),a.traceContextStateHeader&&n.setRequestHeader("tracestate",a.traceContextStateHeader),s=!0),s&&(this.dt=a)}}function d(e,t){var r=this.metrics,i=e[0],o=this;if(r&&i){var a=V(i);a&&(r.txSize=a)}this.startTime=(0,p.z)(),this.listener=function(e){try{"abort"!==e.type||o.loadCaptureCalled||(o.params.aborted=!0),("load"!==e.type||o.called===o.totalCbs&&(o.onloadCalled||"function"!=typeof t.onload)&&"function"==typeof o.end)&&o.end(t)}catch(e){try{n.emit("internal-error",[e])}catch(e){}}};for(var s=0;s 1?e[1]=i:e.push(i)}else e[0]&&e[0].headers&&s(e[0].headers,n)&&(this.dt=n);function s(e,t){var r=!1;return t.newrelicHeader&&(e.set("newrelic",t.newrelicHeader),r=!0),t.traceContextParentHeader&&(e.set("traceparent",t.traceContextParentHeader),t.traceContextStateHeader&&e.set("tracestate",t.traceContextStateHeader),r=!0),r}}function x(e,t){this.params={},this.metrics={},this.startTime=(0,p.z)(),this.dt=t,e.length>=1&&(this.target=e[0]),e.length>=2&&(this.opts=e[1]);var r,n=this.opts||{},i=this.target;"string"==typeof i?r=i:"object"==typeof i&&i instanceof Y?r=i.url:c._A?.URL&&"object"==typeof i&&i instanceof URL&&(r=i.href),T(this,r);var o=(""+(i&&i instanceof Y&&i.method||n.method||"GET")).toUpperCase();this.params.method=o,this.txSize=V(n.body)||0}function A(t,r){var n;this.endTime=(0,p.z)(),this.params||(this.params={}),this.params.status=r?r.status:0,"string"==typeof this.rxSize&&this.rxSize.length>0&&(n=+this.rxSize);var o={txSize:this.txSize,rxSize:n,duration:(0,p.z)()-this.startTime};i("xhr",[this.params,o,this.startTime,this.endTime,"fetch"],this,e.D.ajax)}function E(t){var r=this.params,n=this.metrics;if(!this.ended){this.ended=!0;for(var o=0;o 2&&void 0!==arguments[2])||arguments[2];super(e,t,we.t,r),this.importAggregator()}}new class{constructor(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:(0,_.ky)(16);c._A?(this.agentIdentifier=t,this.sharedAggregator=new y({agentIdentifier:this.agentIdentifier}),this.features={},this.desiredFeatures=new Set(e.features||[]),this.desiredFeatures.add(m),Object.assign(this,(0,a.j)(this.agentIdentifier,e,e.loaderType||"agent")),this.start()):(0,l.Z)("Failed to initial the agent. Could not determine the runtime environment.")}get config(){return{info:(0,t.C5)(this.agentIdentifier),init:(0,t.P_)(this.agentIdentifier),loader_config:(0,t.DL)(this.agentIdentifier),runtime:(0,t.OP)(this.agentIdentifier)}}start(){const t="features";try{const r=n(this.agentIdentifier),i=[...this.desiredFeatures];i.sort(((t,r)=>e.p[t.featureName]-e.p[r.featureName])),i.forEach((t=>{if(r[t.featureName]||t.featureName===e.D.pageViewEvent){const n=function(t){switch(t){case e.D.ajax:return[e.D.jserrors];case e.D.sessionTrace:return[e.D.ajax,e.D.pageViewEvent];case e.D.sessionReplay:return[e.D.sessionTrace];case e.D.pageViewTiming:return[e.D.pageViewEvent];default:return[]}}(t.featureName);n.every((e=>r[e]))||(0,l.Z)("".concat(t.featureName," is enabled but one or more dependent features has been disabled (").concat((0,D.P)(n),"). This may cause unintended consequences or missing data...")),this.features[t.featureName]=new t(this.agentIdentifier,this.sharedAggregator)}})),(0,T.Qy)(this.agentIdentifier,this.features,t)}catch(e){(0,l.Z)("Failed to initialize all enabled instrument classes (agent aborted) -",e);for(const e in this.features)this.features[e].abortHandler?.();const r=(0,T.fP)();return delete r.initializedAgents[this.agentIdentifier]?.api,delete r.initializedAgents[this.agentIdentifier]?.[t],delete this.sharedAggregator,r.ee?.abort(),delete r.ee?.get(this.agentIdentifier),!1}}}({features:[J,m,S,class extends h{static featureName=oe;constructor(t,r){if(super(t,r,oe,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;const n=this.ee;let i;(0,k.QU)(n),this.eventsEE=(0,k.em)(n),this.eventsEE.on(se,(function(e,t){this.bstStart=(0,p.z)()})),this.eventsEE.on(ae,(function(t,r){(0,s.p)("bst",[t[0],r,this.bstStart,(0,p.z)()],void 0,e.D.sessionTrace,n)})),n.on(ce+ne,(function(e){this.time=(0,p.z)(),this.startPath=location.pathname+location.hash})),n.on(ce+ie,(function(t){(0,s.p)("bstHist",[location.pathname+location.hash,this.startPath,this.time],void 0,e.D.sessionTrace,n)}));try{i=new PerformanceObserver((t=>{const r=t.getEntries();(0,s.p)(te,[r],void 0,e.D.sessionTrace,n)})),i.observe({type:re,buffered:!0})}catch(e){}this.importAggregator({resourceObserver:i})}},C,xe,B,class extends h{static featureName=de;constructor(e,r){if(super(e,r,de,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;if(!(0,t.OP)(e).xhrWrappable)return;try{this.removeOnAbort=new AbortController}catch(e){}let n,i=0;const o=this.ee.get("tracer"),a=(0,k._L)(this.ee),s=(0,k.Lg)(this.ee),u=(0,k.BV)(this.ee),d=(0,k.Kf)(this.ee),f=this.ee.get("events"),l=(0,k.u5)(this.ee),h=(0,k.QU)(this.ee),g=(0,k.Gm)(this.ee);function m(e,t){h.emit("newURL",[""+window.location,t])}function v(){i++,n=window.location.hash,this[ve]=(0,p.z)()}function b(){i--,window.location.hash!==n&&m(0,!0);var e=(0,p.z)();this[pe]=~~this[pe]+e-this[ve],this[ye]=e}function y(e,t){e.on(t,(function(){this[t]=(0,p.z)()}))}this.ee.on(ve,v),s.on(be,v),a.on(be,v),this.ee.on(ye,b),s.on(ge,b),a.on(ge,b),this.ee.buffer([ve,ye,"xhr-resolved"],this.featureName),f.buffer([ve],this.featureName),u.buffer(["setTimeout"+le,"clearTimeout"+fe,ve],this.featureName),d.buffer([ve,"new-xhr","send-xhr"+fe],this.featureName),l.buffer([me+fe,me+"-done",me+he+fe,me+he+le],this.featureName),h.buffer(["newURL"],this.featureName),g.buffer([ve],this.featureName),s.buffer(["propagate",be,ge,"executor-err","resolve"+fe],this.featureName),o.buffer([ve,"no-"+ve],this.featureName),a.buffer(["new-jsonp","cb-start","jsonp-error","jsonp-end"],this.featureName),y(l,me+fe),y(l,me+"-done"),y(a,"new-jsonp"),y(a,"jsonp-end"),y(a,"cb-start"),h.on("pushState-end",m),h.on("replaceState-end",m),window.addEventListener("hashchange",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("load",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("popstate",(function(){m(0,i>1)}),(0,O.m$)(!0,this.removeOnAbort?.signal)),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}}],loaderType:"spa"})})(),window.NRBA=o})(); window.jQuery || document.write(' ') CKEDITOR_BASEPATH='https://f1000research.com/js/vendor/ckeditor/' window.reactTheme = 'research'; window.MathJax = { CommonHTML: { linebreaks: { automatic: true } }, 'HTML-CSS': { linebreaks: { automatic: true } }, SVG: { linebreaks: { automatic: true } }, AuthorInit: function() { MathJax.Hub.Register.MessageHook('End Process', function () { let timeout = false; // holder for timeout id const delay = 250; // delay after event is "complete" to run callback const reflowMath = function() { const dispFormulas = document.querySelectorAll('.disp-formula.panel'); if (!dispFormulas) { return; } for (const dispFormula of dispFormulas) { const child = dispFormula.querySelector('.MathJax_Preview').nextSibling.firstChild; const isMultiline = MathJax.Hub.getAllJax(dispFormula)[0].root.isMultiline; if (dispFormula.offsetWidth < child.offsetWidth || isMultiline) { MathJax.Hub.Queue(['Rerender', MathJax.Hub, dispFormula]); } } }; window.addEventListener('resize', function() { clearTimeout(timeout); // clear the timeout timeout = setTimeout(reflowMath, delay); // start timing for event "completion" }); }); }, }; if (window.location.hash == '#_=_'){ window.location = window.location.href.split('#')[0] } !function(f,b,e,v,n,t,s){if(f.fbq)return;n=f.fbq=function() {n.callMethod? n.callMethod.apply(n,arguments):n.queue.push(arguments)} ;if(!f._fbq)f._fbq=n; n.push=n;n.loaded=!0;n.version='2.0';n.queue=[];t=b.createElement(e);t.async=!0; t.src=v;s=b.getElementsByTagName(e)[0];s.parentNode.insertBefore(t,s)}(window, document,'script','https://connect.facebook.net/en_US/fbevents.js'); fbq('init', '1641728616063202'); fbq('track', "PixelInitialized", {}); (function(h,o,t,j,a,r){ h.hj=h.hj||function(){(h.hj.q=h.hj.q||[]).push(arguments)}; h._hjSettings={hjid:2318163,hjsv:6}; a=o.getElementsByTagName('head')[0]; r=o.createElement('script');r.async=1; r.src=t+h._hjSettings.hjid+j+h._hjSettings.hjsv; a.appendChild(r); })(window,document,'https://static.hotjar.com/c/hotjar-','.js?sv='); search file_upload Submit your research search menu close search Browse Gateways & Collections How to Publish Submit your Research My Submissions Article Guidelines Article Guidelines (New Versions) Open Data, Software and Code Guidelines Open Data and Accessible Source Materials Guidelines (HSS) Open Data, Software and Code Guidelines (PSE) Prepublication Checks Production Process Posters and Slides Guidelines Document Guidelines Article Processing Charges Peer Review Finding Article Reviewers About How it Works For Reviewers Our Advisors Policies Glossary FAQs For Developers Newsroom Contact My Research Submissions Content and Tracking Alerts My Details Sign In file_upload Submit your research { "@context": "https://schema.org", "@type": "ScholarlyArticle", "mainEntityOfPage": { "@type": "WebPage", "@id": "https://f1000research.com/articles/6-1976" }, "headline": "bcbioRNASeq: R package for bcbio RNA-seq analysis", "datePublished": "2017-11-08T15:44:01", "dateModified": "2018-06-20T15:36:24", "author": [ { "@type": "Person", "name": "Michael J. Steinbaugh" }, { "@type": "Person", "name": "Lorena Pantano" }, { "@type": "Person", "name": "Rory D. Kirchner" }, { "@type": "Person", "name": "Victor Barrera" }, { "@type": "Person", "name": "Brad A. Chapman" }, { "@type": "Person", "name": "Mary E. Piper" }, { "@type": "Person", "name": "Meeta Mistry" }, { "@type": "Person", "name": "Radhika S. Khetani" }, { "@type": "Person", "name": "Kayleigh D. Rutherford" }, { "@type": "Person", "name": "Oliver Hofmann" }, { "@type": "Person", "name": "John N. Hutchinson" }, { "@type": "Person", "name": "Shannan Ho Sui" } ], "publisher": { "@type": "Organization", "name": "F1000Research", "logo": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 480, "width": 60 } }, "image": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 1200, "width": 150 }, "description": "RNA-seq analysis involves multiple steps, from processing raw sequencing data to identifying, organizing, annotating, and reporting differentially expressed genes. bcbio is an open source, community-maintained framework providing automated and scalable RNA-seq methods for identifying gene abundance counts. We have developed bcbioRNASeq, a Bioconductor package that provides ready-to-render templates, objects and wrapper functions to post-process bcbio RNA sequencing output data. bcbioRNASeq helps automate the generation of high-level RNA-seq reports, facilitating the quality control analyses, identification of differentially expressed genes and functional enrichment analyses." } { "@context": "http://schema.org", "@type": "BreadcrumbList", "itemListElement": [ { "@type": "ListItem", "position": "1", "item": { "@id": "https://f1000research.com/", "name": "Home" } }, { "@type": "ListItem", "position": "2", "item": { "@id": "https://f1000research.com/browse/articles", "name": "Browse" } }, { "@type": "ListItem", "position": "3", "item": { "@id": "https://f1000research.com/articles/6-1976/v2", "name": "bcbioRNASeq: R package for bcbio RNA-seq analysis" } } ] } Home Browse bcbioRNASeq: R package for bcbio RNA-seq analysis ALL Metrics - Views Downloads Get PDF Get XML Cite How to cite this article Steinbaugh MJ, Pantano L, Kirchner RD et al. bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.12688/f1000research.12093.2 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. Close Copy Citation Details Export Export Citation Sciwheel EndNote Ref. Manager Bibtex ProCite Sente EXPORT Select a format first Track Share ▬ ✚ Software Tool Article Revised bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] Michael J. Steinbaugh https://orcid.org/0000-0002-2403-2221 1 , Lorena Pantano https://orcid.org/0000-0002-3859-3249 1 , Rory D. Kirchner https://orcid.org/0000-0003-4814-5885 1 , [...] Victor Barrera https://orcid.org/0000-0003-0590-4634 1 , Brad A. Chapman 1 , Mary E. Piper https://orcid.org/0000-0003-2699-3840 1 , Meeta Mistry 1 , Radhika S. Khetani https://orcid.org/0000-0003-2430-2970 1 , Kayleigh D. Rutherford 1 , Oliver Hofmann 2 , John N. Hutchinson https://orcid.org/0000-0002-7804-7576 1 , Shannan Ho Sui 1 Michael J. Steinbaugh https://orcid.org/0000-0002-2403-2221 1 , Lorena Pantano https://orcid.org/0000-0002-3859-3249 1 , [...] Rory D. Kirchner https://orcid.org/0000-0003-4814-5885 1 , Victor Barrera https://orcid.org/0000-0003-0590-4634 1 , Brad A. Chapman 1 , Mary E. Piper https://orcid.org/0000-0003-2699-3840 1 , Meeta Mistry 1 , Radhika S. Khetani https://orcid.org/0000-0003-2430-2970 1 , Kayleigh D. Rutherford 1 , Oliver Hofmann 2 , John N. Hutchinson https://orcid.org/0000-0002-7804-7576 1 , Shannan Ho Sui 1 PUBLISHED 20 Jun 2018 Author details Author details 1 Harvard T.H. Chan School of Public Health, Boston, MA, 02115, USA 2 University of Melbourne Center for Cancer Research, Melbourne, VIC, 3000, Australia Michael J. Steinbaugh Roles: Conceptualization, Methodology, Software, Validation, Writing – Original Draft Preparation, Writing – Review & Editing Lorena Pantano Roles: Conceptualization, Methodology, Software, Validation, Visualization, Writing – Original Draft Preparation, Writing – Review & Editing Rory D. Kirchner Roles: Conceptualization, Software, Writing – Review & Editing Victor Barrera Roles: Software, Writing – Review & Editing Brad A. Chapman Roles: Writing – Original Draft Preparation, Writing – Review & Editing Mary E. Piper Roles: Software, Writing – Original Draft Preparation, Writing – Review & Editing Meeta Mistry Roles: Writing – Review & Editing Radhika S. Khetani Roles: Writing – Original Draft Preparation, Writing – Review & Editing Kayleigh D. Rutherford Roles: Validation, Writing – Review & Editing Oliver Hofmann Roles: Funding Acquisition, Writing – Review & Editing John N. Hutchinson Roles: Conceptualization, Funding Acquisition, Software, Writing – Original Draft Preparation, Writing – Review & Editing Shannan Ho Sui Roles: Funding Acquisition, Supervision, Writing – Review & Editing OPEN PEER REVIEW DETAILS REVIEWER STATUS This article is included in the RPackage gateway. This article is included in the Bioconductor gateway. This article is included in the Bioinformatics gateway. Abstract RNA-seq analysis involves multiple steps, from processing raw sequencing data to identifying, organizing, annotating, and reporting differentially expressed genes. bcbio is an open source, community-maintained framework providing automated and scalable RNA-seq methods for identifying gene abundance counts. We have developed bcbioRNASeq, a Bioconductor package that provides ready-to-render templates, objects and wrapper functions to post-process bcbio RNA sequencing output data. bcbioRNASeq helps automate the generation of high-level RNA-seq reports, facilitating the quality control analyses, identification of differentially expressed genes and functional enrichment analyses. READ ALL READ LESS Keywords RNA-seq, quality control, differential expression, functional analysis, data import, visualization, report writing, R Markdown Corresponding Author(s) Michael J. Steinbaugh ( [email protected] ) Close Corresponding author: Michael J. Steinbaugh Competing interests: No competing interests were disclosed. Grant information: The author(s) declared that no grants were involved in supporting this work. Copyright: © 2018 Steinbaugh MJ et al . This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. How to cite: Steinbaugh MJ, Pantano L, Kirchner RD et al. bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.12688/f1000research.12093.2 ) First published: 08 Nov 2017, 6 :1976 ( https://doi.org/10.12688/f1000research.12093.1 ) Latest published: 20 Jun 2018, 6 :1976 ( https://doi.org/10.12688/f1000research.12093.2 ) Revised Amendments from Version 1 The new version of the article contains the following updates: We have added a new figure (Figure 1) to describe in more detail the structure of the object the package uses. It has been adapted from the RangedSummarizeExperiment object and shows where the different count data and metadata are stored. We have updated the package to use the GRanges structure to store the genomic position of genes. This should help to integrate gene expression with epigenetic information, such as ChIP-Seq data, when trying to find features close to genes using the already available methods for Granges object. We have updated the name of some functions to match the Bioconductor style. We have added a new section at the end, “Functional Analysis” that describes the first steps for functional analysis after using the two other templates the package generates (quality control and differential expression). This section focuses on how to work with the data generated from the other templates to complete a functional analysis. We have focused only on the first few steps as the functional analysis follows published workflows. All figures and code lines have been updated to match the version of the package. We have added a discussion section after the quality control section to give examples of possible interpretations of the figures the package produces. The intention is to provide guidelines to the user on how to act accordingly based on the information generated in the figures. We have added a link to the session info from the computer used to run the code shown in the article. The new version of the article contains the following updates: We have added a new figure (Figure 1) to describe in more detail the structure of the object the package uses. It has been adapted from the RangedSummarizeExperiment object and shows where the different count data and metadata are stored. We have updated the package to use the GRanges structure to store the genomic position of genes. This should help to integrate gene expression with epigenetic information, such as ChIP-Seq data, when trying to find features close to genes using the already available methods for Granges object. We have updated the name of some functions to match the Bioconductor style. We have added a new section at the end, “Functional Analysis” that describes the first steps for functional analysis after using the two other templates the package generates (quality control and differential expression). This section focuses on how to work with the data generated from the other templates to complete a functional analysis. We have focused only on the first few steps as the functional analysis follows published workflows. All figures and code lines have been updated to match the version of the package. We have added a discussion section after the quality control section to give examples of possible interpretations of the figures the package produces. The intention is to provide guidelines to the user on how to act accordingly based on the information generated in the figures. We have added a link to the session info from the computer used to run the code shown in the article. See the authors' detailed response to the review by Charlotte Soneson See the authors' detailed response to the review by Davide Risso READ REVIEWER RESPONSES Introduction RNA sequencing (RNA-seq) analysis seeks to identify differential expression among groups of samples, providing insights into the underlying biology of a system under study 1 . Automating a full analysis from raw sequence data to functionally annotated gene results requires the coordination of multiple steps and tools. From the first data processing steps to quantify gene expression, to the data quality checks necessary for identification of differentially expressed genes 2 and functionally enriched categories, RNA-seq analysis involves the repetition of commands using various tools. This is done on a per-sample basis, and each step can require varying degrees of user intervention. As a bioinformatics core facility that processes a large number of RNA-seq datasets, we have developed a Bioconductor (BioC) 3 package called bcbioRNASeq to aggregate the outputs of tools for RNA-seq quality control (QC), differential expression and functional enrichment analysis as much as possible, while still retaining full, flexible control of critical parameters. This package relies on the output of bcbio , a Python framework that implements best-practice pipelines for fully automated high-throughput sequencing analysis (including RNA-seq, variant discovery, and ChIP-seq). bcbio is a community driven resource that handles the data processing component of sequencing analysis, providing researchers with more time to focus on the downstream biology. For RNA-seq data, bcbio generates QC and gene abundance information compatible with multiple downstream BioC packages. We briefly describe some of the tools included in the bcbio RNA-seq pipeline to help our users understand the outputs of bcbio that are used in the bcbioRNASeq package. To ensure that the library generation and sequencing quality are suitable for further analysis, tools like FastQC 4 examine the raw reads for quality issues. Cutadapt 5 can optionally be used to trim reads for adapter sequences, along with other contaminant sequences such as polyA tails and low quality sequences with PHRED 6 , 7 quality scores less than five. Salmon 8 generates abundance estimates for known splice isoforms. In parallel, STAR 9 aligns the reads to the specified reference genome, and featureCounts 10 generates counts associated with known genes. bcbio assesses the complexity and quality of the RNA-seq data by quantifying ribosomal RNA (rRNA) content and the genomic context of alignments in known transcripts and introns using a combination of custom tools and Qualimap 11 . Finally, MultiQC 12 generates an interactive HTML report in which the metrics from all tools used during the analysis are combined into a single dynamic file. bcbio handles these first stages of RNA-seq data processing with little user intervention. The next stages of an RNA-seq analysis include assessing read and alignment qualities, identifying outlier samples, clustering samples, assessing model fit, choosing cutoffs and finally, identifying differentially expressed genes. These steps often occur in multiple iterations, and require more active analyst involvement to integrate multiple tools that accept input data with incompatible formats and properties (see Use Case section). For example, the featureCounts gene counts from STAR-based alignments (a simple matrix) are useful for quality control, providing many more quality metrics than the quasi-alignments from Salmon. However, the quasi-alignments from Salmon (which are imported by tximport into a list of matrices) have been shown to be more accurate when testing for differential gene expression 13 , 14 . Managing these disparate data types and tools can make analyses unnecessarily time consuming, and increases the risk of inconsistency between analyses. Given the complexity of the analysis, it is essential to report the final parameters and associated results in a cohesive, reproducible manner. bcbioRNASeq was developed to address these issues and ease the process of documentation and report generation. The package offers multiple R Markdown templates that are ready-to-render after configuration of a few parameters and include example text and code for quality control metrics, differential expression, and functional enrichment analyses. Although other packages have been developed to solve similar issues, bcbioRNASeq allows for tight integration with the bcbio framework, and provides a unified package with objects, functions and pre-made templates for fast and simple RNA-seq analysis and reporting. Methods Reading data As noted, bcbio runs a number of tools to generate QC metrics and compute gene counts from RNA-seq data. Additional information about the bcbio RNA-seq pipeline is available on readthedocs ). At the end of a bcbio run, the most important files are stored in a separate directory specified by the user in the bcbio configuration YAML file under the " upload: " parameter; this directory is named " final " by default. Within this directory, there is a dated project directory containing quality metrics, provenance information, and data derived from the analysis that have been aggregated across all samples, e.g. count files. In addition, there is a directory corresponding to each sample that contains the binary alignment map (BAM) files and Salmon count data for that sample. final/ 2018-01-01_illumina_rnaseq/ annotated_combined.counts bcbio-nextgen-commands.log bcbio-nextgen.log combined.counts combined.dexseq combined.dexseq.ann data_versions.csv multiqc/ programs.txt project-summary.yaml tx2gene.csv sample1/ qc/ salmon/ sample1-ready.bam sample1-ready.bam.bai sample1-ready.counts sample1-transcriptome.bam The final upload directory generated by bcbio is used as the input for bcbioRNASeq. Once the bcbio run is complete, you can open an R session and load the bcbioRNASeq package (source code is available at our GitHub repository ). Use the bcbioRNASeq() constructor function (see example below) to create a structured S4 object that contains all of the necessary information for downstream analysis. The only required argument when creating this object is uploadDir , which specifies the path to the bcbio final upload directory. Note that bcbioRNASeq will transform all sample metadata column names to lowerCamelCase format without spaces, dashes, periods or underscores; therefore the interestingGroups argument should be specified in the same format. The values that can be used for interestingGroups are the same as the column names found in the CSV file used to bcbio_nextgen.py . Additionally, specifying the organism of the dataset is strongly recommended, and must use the full Latin name (e.g. Mus musculus ); this enables automatic downloading of gene annotations from Ensembl. Once the S4 object is assigned, use the saveData() function to write the dataset to disk as an R Data file. Note that the following code block does not need to be run to reproduce the figures in this paper; its purpose is to describe how to load the data from a bcbio analysis into R using this package, facilitating use of the downstream functions. # Non-working example demonstrating how to load a bcbio run. # Load the pre-saved object instead (See use case below). library (bcbioRNASeq) bcb <- bcbioRNASeq ( uploadDir = file.path ( "gse65267" , "final" ), organism = "Mus musculus" , interestingGroups = "day" ) saveData (bcb) This S4 object is unique to the bcbioRNASeq package, as it contains all of the necessary data from the bcbio run required for analysis. From here, you can use various functions in bcbioRNASeq to perform analyses, make figures, and generate data tables and results files as we describe in later sections. This object is also used as the input for the R Markdown templates for report generation. First, we begin by describing the object in more detail. Object description We have designed a new S4 class named bcbioRNASeq ( Figure 1 ) that is an extension of RangedSummarizedExperiment 15 . The assays() slot contains Salmon quasi-alignment data imported with tximport 13 , and automatically generated DESeq2 16 count transformations that provide support for quality control plots. To see all available assays in the bcbioRNASeq object listed by name, use assayNames(bcb) . These matrices are described in more detail below: @assays : ShallowSimpleListAssays containing count matrices derived from Salmon quasi-aligned counts imported with tximport and processed with DESeq2. Accessible with assays() . – counts : raw counts, generated by salmon and imported with tximport. Also accessible with assay() . – tpm : transcripts per million (TPM), calculated by tximport. – length : gene length matrix, calculated by tximport. – normalized : Normalized counts, with DESeq2 sizeFactors applied. – rlog : regularized log transformation, calculated by DESeq2. – vst : variance stabilizing transformation, calculated by DESeq2. @rowRanges : GRanges describing the rows (genes) of the count matrices slotted in assays() . When organism is specified in the bcbioRNASeq() function call, gene annotations will be downloaded from Ensembl using AnnotationHub and ensembldb. Accessible with rowRanges() and/or rowData() , which returns a DataFrame() instead of GRanges . @colData : DataFrame describing the columns (samples) of the count matrices slotted in assays() . Also contains sample quality metrics from bcbio analysis, generated from aligned counts produced by STAR and featureCounts. These aligned counts are not saved in the object. Accessible with colData() . @metadata : list containing additional metadata relevant to the dataset and the information pertaining to/generated from previous steps in the workflow. Accessible with metadata() . – version : Version of bcbioRNASeq package used to generate the object. – level : Whether counts are loaded at gene (default) or transcript level. – caller : Caller used to generate the counts. Defaults to salmon. – countsFromAbundance : Parameter used when reading transcript abundance with tximport. – uploadDir : Path to bcbio final upload directory. – sampleDirs : Paths of sample directories contained in bcbio upload directory. – sampleMetadataFile : Path to custom sample metadata file. Can be used to override the sample metadata saved in the bcbio run YAML, but is not normally needed. – projectDir : Path to project directory in bcbio upload. – template : Name of YAML file used to configure bcbio run. – runDate : Date of bcbio run completion. – interestingGroups : Groups of interest to use by default for quality control plot colors. – organism : Latin species name (e.g. " Homo sapiens "). – genomeBuild : Genome build (e.g. "GRCm38"). – ensemblRelease : Ensembl release version (e.g. 90). – rowRangesMetadata : Metadata describing the ensembldb package used with AnnotationHub to define the rowRanges . – gffFile : Transcript annotation file path, if used instead of Ensembl metadata. – tx2gene : Transcript-to-gene identifier mappings. – lanes : Number of flow cell lanes used during sequencing. – yaml : bcbio run YAML containing summary statistics and sample metadata saved during configuration. – dataVersions : Genome versions used by bcbio. – programVersions : Program versions used by bcbio. – bcbioLog : bcbio run log. – bcbioCommandsLog : bcbio commands log. – allSamples : Whether the object contains all samples from the run. – call : bcbioRNASeq() function call used to create the S4 object. – date : Date the bcbio run was loaded into R with bcbioRNASeq() . – wd : Working directory path. – utilsSessionInfo : utils::sessionInfo() output. – devtoolsSessionInfo : devtools::session_info() output. Figure 1. bcbioRNASeq S4 object structure. The bcbioRNASeq object is an S4 class extension of the RangedSummarizedExperiment container. The assays slot contains several matrices with sample IDs as column names and gene IDs as row names, including raw counts and various derivations of the raw counts, and can be accessed via the assays() function. Gene IDs tie these assays to additional information about the genes stored as a GRanges object, and can be accessed via the rowRanges() function. Similarly, sample IDs tie the assays to further sample information such as run metrics from bcbio and experimental factors also stored as a DataFrame , and can be accessed via the colData() function. Non-gene or sample specific metadata about the entire run such as tool versions and genome builds is stored as a list and can be accessed via the metadata() function. Use case To demonstrate the functionality and configuration of the package, we have used an experiment from the Gene Expression Omnibus (GEO) public repository of expression data as an example use case. The RNA-seq data is from a study of acute kidney injury in a mouse model ( GSE65267 ) 17 . The study aims to identify differentially expressed genes in progressive kidney fibrosis and contains samples from mouse kidneys at several time points (n = 3, per time point) after folic acid treatment. From this dataset, we are using a subset of the samples for our use case: before folic acid treatment, and 1, 3, 7 days after treatment. A pre-computed version of the example bcbioRNASeq object used in this workflow ( bcb.rda ) and the code for reproduction are available at the f1000v2 branch of our package repository on GitHub . First, load the bcbioRNASeq object and a few other libraries to demonstrate how to access the different types of information contained in the object. # Load the pre-saved object library (bcbioRNASeq) loadRemoteData ( "https://github.com/hbc/bcbioRNASeq/raw/f1000v2/data/bcb.rda" ) The counts() function returns the abundance estimates generated by Salmon. Read counts for each sample in the dataset are aggregated into a matrix, in which columns correspond to samples and rows represent genes. Multiple normalized counts matrices are saved in the bcbioRNASeq object, and are accessible with the normalized argument: 1. Raw counts ( normalized = FALSE ). 2. DESeq2 normalized counts, with sizeFactors applied ( normalized = TRUE ). 3. Transcripts per million ( normalized = "tpm" ). 4. Trimmed mean of M-values normalization method ( normalized = "tmm" ). 5. Relative log expression ( normalized = "rle" ). 6. Regularized log transformation ( normalized = "rlog" ). 7. Variance stabilization transformation ( normalized = "vst" ). Exporting quantified data The package contains multiple convenience functions to extract the expression abundances described above. Outlined below are the steps to save these counts external to the bcbioRNASeq object. These steps utilize functions from the DESeq2, edgeR and tximport packages both directly as well as within wrapper functions. For discussions on RNA-seq data normalization methods and count formats see 18 ; we typically save at least the DESeq2 normalized counts (library size adjusted) and transcripts per million counts (gene length adjusted) for further analyses. raw <- counts (bcb, normalized = FALSE ) normalized <- counts (bcb, normalized = TRUE ) tpm <- counts (bcb, normalized = "tpm" ) rlog <- counts (bcb, normalized = "rlog" ) vst <- counts (bcb, normalized = "vst" ) saveData (raw, normalized, tpm, rlog, vst) writeCounts (raw, normalized, tpm, rlog, vst) Quality control steps A typical RNA-seq analysis requires multiple quality control assessments at the read, alignment, sample and model level. Most of the data required to make these assessments is automatically generated by bcbio; the bcbioRNASeq package makes it easier for users to access it. For instance, the Qualimap tool runs as part of the bcbio pipeline and generates various metrics that can be used to assess the quality of the data and consistency across samples. The output of Qualimap is stored in the bcbioRNASeq object, and the package has several functions to visualize this output in a graphical format. These plots can be used to check data quality. Visual thresholds appear in many plots to help the user to assess quality. For example, a vertical dashed line is used as a cutoff threshold, helping to identify samples that require additional inspection. These default suggested cutoff values are suitable for human/mouse RNA-seq data sets (based on our experience), but should be manually modified for other species. The package does not focus on automatically determining threshold values, but instead provides a mechanism by which to visually communicate the quality of the data. The default cutoffs for these thresholds can be changed using function arguments. The metrics() function allows the user to extract a data.frame containing all metrics information used by the QC report functions. In this way, custom figures can also be easily created using the same data but with user-preferred packages. Below, we provide several examples of recommended QC steps for RNA-seq data with a short explanation outlining their utility. By default, normalized counts generated with the DESeq2 varianceStabilizingTransformation() function is used to plot gene expression. This can be changed using the option named normalized in each function used to plot expression values. Read statistics Total reads per sample and mapping rate are metrics that can help identify imbalances in sequencing depth or failures among the samples in a dataset ( Figure 2A–B ). Generally, we expect to see a similar sequencing depth across all samples in a dataset and mapping rates greater than 75%. Low genomic mapping rates are indicative of sample contamination, poor sequencing quality or other artifacts. Some figures show lines indicating the optimal minimum number for the quality metric. plotTotalReads (bcb) plotMappingRate (bcb) Figure 2. Read, alignment, genomic context, and gene detection statistics. Sample classes, as defined by the interestingGroups argument, are represented by the different colors defined in the legend of each plot. Vertical dashed black lines indicate suggested cutoff values. The total reads plot ( A ) indicates the total number of reads sequenced per sample. There is some variability, but in all cases the coverage is higher than 20 million reads, which we consider as sufficient to give good quality gene quantification. ( B ) shows the percentage of reads mapping to the reference genome. Here, all samples are well within recommended ranges, having well over 25 million reads and almost 90% of reads mapping. The exonic and intronic mapping rate plots ( C and D ) indicate the percentage of reads mapping to exons or introns, respectively. Here, all samples are within recommended ranges, with samples from day 3 and day 7 showing higher proportions of reads mapping to intronic versus exonic regions as compared to the day 1 and normal sample classes. The genes detected plot ( E ) indicates the total number of genes for each sample with at least one mapped read. Optimal minimum gene detection values will vary based on an organism’s transcriptome size. The gene detection saturation plot ( F ) shows the relationship between the number of reads mapped and the number of genes detected. If this trend is not linear, it indicates that the sequencing may have been saturated in terms of detecting gene expression. For this specific data, it shows no saturation yet for gene detection, meaning that more genes could be detected if coverage were increased for the samples with lower number of reads. Genomic context For RNA-seq, the majority of reads should map to exons and not introns. Ideally, at least 60% of total reads should map to exons. High levels of intronic mapping may indicate high proportions of nuclear RNA or DNA contamination. Samples should also have rRNA contamination rates below 10% (not shown) ( Figure 2C–D ). plotExonicMappingRate (bcb) plotIntronicMappingRate (bcb) Number of genes detected Determining how many genes are detected relative to the number of mapped reads is another good way to assess the sample quality ( Figure 2D–E ). Ideally, all samples will have similar numbers for genes detected, and samples with higher number of mapped reads will have more genes detected. Large differences in gene detection numbers between samples can introduce biases and should be monitored at later steps for potential influence on sample clustering. plotGenesDetected (bcb) plotGeneSaturation (bcb) Counts per gene Comparing the distribution of normalized gene counts across samples is one way to assess sample similarity within a dataset. For this figure, normalized counts comes from the trimmed mean of M-values, calculated by edgeR. This is output is from the function cpm(normalized.lib.sizes = TRUE) after applying calcNormFactors to a DGEList object containing the raw counts. We would expect similar count distributions for all genes across the samples unless the library sizes or total RNA expression are different ( Figure 3 ). The plotCountsPerGene() and plotCountDensity() functions provide two ways to visualize this comparison. Zero counts are removed in these figures and the values are log2 scale. plotCountsPerGene (bcb) plotCountDensity (bcb) Figure 3. Gene count distributions. Normalized count distributions are displayed as box plots ( A ) and density plots ( B ). The log2 TMM-normalized counts per gene normalization method equates the overall expression levels of genes between samples under the assumption that a majority of them are not differentially expressed 19 . Therefore, by normalizing for total RNA expression by sample, we expect the spread of the log2 TMM-normalized counts per gene to be similar for every sample. Sample classes (as set with the interestingGroups argument) are highlighted in different colors. Here, there is high similarity among the samples. This plot shows how all the samples have very similar gene expression profile what makes them comparable for the differential expression analysis. Model fitting It is important to explore the fit of the model for a given dataset before performing differential expression analysis. The normalized and transformed data can be used to assess the variance-expression level relationship in the data, to identify which method is best at stabilizing the variance across the mean for downstream visualization. The plotMeanSD() function wraps the output of different variance stabilizing methods (including the varianceStabilizingTransformation() and rlog() transformations from the DESeq2 package) and plots them with the vsn package’s meanSdplot() function 20 ( Figure 4 ). For the example data, the vst and rlog transformations work well with respect to reducing the noise of low expression genes, illustrated by a flatter distribution observed in these plots. The user may choose the desired transformation for their figures using the normalized argument. plotMeanSD (bcb) Figure 4. Variance stabilizing transformations. Plots showing the standard deviation of normalized counts using log2() (top left), rlog() (top right), VST ( varianceStabilizingTransformation() ) (bottom left), and TMM (bottom right) transformations by rank(mean) . The red line denotes the running median estimator. As these panels show, the rlog and VST transformations greatly reduce the standard deviation and variance across the mean. Dispersion Another plot that is important to evaluate when performing QC on RNA-seq data is the plot of dispersion versus the mean of normalized counts. For a good dataset, we expect the dispersion to decrease as the mean of normalized counts increases for each gene. The plotDispEsts() function provides easy access to model information stored in the bcbioRNASeq object, using the plotting code provided in the DESeq2 package ( Figure 5 ). plotDispEsts (bcb) Figure 5. DESeq2 dispersion plot. Sample similarity within groups The QC metrics assessed up to this point are performed to get a global assessment across the dataset and look for similar trends across all samples. However, often we have a dataset in which samples can be classified into groups and it is common to interrogate how similar replicates are to each other within those groups, and the relationship between groups. To this end, bcbioRNASeq provides functions to perform this level of QC with Inter-Correlation Analyses (ICA) and Principal Components Analyses (PCA) between samples. Furthermore, we can use the results of the PCA to identify covariates that correlate with principal components. Since these analyses are based on variance measures, it is recommended that the variance stabilized (vst) or rlog transformed counts be used. Using these transformed counts minimizes large differences in sequencing depth and helps normalize all samples to a similar dynamic range. Simple logarithmic transformations of normalized count values tend to generate a lot of noise for low expressed genes, which can consequently dominate the calculations in the similarity analysis. An rlog transformation will shrink the values of low counts towards the genes’ average across samples, without affecting the high expression genes. Inter-correlation analysis and hierarchical clustering Inter-correlation analysis allows us to look at how closely samples are related to each other by first computing pair-wise correlations between expression profiles of all samples in the dataset and then clustering based on those correlation values. Samples that are similar to one another will be highly correlated and will cluster together. We expect samples from the same group to cluster together ( Figure 6 ), although this is not always the case. We can also identify potential outlier samples using a correlation heatmap, if there are samples that show low correlation with all other samples in the dataset. For more control over the graphic parameters of the correlation heatmap, other packages can be used to generate these plots by using the normalized data, accessed with counts(bcb, normalized = "vst") , as input. One example is the pheatmap() function from the pheatmap package 21 , which underlies this plot. plotCorrelationHeatmap (bcb) Figure 6. Sample correlations. All pairwise sample Pearson correlations are shown. Correlations are clustered by both row and column, with sample classes (as set with the interestingGroups argument) highlighted across the top of the heatmap. Here, the sample classes cluster well, with the normal and day 1 samples showing the highest intra-group correlations. Principal component analysis (PCA) PCA is a multivariate technique that allows us to summarize the systematic patterns of variations in the data 22 . PCA takes the expression levels for genes and transforms them in principal component space, reducing each sample to a single point. It allows us to separate samples by expression variation, and identify potential sample outliers. The PCA plot is a valuable way to explore both inter- and intra-group clustering ( Figure 7 ). As with the correlation heatmap plots, other packages can be used to generate these kinds of plots from the normalized data, which can be accessed with counts(bcb, normalized = "vst") . plotPCA (bcb, label = FALSE ) Figure 7. Principal component analysis. The first two principal components of the gene expression dataset are plotted here for each of the samples. Sample classes (as set with the interestingGroups argument) are highlighted in different colors. Alternatively, sample labels can be added with the " label = TRUE " argument (not shown) to identify individual samples, which is particularly useful for identifying outliers. Here, we see good clustering of the samples by group with no apparent outliers. Correlation of covariates with PCs When there are multiple factors that can influence the results of a given experiment, it is useful to assess which of them is responsible for the most variance as determined by PCA ( Figure 8 ). The plotPCACovariates() function passes transformed count data and metadata from the bcbioRNASeq object to the degCovariates() function of the DEGreport package 23 . This method adapts the method described by Daily et al . for which they integrated a method to correlate covariates with principal components values to determine the importance of each factor 24 ( Figure 8 ). For the example data, no covariate shows a strong correlation with the PCs with FDR < 0.05, indicating that there is no need for correction for any covariates during the differential expression analysis. plotPCACovariates (bcb, fdr = 0.1 ) Figure 8. Principal component covariate correlations. PCA is performed on transformed counts and correlations between principal components and the different metadata covariates are computed. Significant correlations (FDR < 0.1) are shaded from blue (anti-correlated) to orange (correlated), with non-significant correlations shaded in gray. Here, non of variables show a significant correlation with the any principal component of the data. Asterisks indicates p-value < 0.05. QC discussion With the QC figures, outlier samples can be identified and may be removed from the set of samples. For instance, in Figure 2B , if one samples shows a very low mapping rate (< 50%) compared to others, this could indicate an issue with the RNA material or the library preparation. The same reasoning applies to Figure 2C , that shows exonic mapping rates. In this case, low mapping rates may indicate DNA contamination if the rate is very low, and should be removed if only a minority of samples are affected. Moreover, Figure 8 can reveal clustering of samples correlated to some covariates, providing guidance on which variables to add to the formula during differential expression analysis. Differential expression analysis using DESeq2 Once the QC is complete and the dataset looks good, the next step is to identify differentially expressed genes (DEG). For this part of the workflow, we follow instructions and guidelines from the DESeq2 vignette, using Salmon-derived abundance estimates imported with tximport. As previously noted, a Differential Expression R Markdown template is available with the bcbioRNASeq package for these steps. The first step is to define the factors to include in the statistical analysis as a design formula. We chose to study the difference between the normal group and the day 7 group in our dataset. In our example, we have only one variable of interest; however, DESeq2 is able to model additional covariates. When additional variables are included, the last variable entered in the design formula should generally be the main condition of interest. More detailed instructions and examples are available in the DESeq2 vignette. # DESeqDataSet dds <- as (bcb, "DESeqDataSet" ) design (dds) <- ∼day dds <- DESeq (dds) vst <- varianceStabilizingTransformation (dds) saveData (dds, vst) Alpha level (FDR) cutoffs The results from DESeq2 include a column for the P values associated with each gene/test as well as a column containing P values that have been corrected for multiple testing (i.e. false discovery rate values). The multiple test correction method performed by default is the Benjamini Hochberg (BH) method 25 . Since it can be difficult to arbitrarily select an adjusted P value cutoff, the alphaSummary() function is useful for summarizing results for multiple adjusted P value or FDR cutoff values ( Table 1 ). alphaSummary ( object = dds, contrast = c ( factor = "day" , numerator = "7" , denominator = "0" ), alpha = c (0.1, 0.05) ) Table 1. Differentially expressed genes at different adjusted P value cutoffs. 0.1 0.05 LFC > 0 (up) 1521, 4% 1102, 2.9% LFC < 0 (down) 1694, 4.5% 1282, 3.4% outliers 632, 1.7% 632, 1.7% low_counts 12883, 34% 12883, 34% cutoff (mean count < 5) (mean count < 5) Differential expression analysis Use the results() function to generate a DESeqResults object containing the output of the differential expression analysis. The desired BH-adjusted P value cutoff value is specified here with the alpha argument (< 0.05 shown). res <- results ( object = dds, name = "day_7_vs_0" , alpha = 0.05 ) saveData (res) Mean average (MA) plot The plotMeanAverage() function plots the mean of the normalized counts versus the log2 fold changes for all genes tested ( Figure 9 ). plotMeanAverage (res) Figure 9. MA plot. Each point represents one gene, with mean expression levels across all samples plotted on the x-axis and the log2 fold change observed in the contrast of interest on the y-axis. Significant differentially expressed genes (with an adjusted P value less than the adjusted cutoff P value we chose earlier) are colored red 28 . Volcano plot The plotVolcano() function produces a volcano plot comparing significance (here the BH-adjusted P value) for each gene against the fold change (here on a log2 scale) observed in the contrast of interest 26 ( Figure 10 ). This function requires an Internet connection to map Ensembl gene IDs to gene names (symbols). plotVolcano (res) Figure 10. Volcano plot. This plot compares the amount of gene expression change (not strictly interpretable here as these fold changes were derived from an LRT test) to the significance of that change (here plotted as the -log10 transformation of the multiple test adjusted P value), with each point representing a single gene. Options are available to highlight top gene candidates by shading (here, genes in the green shaded areas are bounded by a minimal fold change and -log10 adjusted P value cutoffs) and by text labeling. The two (optional) marginal plots showing the distributions of the log2 fold changes and negative log10 adjusted P values are useful in assessing cutoff choices and trade-offs. Gene expression heatmap The plotDEGHeatmap() function produces a gene expression heatmap for visualizing the expression of differentially expressed (DE) genes across samples. The heatmap shows only DE genes on a per-sample basis, using an additional log2 fold change cutoff. By default, this plot is scaled by row and uses the ward.D2 method for clustering 27 . The gene expression heatmap is a nice way to explore the consistency in expression across replicates or differences in expression between sample groups for each of the DE genes ( Figure 11 ). plotDEGHeatmap ( results = res, counts = vst) Figure 11. Differentially Expressed Gene Heatmap. The heatmap shows only DE genes on a per-sample basis, using an additional log2 fold change cutoff. By default, this plot is scaled by row (centered and scaled) and uses the ward.D2 method for clustering 27 . Results are clustered by both row and column, with sample classes (as set with the interestingGroups parameter) highlighted across the top of the heatmap. Heatmaps are drawn internally with the pheatmap() function of the pheatmap package 21 . Specific genes plot In addition to looking at the overall results from the differential expression analysis, it is useful to plot the expression differences for a handful of the top differentially expressed genes. This helps to check the quality of the analysis by validating the expression for genes that are identified as significant. It is also helpful to visualize trends in expression change across the various sample groups ( Figure 12 ). As bcbioRNASeq is integrated with the DEGreport package 23 , we can use DEGreport’s degPlot() function to view the expression of individual DE genes. degPlot ( bcb, res = res, n = 3 , slot = "vst" , log = TRUE, ann = c ( "geneID" , "geneName" ), xs = "day" ) Figure 12. Individual gene expression patterns. Gene expression patterns are shown for the top 3 (as selected in the function options) differentially expressed genes (by BH-adjusted P value). Normalized, transformed counts are shown for each replicate from each sample group; groups are set within the function options. Interestingly, even though our analysis compared day 7 to normal, all 3 genes show their greatest increases in gene expression at day 3, with some leveling off or relative decreases in expression at day 7. Detecting patterns The full set of example data is from a time course experiment, as described previously. Up to this point, we have only compared gene expression between two time points (normal and day 7), but we can also analyze the whole dataset to identify genes that show any change in expression across the different time points. As recommended by DESeq2, the best approach for this type of experimental design is to perform a likelihood ratio test (LRT) to test for differences in gene expression between any of the sample groups in the context of the time course. More information about time-course experiments and LRT is available in the DESeq2 vignette . This approach will yield a list of differentially expressed genes, but will not report how the expression is changing. Visualizing patterns of expression change amongst the significant genes is helpful in identifying groups of genes that have similar trends, which in turn can help determine a biological reason for the changes we observe ( Figure 13 ). The DEGreport package includes the degPatterns() function, which is designed to extract and plot genes that have a similar trend across the various time points. More information about this function can be found in the DEGreport package 23 . Note that this function works only with significant genes; significants() returns the significant genes based on log2FoldChange and padj values (0 and 0.05 respectively, by default). dds_lrt <- DESeq (dds, test = "LRT" , reduced = ˜ 1 ) res_lrt <- results (dds_lrt) ma <- counts (bcb, "vst" )[ significants (res, fc = 2 ), ] res_patterns <- degPatterns ( ma = ma, metadata = colData (bcb), time = "day" , minc = 60 ) saveData (dds_lrt, res_lrt, res_patterns) res_patterns[[ "plot" ]] Figure 13. Clustered expression patterns. Clustered and scaled expression patterns for the top 500 differentially expressed genes. Each group represents a gene expression pattern shared among different DE genes, with the number of groups determined by the expression correlation patterns of the groups. Box plots are shown for the expression patterns of each gene within the group to give a better idea of how well the groupings fit the expression data (DEGreport_1.14.1 used for this plot.) The output of the degPatterns() function is the plot, as well as a list object that contains a data.frame with two columns – the "genes" and the corresponding "cluster" number. To extract genes from a specific cluster for further analysis, the base function subset() can be utilized as follows to obtain the list of genes from cluster 3: subset (res_patterns[[ "df" ]], cluster == 3, select = "genes" ) Summarize analysis Finally, at the end of the analysis, a results table can be extracted containing the DE genes at a specified log fold change threshold. These results can be written to files in the specified output folder. In addition to the DE genes, a detailed summary and description of results and the output files are generated along with the cutoffs used to identify the significant genes. This function requires an Internet connection to map Ensembl gene IDs to gene names (symbols). res_tbl <- resultsTables (res, lfc = 1) Top tables Once the results table object is ready, the top up- and down-regulated genes can now be displayed. Here, we output the top 5 DE genes in each direction ( Table 2 ). topTables (res_tbl, n = 5 ) Table 2. Top differentially expressed genes. ensgene baseMean log2FoldChange padj ENSMUSG00000024164 ENSMUSG00000029304 ENSMUSG00000027962 ENSMUSG00000040405 ENSMUSG00000022037 16284 199735 2744 20543 7752 8.21 5.14 5.23 3.44 3.21 1.96e-185 4.25e-68 1.92e-30 4.18e-29 3.80e-28 ensgene baseMean log2FoldChange padj ENSMUSG00000049152 ENSMUSG00000011179 ENSMUSG00000060002 ENSMUSG00000062908 ENSMUSG00000064370 13001 3090 23614 2594 183932 -2.30 -2.74 -2.14 -2.09 -1.87 1.23e-25 8.15e-21 4.43e-19 6.97e-19 7.14e-19 File outputs Output files from this use case include the following gene counts files (output at the count normalization step): normalized_counts.csv.gz : Use to evaluate individual genes and/or generate plots. These counts are normalized for the variation in sequencing depth across samples. tpm.csv.gz : Transcripts per million, scaled by length and also suitable for plotting. raw_counts.csv.gz : Only use to perform a new differential expression analysis. These counts will vary across samples due to differences in sequencing depth, and have not been normalized. Do not use this file for plotting genes. If desired, variance stabilized counts (e.g. rlog, vst) can also be included for future use in variance based plotting methods such as PCA. DEG tables containing the differential expression results from the analysis summary step are sorted by BH-adjusted P value, and contain the following columns: geneID : Ensembl gene identifier. baseMean : Mean of the normalized counts per gene for all samples. log2FoldChange : log2 fold change. pvalue : Wald test P value. padj : BH-adjusted Wald test P value (corrected for multiple comparisons; false discovery rate). Functional analysis To gain greater biological insight into the list of DE genes, it is helpful to perform functional analysis. We provide the Functional Analysis R Markdown template, which contains code from the clusterProfiler package 29 . The functional analysis includes over-representation analysis to identify significantly enriched biological processes among a list of DE genes, and gene set enrichment analysis (GSEA) to identify pathways significantly enriched for genes with larger fold changes. The functional analysis results suggest genes/pathways that may be involved with the condition of interest; however, the results are not conclusive and all identified processes/pathways require experimental validation. For over-representation analysis, clusterProfiler requires a vector of background genes tested for differential expression, a vector of significant DE genes, and a named vector of fold changes associated with each DE gene. # Generate the gene IDs for the DE list of genes and the background list of genes library (clusterProfiler) all_genes % .[! is.na (res[[ "padj" ]])] %>% as.character () sig_genes <- significants ( object = res, fc = 1, padj = 0.05 ) # Generate fold change values for significant results sig_results <- as.data.frame (res)[sig_genes, ] fold_changes <- sig_results$log2FoldChange names (fold_changes) <- rownames (sig_results) The enrichGO() function takes the significant gene list, the background gene list, and the ontology to test as input to perform statistical enrichment analysis of gene ontology (GO) terms using hypergeometric testing. # Run GO enrichment analysis ego <- enrichGO ( gene = sig_genes, OrgDb = "org.Mm.eg.db", keyType = "ENSEMBL", ont = "BP" universe = all_genes, qvalueCutoff = 0.05, readable = TRUE ) The GO enrichment analysis output is an S4 class object named enrichResult . The results can be extracted from this object using the slot () accessor function: ego_summary % as_tibble () %>% camel () The results table contains the following columns: id : GO identifier. description : GO process. geneRatio : number of DE genes associated with GO process / total number of DE genes. bgRatio : number of background genes associated with GO process / total number of background genes. pvalue : hypergeometric test P value. padj : BH-adjusted hypergeometric test P value (corrected for multiple comparisons; false discovery rate). qvalue : FDR-adjusted hypergeometric test P value (corrected for multiple comparisons; positive false discovery rate). geneID : gene symbols of all DE genes associated with GO process. There are a variety of plots that can be used to explore the enriched processes using the enrichResult object, including the dot plot, enrichment GO plot, and category netplot. The dot plot shows the number of DE genes associated with the top 25 GO terms (size) and the P -adjusted values for these terms. The order of GO terms is based on the gene ratio (number of significant genes associated with the GO term / total number of significant genes) ( Figure 14 ). # Dotplot of top 25 dotplot (ego, showCategory = 25) Figure 14. Enriched GO terms: dot plot. The 25 GO processes with the largest gene ratios are plotted in order of gene ratio. The size of the dots represent the number of genes in the significant DE gene list associated with the GO term and the color of the dots represent the P -adjusted values (BH). The enrichment GO plot shows the relationship between the top 25 most significantly enriched GO terms, by grouping similar terms together ( Figure 15 ). # Enrichment plot of top 25 enrichMap (ego, n = 25, vertex.label.cex = 0.5) Figure 15. Enriched GO terms: enrichment GO plot. The top 25 most significantly enriched GO terms (by P -adjusted values) are shown with similar terms grouped together. The color represents the P values relative to the other displayed terms (brighter red is more significant) and the size of the terms represents the number of genes that are significant from the significant genes list. The category netplot shows the relationships between the genes associated with the top five most significant GO terms and the fold changes of the significant genes associated with these terms (color) ( Figure 16 ). The size of the GO terms reflects the P values of the terms, with the more significant terms being larger. # Cnet plot for top 5 most significant GO processes cnetplot ( x = ego, showCategory = 5, foldChange = fold_changes, vertex.label.cex = 0.5 ) Figure 16. Enriched GO terms: category netplot. The top five most significant GO terms ( P -adjusted values) are plotted and connected with lines to associated DE genes. The fold changes of the significant genes associated with these terms are represented by color. The size of the GO terms reflects the P values of the terms, with the more significant terms being larger. The Functional Analysis template also includes code for performing over-representation analysis with GO terms and code to perform GSEA analysis using KEGG and GO terms. The final step in the template is the visualization of the genes and corresponding fold changes associated with the GSEA enriched pathways using the pathview package. R Markdown Templates To facilitate analyses and compile results into a report format, we have created easy-to-use R Markdown templates that are accessible in RStudio 30 , 31 . Once the bcbioRNASeq package has been installed, you can find these templates under F ile -> N ew F ile -> R M arkdown ... -> F rom T emplate . There are three main templates for RNA-seq: (1) Quality Control , (2) Differential Expression , and (3) Functional Analysis . You may need to restart RStudio to see the templates. It is recommended that users run the reports in the order described above, as there may be functions that depend on data generated from the previous report. If you are not using RStudio, you can create new documents based on the templates using the rmarkdown::draft() function: library (rmarkdown) draft ( file = "quality_control.Rmd" , template = "quality_control" , package = "bcbioRNASeq" ) draft ( file = "differential_expression.Rmd" , template = "differential_expression" , package = "bcbioRNASeq" ) draft ( file = "functional_analysis.Rmd" , template = "functional_analysis" , package = "bcbioRNASeq" ) The instructions above will create an R Markdown file from each of the templates. Each file begins with a YAML header, followed by sub-sections containing code chunks and some relevant text and/or a sub-heading to describe that step of the analysis. Each R Markdown file takes as input the bcbioRNASeq object, such that various functions from the package can be run on the data stored within the object to output figures, tables and carefully formatted results. Note you can add more text, headings and code chunks to the body of the R Markdown files to customize the reports as desired. Before rendering the file into a report you will want to run the prepareRNASeqTemplate() function in order to obtain the accessory files necessary for a fully working template. library (bcbioRNASeq) prepareRNASeqTemplate () Finally, the main analysis parameters need to be specified in the YAML section on the top part of each document. For this example we assume R objects are stored in a data subdirectory. Edit the following parameters in the Quality Control template: params: bcb_file: "data/bcb.rda" data_dir: "data" results_dir: "results/quality_control" bcb.rda refers to the bcbioRNASeq object. In the YAML section on the top of the Differential Expression template, modify the following parameters: params: bcb_file: "data/bcb.rda" design: !r formula(~day) contrast: !r c("day", "7", "0") alpha: 0.05 lfc_threshold: 0 data_dir: "data" results_dir: "results/differential_expression" dropbox_dir: "NULL" In the YAML section on the top of the Functional Analysis template, modify the following parameters (note that clusterProfiler and pathview must be installed): params: dds_file: "data/dds.rda" res_file: data/res.rda organism: "Mus musculus" go_class: "BP" alpha: 0.05 lfc_threshold: 0 data_dir: "data" results_dir: "results/functional_analysis" res.rda refers to a DESeqResults object from the DESeq2 package. The downloaded accessory files will be saved to your current working directory and should be kept together with your main analysis R Markdown files that were generated from the templates. These accessory files include: a) _output.yaml , to specify the R Markdown render format; b) _header.Rmd and c) _footer.Rmd , to add header/footer sections to the report; d) _setup.R , to fill in some of the parameters related to figures and rendering format; and e) bibliography.bib , BibTex file for citations. Conclusions Here we describe bcbioRNASeq, a Bioconductor package that provides functionality for quality assessment and differential expression analysis of RNA-seq experiments. This package supplements the bcbio community project, as it takes the output from automated bcbio RNA-seq runs as input, allowing for the data to be stored in a special S4 object that can easily be accessed for various steps of the RNA-seq workflow downstream of bcbio. Built as an open source project, bcbio is a well-supported and documented platform for effectively using current state-of-the-art RNA-seq methods. Taken together, bcbio and bcbioRNASeq provide a full framework for rapidly and accurately processing RNA-seq data. The package also provides a set of configurable templates to generate comprehensive HTML reports suitable for biological researchers. With the use of R Markdown, all steps of the analysis are fully configurable and traceable. We provide a full set of instructions for using bcbioRNASeq, including an example use case that demonstrate all of the main functionality. Quality control (pre- and post-quantification), model fitting, differential expression, and functional analysis provide a comprehensive set of metrics for evaluating the robustness of the RNA-seq results. Methods such as hierarchical clustering, principal components analysis, and time point analysis allow for an interactive examination of the data’s structure. Collectively, the workflow we describe can help researchers to identify true biological signal from technical noise and batch effects when analyzing RNA-seq experiments. Software and data availability Current source code: https://github.com/hbc/bcbioRNASeq Archived source code (v0.2.4), as at time of publication: http://doi.org/10.5281/zenodo.1256861 32 Workflow code: https://github.com/hbc/bcbioRNASeq/tree/f1000v2 License: MIT The data used in the use case can be accessed from NCBI GEO using accession GSE65267 . Competing interests No competing interests were disclosed. Grant information The author(s) declared that no grants were involved in supporting this work. Acknowledgments Thanks to Mike Love of the Department of Biostatistics and Department of Genetics at the University of North Carolina at Chapel Hill for providing advice regarding RNA-seq differential expression and the use of the DESeq2 and tximport packages. We thank Laura Lin of the Harvard Chan Bioinformatics Core for reviewing the code showed here. Faculty Opinions recommended References 1. Wang Z, Gerstein M, Snyder M: RNA-Seq: a revolutionary tool for transcriptomics. Nat Rev Genet. 2009; 10 (1): 57–63. PubMed Abstract | Publisher Full Text | Free Full Text 2. Love MI, Anders S, Kim V, et al. : RNA-Seq workflow: gene-level exploratory analysis and differential expression [version 2; referees: 2 approved]. F1000Res. 2016; 4 : 1070. Publisher Full Text 3. Huber W, Carey VJ, Gentleman R, et al. : Orchestrating high-throughput genomic analysis with bioconductor. Nat Methods. 2015; 12 (2): 115–121. PubMed Abstract | Publisher Full Text | Free Full Text 4. Andrews S: FastQC: a quality control tool for high throughput sequence data. 2010. Reference Source 5. Martin M: Cutadapt removes adapter sequences from high-throughput sequencing reads. EMBnet J. 2011; 17 (1): 10–12. Publisher Full Text 6. Ewing B, Hillier L, Wendl MC, et al. : Base-calling of automated sequencer traces using phred . i. accuracy assessment. Genome Res. 1998; 8 (3): 175–85. PubMed Abstract | Publisher Full Text 7. Ewing B, Green P: Base-calling of automated sequencer traces using phred . II. error probabilities. Genome Res. 1998; 8 (3): 186–94. PubMed Abstract | Publisher Full Text 8. Patro R, Duggal G, Love MI, et al. : Salmon provides fast and bias-aware quantification of transcript expression. Nat Methods. 2017; 14 (4): 417–419. PubMed Abstract | Publisher Full Text | Free Full Text 9. Dobin A, Davis CA, Schlesinger F, et al. : STAR: ultrafast universal RNA-seq aligner. Bioinformatics. 2013; 29 (1): 15–21. PubMed Abstract | Publisher Full Text | Free Full Text 10. Liao Y, Smyth GK, Shi W: featurecounts: an efficient general purpose program for assigning sequence reads to genomic features. Bioinformatics. 2014; 30 (7): 923–30. PubMed Abstract | Publisher Full Text 11. Okonechnikov K, Conesa A, García-Alcalde F: Qualimap 2: advanced multi-sample quality control for high-throughput sequencing data. Bioinformatics. 2016; 32 (2): 292–94. PubMed Abstract | Publisher Full Text | Free Full Text 12. Ewels P, Magnusson M, Lundin S, et al. : MultiQC: summarize analysis results for multiple tools and samples in a single report. Bioinformatics. 2016; 32 (19): 3047–3048. PubMed Abstract | Publisher Full Text | Free Full Text 13. Soneson C, Love MI, Robinson MD: Differential analyses for RNA-seq: transcript-level estimates improve gene-level inferences [version 2; referees: 2 approved]. F1000Res. 2016; 4 : 1521. PubMed Abstract | Publisher Full Text | Free Full Text 14. Robert C, Watson M: Errors in RNA-Seq quantification affect genes of relevance to human disease. Genome Biol. 2015; 16 : 177. PubMed Abstract | Publisher Full Text | Free Full Text 15. Morgan M, Obenchain V, Hester J, et al. : SummarizedExperiment: SummarizedExperiment container. 2017. Reference Source 16. Love MI, Huber W, Anders S: Moderated estimation of fold change and dispersion for RNA-seq data with DESeq2. Genome Biol. 2014; 15 (12): 550. PubMed Abstract | Publisher Full Text | Free Full Text 17. Craciun FL, Bijol V, Ajay AK, et al. : RNA Sequencing Identifies Novel Translational Biomarkers of Kidney Fibrosis. J Am Soc Nephrol. 2016; 27 (6): 1702–1713. PubMed Abstract | Publisher Full Text | Free Full Text 18. Li P, Piao Y, Shon HS, et al. : Comparing the normalization methods for the differential analysis of illumina high-throughput RNA-Seq data. BMC Bioinformatics. 2015; 16 : 347. PubMed Abstract | Publisher Full Text | Free Full Text 19. Robinson MD, Oshlack A: A scaling normalization method for differential expression analysis of RNA-seq data. Genome Biol. 2010; 11 (3): R25. PubMed Abstract | Publisher Full Text | Free Full Text 20. Huber W, von Heydebreck A, Sültmann H, et al. : Variance stabilization applied to microarray data calibration and to the quantification of differential expression. Bioinformatics. 2002; 18 Suppl 1 : S96–104. PubMed Abstract | Publisher Full Text 21. Kolde R: pheatmap: Pretty Heatmaps. 2015. Reference Source 22. Jolliffe I: Principal component analysis. Wiley Online Library, 2002. Publisher Full Text 23. Pantano L: DEGreport: Report of DEG analysis. 2017. Publisher Full Text 24. Daily K, Ho Sui SJ, Schriml LM, et al. : Molecular, phenotypic, and sample-associated data to describe pluripotent stem cell lines and derivatives. Sci Data. 2017; 4 : 170030. PubMed Abstract | Publisher Full Text | Free Full Text 25. Benjamini Y, Hochberg Y: Controlling the False Discovery Rate: A Practical and Powerful Approach to Multiple Testing. J R Stat Soc Series B Stat Methodol. 1995; 57 (1): 289–300. Reference Source 26. Cui X, Churchill GA: Statistical tests for differential expression in cDNA microarray experiments. Genome Biol. 2003; 4 (4): 210. PubMed Abstract | Publisher Full Text | Free Full Text 27. Ward JH Jr: Hierarchical grouping to optimize an objective function. J Am Stat Assoc. 1963; 58 (301): 236–244. Publisher Full Text 28. Dudoit S, Yang YH, Callow MJ, et al. : Statistical methods for identifying differentially expressed genes in replicated cDNA microarray experiments. Stat Sin. 2002; 12 (1): 111–139. Reference Source 29. Yu G, Wang LG, Han Y, et al. : clusterprofiler: an R package for comparing biological themes among gene clusters. OMICS. 2012; 16 (5): 284–287. PubMed Abstract | Publisher Full Text | Free Full Text 30. Allaire JJ, Cheng J, Xie Y, et al. : rmarkdown: Dynamic Documents for R. 2017. Reference Source 31. RStudio Team: RStudio: Integrated Development Environment for R. RStudio, Inc., Boston, MA. 2016. Reference Source 32. Steinbaugh M, Pantano L, Kirchner R, et al. : hbc/bcbioRNASeq: bcbioRNASeq 0.2.4 (Version v0.2.4). Zenodo. 2018. Data Source Comments on this article Comments (0) Version 2 VERSION 2 PUBLISHED 08 Nov 2017 ADD YOUR COMMENT Comment Author details Author details 1 Harvard T.H. Chan School of Public Health, Boston, MA, 02115, USA 2 University of Melbourne Center for Cancer Research, Melbourne, VIC, 3000, Australia Michael J. Steinbaugh Roles: Conceptualization, Methodology, Software, Validation, Writing – Original Draft Preparation, Writing – Review & Editing Lorena Pantano Roles: Conceptualization, Methodology, Software, Validation, Visualization, Writing – Original Draft Preparation, Writing – Review & Editing Rory D. Kirchner Roles: Conceptualization, Software, Writing – Review & Editing Victor Barrera Roles: Software, Writing – Review & Editing Brad A. Chapman Roles: Writing – Original Draft Preparation, Writing – Review & Editing Mary E. Piper Roles: Software, Writing – Original Draft Preparation, Writing – Review & Editing Meeta Mistry Roles: Writing – Review & Editing Radhika S. Khetani Roles: Writing – Original Draft Preparation, Writing – Review & Editing Kayleigh D. Rutherford Roles: Validation, Writing – Review & Editing Oliver Hofmann Roles: Funding Acquisition, Writing – Review & Editing John N. Hutchinson Roles: Conceptualization, Funding Acquisition, Software, Writing – Original Draft Preparation, Writing – Review & Editing Shannan Ho Sui Roles: Funding Acquisition, Supervision, Writing – Review & Editing Competing interests No competing interests were disclosed. Grant information The author(s) declared that no grants were involved in supporting this work. Article Versions (2) version 2 Revised Published: 20 Jun 2018, 6:1976 https://doi.org/10.12688/f1000research.12093.2 version 1 Published: 08 Nov 2017, 6:1976 https://doi.org/10.12688/f1000research.12093.1 Copyright © 2018 Steinbaugh MJ et al . This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. Download Export To Sciwheel Bibtex EndNote ProCite Ref. Manager (RIS) Sente metrics Views Downloads F1000Research - - PubMed Central info_outline Data from PMC are received and updated monthly. - - Citations open_in_new 0 open_in_new 0 open_in_new SEE MORE DETAILS CITE how to cite this article Steinbaugh MJ, Pantano L, Kirchner RD et al. bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.12688/f1000research.12093.2 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS track receive updates on this article Track an article to receive email alerts on any updates to this article. TRACK THIS ARTICLE Share Open Peer Review Current Reviewer Status: ? Key to Reviewer Statuses VIEW HIDE Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Version 2 VERSION 2 PUBLISHED 20 Jun 2018 Revised Views 0 Cite How to cite this report: Risso D. Reviewer Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.16697.r35236 ) The direct URL for this report is: https://f1000research.com/articles/6-1976/v2#referee-response-35236 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 13 Jul 2018 Davide Risso , Division of Biostatistics and Epidemiology, Department of Healthcare Policy and Research, Weill Cornell Medicine, New York, NY, USA Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.16697.r35236 The updated manuscript is well written and describe a very useful R package for the quality control, exploration and statistical analysis of RNA-seq data. The changes in the new version make this a better resource, by adding a useful ... Continue reading READ ALL The updated manuscript is well written and describe a very useful R package for the quality control, exploration and statistical analysis of RNA-seq data. The changes in the new version make this a better resource, by adding a useful section on Functional Analysis and by fully leveraging the SummarizedExperiment and GRanges Bioconductor infrastructures. However, the following (mostly minor) issues are still outstanding and the authors need to address them for me to approve without reservations: 1. As pointed out in my previous review and by the other reviewer, the authors refer to bcbioRNASeq as a Bioconductor package, when it is not part of the Bioconductor project. I again suggest that the authors submit their package to Bioconductor. Until then, this should be referred to as a R package. 2. Although the authors addressed my comment on the description of the plots, some figures remain a bit opaque. For instance, what is the interpretation of Figure 2F? Similarly, the range of the x-axis in Figure 3A makes the plot hard to interpret. 3. Figure 8 is not rendered correctly in the PDF and HTML versions of the article. 4. There are some inconsistencies between the code in the article and the code in the Github repository. For instance in Figure 13 the code in the text says "vst", while in the repository it uses "rlog". It is unclear which is shown in the figure. In addition, when describing plotPCACovariates, in the text the authors say that the used FDR<0.05 but in the code they use 0.1. Please double check all the code. 5. The function `enrichMap` is deprecated. The authors should use the function `emapplot`. In addition, it should be made clear that these functions are part of the DOSE package. Competing Interests: No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Risso D. Reviewer Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.16697.r35236 ) The direct URL for this report is: https://f1000research.com/articles/6-1976/v2#referee-response-35236 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Author Response 16 Jul 2018 Lorena Pantano Rubino , Harvard T.H. Chan School of Public Health, Boston, 02115, USA 16 Jul 2018 Author Response Thanks for the review statement. We appreciate the time to review our work and the comments that will make this article more reproducible and accurate. I’ll leave the first author ... Continue reading Thanks for the review statement. We appreciate the time to review our work and the comments that will make this article more reproducible and accurate. I’ll leave the first author to lead this round of reviewing to address all the comments you shared with us. Thanks for the review statement. We appreciate the time to review our work and the comments that will make this article more reproducible and accurate. I’ll leave the first author to lead this round of reviewing to address all the comments you shared with us. Competing Interests: No competing interests were disclosed. Close Report a concern Respond or Comment COMMENTS ON THIS REPORT Author Response 16 Jul 2018 Lorena Pantano Rubino , Harvard T.H. Chan School of Public Health, Boston, 02115, USA 16 Jul 2018 Author Response Thanks for the review statement. We appreciate the time to review our work and the comments that will make this article more reproducible and accurate. I’ll leave the first author ... Continue reading Thanks for the review statement. We appreciate the time to review our work and the comments that will make this article more reproducible and accurate. I’ll leave the first author to lead this round of reviewing to address all the comments you shared with us. Thanks for the review statement. We appreciate the time to review our work and the comments that will make this article more reproducible and accurate. I’ll leave the first author to lead this round of reviewing to address all the comments you shared with us. Competing Interests: No competing interests were disclosed. Close Report a concern COMMENT ON THIS REPORT Views 0 Cite How to cite this report: Soneson C. Reviewer Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.16697.r35237 ) The direct URL for this report is: https://f1000research.com/articles/6-1976/v2#referee-response-35237 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 29 Jun 2018 Charlotte Soneson , Institute of Molecular Life Sciences, University of Zurich (UZH), Zürich, Switzerland Approved VIEWS 0 https://doi.org/10.5256/f1000research.16697.r35237 The second version of this manuscript has been extended and now contains additional descriptions of the object class used by the bcbioRNASeq package as well as the functional analysis aspects. The text is well written and easy to follow, and ... Continue reading READ ALL The second version of this manuscript has been extended and now contains additional descriptions of the object class used by the bcbioRNASeq package as well as the functional analysis aspects. The text is well written and easy to follow, and the package is likely to be a welcome addition to the toolbox for users working with the bcbio pipeline to process their RNA-seq data. However, I think there are still aspects that can be improved, in particular with respect to the reproducibility of the provided results and figures. The package is still referred to as a "Bioconductor package" in the Introduction, but as far as I can see, it is not submitted to Bioconductor. If the authors don't intend to submit it there, I would recommend describing it as an "R package" instead, as was previously pointed out. Judging from the session info provided in the GitHub repository, the code in the workflow was run with an old version of R (3.4.1) and Bioconductor. One of the used functions (enrichMap, from the DOSE package) is deprecated in the current Bioconductor release, and may be removed in the near future. Moreover, running the commands in the manuscript (or the ones provided in the workflow.R script in the GitHub repository) do not always generate the same figures as are being shown in the article (and in addition, the code in the manuscript is different from the one in the workflow.R script). The differences are mostly minor, but this still reduces the reproducibility of the workflow. Minor points: The "counts" assay is described as containing "raw counts, generated by salmon and imported with tximport". However, if I understand correctly this slot may also contain other types of "count-like" abundances generated by tximport (depending on the value used for countsFromAbundance). The legend of Figure 2B says that all samples "are well within recommended ranges, having [...] almost 90% of reads mapping". However, there are several samples with mapping rate below 70% in the figure. The legend of Figure 3 says that sample classes (defined by the interestingGroups argument) are indicated in different colors, but in the figure, each sample has a different color. In the "Genomic context" part of the QC section, the text suggests that the results of plotExonicMappingRate() and plotIntronicMappingRate() show the exonic and intronic mapping reads as a fraction of the total number of reads ("Ideally, at least 60% of total reads should map to exons"). However, I believe the plots show the fraction of the mapped reads that map to exons and introns, respectively. This could be clarified. In the "Number of genes detected" section, Figure 2D-E should be Figure 2E-F. In the "Correlation of covariates with PCs, the text says "PCs with FDR < 0.05" while the code uses fdr = 0.1. The code in the "Specific genes plot" requires a "library(DEGreport)" to run. The third code chunk in the "Functional analysis" section doesn't run (missing a comma after ont = "BP"), the fourth code chunk requires a "library(tibble)" to run, and the sixth code chunk similarly requires a "library(DOSE)". In the legends of Figures 14-16: P-adjusted values -> adjusted P-values. Competing Interests: No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Soneson C. Reviewer Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.16697.r35237 ) The direct URL for this report is: https://f1000research.com/articles/6-1976/v2#referee-response-35237 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Author Response 03 Jul 2018 Lorena Pantano Rubino , Harvard T.H. Chan School of Public Health, Boston, 02115, USA 03 Jul 2018 Author Response Thanks for the excellent work at reviewing our work. We'll work on fix all minus issues and decide a time for submitting a Bioconductor. We'll report back here with the ... Continue reading Thanks for the excellent work at reviewing our work. We'll work on fix all minus issues and decide a time for submitting a Bioconductor. We'll report back here with the comments. Thanks again for all your time. Thanks for the excellent work at reviewing our work. We'll work on fix all minus issues and decide a time for submitting a Bioconductor. We'll report back here with the comments. Thanks again for all your time. Competing Interests: No competing interests were disclosed. Close Report a concern Respond or Comment COMMENTS ON THIS REPORT Author Response 03 Jul 2018 Lorena Pantano Rubino , Harvard T.H. Chan School of Public Health, Boston, 02115, USA 03 Jul 2018 Author Response Thanks for the excellent work at reviewing our work. We'll work on fix all minus issues and decide a time for submitting a Bioconductor. We'll report back here with the ... Continue reading Thanks for the excellent work at reviewing our work. We'll work on fix all minus issues and decide a time for submitting a Bioconductor. We'll report back here with the comments. Thanks again for all your time. Thanks for the excellent work at reviewing our work. We'll work on fix all minus issues and decide a time for submitting a Bioconductor. We'll report back here with the comments. Thanks again for all your time. Competing Interests: No competing interests were disclosed. Close Report a concern COMMENT ON THIS REPORT Version 1 VERSION 1 PUBLISHED 08 Nov 2017 Views 0 Cite How to cite this report: Risso D. Reviewer Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.13084.r27759 ) The direct URL for this report is: https://f1000research.com/articles/6-1976/v1#referee-response-27759 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 12 Dec 2017 Davide Risso , Division of Biostatistics and Epidemiology, Department of Healthcare Policy and Research, Weill Cornell Medicine, New York, NY, USA Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.13084.r27759 The authors present bcbioRNASeq, an R package for the QC and differential expression analysis of RNA-seq data. The package takes as input the output of the bcbio software, not presented in this work. bcbio is a community ... Continue reading READ ALL The authors present bcbioRNASeq, an R package for the QC and differential expression analysis of RNA-seq data. The package takes as input the output of the bcbio software, not presented in this work. bcbio is a community driven resource that handles the data processing of several high-throughput sequencing applications, ranging from variant calling to ChIP-seq and RNA-seq analyses. The bcbioRNASeq package focuses on RNA-seq. Overall, I enjoyed the article and I think it represents a nice resource for practitioners looking to perform standardized RNA-seq analyses of several datasets and for bcbio users that want to perform high-level statistical analyses after RNA-seq preprocessing. The article is well written and provides a reproducible example that walks the reader through the proposed pipelines. I am glad to report that (except for some minor issues reported below) I was able to fully reproduce the analysis. The following points will hopefully help the authors improve the manuscript. Currently, the package does not appear to be available through Bioconductor, but only through the authors' Github. Are the authors planning to submit it to Bioconductor? I definitely encourage them to do so, as a way to manage package versions and dependencies (see next point). If not, I would ask the authors to refer to bcbioRNASeq as an R package rather than a Bioconductor package. The authors do not specify the versions of the packages needed for their workflow to work. Although the DESCRIPTION file of the package provides such information, adding it to the manuscript would allow readers to reproduce the workflow example. This would be automatically taken care of if the package was part of a Bioconductor release. In addition, at the beginning of the paper, the authors load the packages "DESeq2" and "DEGreport". Aren't these packages in the Import: field of the DESCRIPTION file of bcbioRNASeq? S4 object. Is it really needed to store all the normalized data in the S4 object? This could lead to a huge object when the analysis is run on hundreds of samples. Since scaling normalization is very fast wouldn’t it be better to compute normalized data on the fly and only store the raw and tpm data computed by tximport and featureCounts? On a related note, wouldn’t it be better for the authors to store the featureCounts data in an additional element of the assays() slot and provide coercion methods from their object to the DESeqDataSet and tximport objects? Is it really needed to store both tximport results in the assays slot and in the bcbio slot? Overall, I have the feeling that the object is needlessly big and this could lead to a big memory footprint. Are the plots based on raw or normalized data? If the latter, which normalization / transformation is used by default? How does the user change it? Interpretation of the plots. Although the authors describe the plots in generic terms, it would be useful to explain them more specifically referring to the actual example analysis. For instance, which of the three transformation of Figure 3 is best for the example data? Is Figure 1F a typical pattern or does it uncover something unusual with the data? Same for Figure 7. A better metric to highlight the difference in distribution among samples (Figure 2) is the Relative Log Expression (RLE) plot. The authors might want to include such plot to their already excellent array of QC plots. What to do if the data fail the QC step? The authors present all their QC plots but then move on by simply stating "Once the QC is complete and the dataset looks good [...]". What if the data do not look good? It would be good to advice on what to do (just as a discussion perhaps). E.g., there could be outlying samples to be removed or batch effects to be accounted for in the model. Minor issues: The last sentence of the first paragraph of the Introduction seems to indicate that the R package actually runs the QC tools, while these are run in bcbio and the results are loaded in the R package for exploration. First chunk of R code (page 4 of the PDF version of the paper): at my first read I was wondering how to get the data to run this command. It should be made clearer that this is only meant to show the syntax and is not part of the runnable example. The link to the bcb.rda file is broken. `normalized <- counts(dds, "normalized") ` This line doesn’t work. Did the author mean normalized=TRUE? `tpm <- tpm(txi) ` This line doesn’t work. Did the author mean `tpm <- counts(dds, normalized=“tpm”)`? Please describe writeCounts(). Please consider removing the "+" in the R chunks (e.g., at page 13 of the PDF) so that readers could run the code by copying and pasting into an R session. In statistics, ICA is often used to refer to Independent Component Analysis, so the authors may want to avoid this acronym for the correlation analysis to avoid confusion. In my RStudio session the plotPCACovariates() plot did not work (I couldn't see any points but just a gray background). YAML parameters for the Functional Analysis Rmarkdown: I believe that the line "res" should be "resFile". Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Yes Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Yes Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes Competing Interests: No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Risso D. Reviewer Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.13084.r27759 ) The direct URL for this report is: https://f1000research.com/articles/6-1976/v1#referee-response-27759 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Views 0 Cite How to cite this report: Soneson C. Reviewer Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.13084.r27761 ) The direct URL for this report is: https://f1000research.com/articles/6-1976/v1#referee-response-27761 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 20 Nov 2017 Charlotte Soneson , Institute of Molecular Life Sciences, University of Zurich (UZH), Zürich, Switzerland Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.13084.r27761 This article describes bcbioRNASeq, an R package for analysis of RNA-seq data for which gene expression estimation and quality assessment have been done via the bcbio pipeline. The package contains functions for generating plots and performing downstream analysis as well ... Continue reading READ ALL This article describes bcbioRNASeq, an R package for analysis of RNA-seq data for which gene expression estimation and quality assessment have been done via the bcbio pipeline. The package contains functions for generating plots and performing downstream analysis as well as Rmarkdown templates for generating stand-alone reports of the quality control, differential expression analysis and functional analysis steps. In general, I find the article well written and easy to follow. The included functionality covers the most important parts of a typical RNA-seq analysis, and the package is likely to be a useful tool for bcbio users. Also, the provided templates are easy to extend with additional analyses if necessary. Following are some suggestions for improvement of specific parts of the article. 1. It would be good to indicate any dependencies on particular versions of R/Bioconductor/specific packages, and preferably also give the session info for the session with which the manuscript was generated. I ran the code with R v3.4.2 (Bioconductor v3.6, DESeq2 v1.18.1, bcbioRNASeq v0.1.2), and while most of the code executes correctly, there are some lines that do not. In particular: > normalized tpm writeCounts()(raw, normalized, rlog, tpm) ## formatting error > resTbl plotVolcano(res) > resTbl plotPCACovariates(bcb, fdr = 0.1) does not generate Figure 7 (the asterisks are missing). > alphaSummary(dds, contrast = c(factor = "group", numerator = "day7", denominator = "normal")) does not generate the numbers in Table 1. > resPatterns topTables(resTbl, n = 5) does not generate the numbers in Table 2. 2. If I read correctly, the "tximport" slot of the bcbio object contains length-scaled TPMs, not aggregated transcript counts from tximport. This should be made clearer in the article. Is there an explicit choice in the bcbio pipeline that determines the type of count-scale abundances that are generated? 3. It would be useful to indicate in the beginning of the article where the metadata is stored in the output from bcbio. I.e., where should one look for the values available to supply to the "interestingGroups" argument of loadRNASeq()? 4. In the object description, it would be worth explaining a bit more clearly how the values contained in the slot "-tmm: trimmed mean of M-values, calculated by edgeR" were calculated. 5. In the object description, devtools::sessionInfo() should be devtools::session_info() 6. In the Use case, "Also accessible with tpm()" should presumably be under point 3. 7. Regarding the visual thresholds in the plots (warning thresholds and optimal values), how are the default values determined? Are they fixed, or do they depend on some characteristics of the data? And are they particularly suitable for data generated under specific conditions, in specific organisms or with particular protocols? I am also wondering whether the use of the word "optimal" to designate one threshold may cause confusion. For example, if the "optimal" total number of reads is ~20M, and the "optimal" mapping rate is ~90%, it may not be immediately clear how one should interpret values exceeding (and potentially far away from) these "optimal" values. Finally, is there a reason for only having one line in the exonic and intronic mapping rate plots, but both lines in the other plots? 8. In the "Model fitting" section, it is suggested that it is important to evaluate the variance stabilizing performance of different transformations before the differential expression analysis. However, the transformed data are never actually used for the DE analysis (which is performed with DESeq2). Thus, it should be clarified how the results obtained here are used to inform the downstream analysis. 9. For the QC and differential expression analysis, the article outlines the analysis steps in detail. However, the functional analysis is only described through the existence of an Rmarkdown template. It would be nice to have at least part of the functional analysis also explained and written out in the article. 10. In resPatterns[["plot"]], the x-axis labels are not centered under the respective boxplots. 11. The package is referred to as a "Bioconductor package", but as far as I can see it is not (yet) in Bioconductor. 12. It seems that zero counts are excluded from the plots in Figure 2. This could be clarified in the text. 13. In some places, "library" is used in the place of "package". 14. For the degPatterns() call, it is indicated that it is "CPU intensive". It might be useful to indicate approximately *how* CPU/time intensive, since all other steps in the workflow execute quickly. 15. In the code blocks, it would be easier if non-code characters like > and + were removed, so that the code could be directly copied into an R session. Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Partly Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Partly Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes Competing Interests: No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Soneson C. Reviewer Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.13084.r27761 ) The direct URL for this report is: https://f1000research.com/articles/6-1976/v1#referee-response-27761 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Comments on this article Comments (0) Version 2 VERSION 2 PUBLISHED 08 Nov 2017 ADD YOUR COMMENT Comment keyboard_arrow_left keyboard_arrow_right Open Peer Review Reviewer Status info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Reviewer Reports Invited Reviewers 1 2 Version 2 (revision) 20 Jun 18 read read Version 1 08 Nov 17 read read Charlotte Soneson , University of Zurich (UZH), Zürich, Switzerland Davide Risso , Weill Cornell Medicine, New York, USA Comments on this article All Comments (0) Add a comment Sign up for content alerts Sign Up You are now signed up to receive this alert Browse by related subjects keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2018 Risso D. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 13 Jul 2018 | for Version 2 Davide Risso , Division of Biostatistics and Epidemiology, Department of Healthcare Policy and Research, Weill Cornell Medicine, New York, NY, USA 0 Views copyright © 2018 Risso D. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (1) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions The updated manuscript is well written and describe a very useful R package for the quality control, exploration and statistical analysis of RNA-seq data. The changes in the new version make this a better resource, by adding a useful section on Functional Analysis and by fully leveraging the SummarizedExperiment and GRanges Bioconductor infrastructures. However, the following (mostly minor) issues are still outstanding and the authors need to address them for me to approve without reservations: 1. As pointed out in my previous review and by the other reviewer, the authors refer to bcbioRNASeq as a Bioconductor package, when it is not part of the Bioconductor project. I again suggest that the authors submit their package to Bioconductor. Until then, this should be referred to as a R package. 2. Although the authors addressed my comment on the description of the plots, some figures remain a bit opaque. For instance, what is the interpretation of Figure 2F? Similarly, the range of the x-axis in Figure 3A makes the plot hard to interpret. 3. Figure 8 is not rendered correctly in the PDF and HTML versions of the article. 4. There are some inconsistencies between the code in the article and the code in the Github repository. For instance in Figure 13 the code in the text says "vst", while in the repository it uses "rlog". It is unclear which is shown in the figure. In addition, when describing plotPCACovariates, in the text the authors say that the used FDR<0.05 but in the code they use 0.1. Please double check all the code. 5. The function `enrichMap` is deprecated. The authors should use the function `emapplot`. In addition, it should be made clear that these functions are part of the DOSE package. Competing Interests No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (1) Author Response 16 Jul 2018 Lorena Pantano Rubino, Harvard T.H. Chan School of Public Health, Boston, 02115, USA Thanks for the review statement. We appreciate the time to review our work and the comments that will make this article more reproducible and accurate. I’ll leave the first author to lead this round of reviewing to address all the comments you shared with us. View more View less Competing Interests No competing interests were disclosed. reply Respond Report a concern Risso D. Peer Review Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.16697.r35236) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/6-1976/v2#referee-response-35236 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2018 Soneson C. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 29 Jun 2018 | for Version 2 Charlotte Soneson , Institute of Molecular Life Sciences, University of Zurich (UZH), Zürich, Switzerland 0 Views copyright © 2018 Soneson C. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (1) Approved info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions The second version of this manuscript has been extended and now contains additional descriptions of the object class used by the bcbioRNASeq package as well as the functional analysis aspects. The text is well written and easy to follow, and the package is likely to be a welcome addition to the toolbox for users working with the bcbio pipeline to process their RNA-seq data. However, I think there are still aspects that can be improved, in particular with respect to the reproducibility of the provided results and figures. The package is still referred to as a "Bioconductor package" in the Introduction, but as far as I can see, it is not submitted to Bioconductor. If the authors don't intend to submit it there, I would recommend describing it as an "R package" instead, as was previously pointed out. Judging from the session info provided in the GitHub repository, the code in the workflow was run with an old version of R (3.4.1) and Bioconductor. One of the used functions (enrichMap, from the DOSE package) is deprecated in the current Bioconductor release, and may be removed in the near future. Moreover, running the commands in the manuscript (or the ones provided in the workflow.R script in the GitHub repository) do not always generate the same figures as are being shown in the article (and in addition, the code in the manuscript is different from the one in the workflow.R script). The differences are mostly minor, but this still reduces the reproducibility of the workflow. Minor points: The "counts" assay is described as containing "raw counts, generated by salmon and imported with tximport". However, if I understand correctly this slot may also contain other types of "count-like" abundances generated by tximport (depending on the value used for countsFromAbundance). The legend of Figure 2B says that all samples "are well within recommended ranges, having [...] almost 90% of reads mapping". However, there are several samples with mapping rate below 70% in the figure. The legend of Figure 3 says that sample classes (defined by the interestingGroups argument) are indicated in different colors, but in the figure, each sample has a different color. In the "Genomic context" part of the QC section, the text suggests that the results of plotExonicMappingRate() and plotIntronicMappingRate() show the exonic and intronic mapping reads as a fraction of the total number of reads ("Ideally, at least 60% of total reads should map to exons"). However, I believe the plots show the fraction of the mapped reads that map to exons and introns, respectively. This could be clarified. In the "Number of genes detected" section, Figure 2D-E should be Figure 2E-F. In the "Correlation of covariates with PCs, the text says "PCs with FDR < 0.05" while the code uses fdr = 0.1. The code in the "Specific genes plot" requires a "library(DEGreport)" to run. The third code chunk in the "Functional analysis" section doesn't run (missing a comma after ont = "BP"), the fourth code chunk requires a "library(tibble)" to run, and the sixth code chunk similarly requires a "library(DOSE)". In the legends of Figures 14-16: P-adjusted values -> adjusted P-values. Competing Interests No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. reply Respond to this report Responses (1) Author Response 03 Jul 2018 Lorena Pantano Rubino, Harvard T.H. Chan School of Public Health, Boston, 02115, USA Thanks for the excellent work at reviewing our work. We'll work on fix all minus issues and decide a time for submitting a Bioconductor. We'll report back here with the comments. Thanks again for all your time. View more View less Competing Interests No competing interests were disclosed. reply Respond Report a concern Soneson C. Peer Review Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.16697.r35237) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/6-1976/v2#referee-response-35237 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2017 Risso D. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 12 Dec 2017 | for Version 1 Davide Risso , Division of Biostatistics and Epidemiology, Department of Healthcare Policy and Research, Weill Cornell Medicine, New York, NY, USA 0 Views copyright © 2017 Risso D. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions The authors present bcbioRNASeq, an R package for the QC and differential expression analysis of RNA-seq data. The package takes as input the output of the bcbio software, not presented in this work. bcbio is a community driven resource that handles the data processing of several high-throughput sequencing applications, ranging from variant calling to ChIP-seq and RNA-seq analyses. The bcbioRNASeq package focuses on RNA-seq. Overall, I enjoyed the article and I think it represents a nice resource for practitioners looking to perform standardized RNA-seq analyses of several datasets and for bcbio users that want to perform high-level statistical analyses after RNA-seq preprocessing. The article is well written and provides a reproducible example that walks the reader through the proposed pipelines. I am glad to report that (except for some minor issues reported below) I was able to fully reproduce the analysis. The following points will hopefully help the authors improve the manuscript. Currently, the package does not appear to be available through Bioconductor, but only through the authors' Github. Are the authors planning to submit it to Bioconductor? I definitely encourage them to do so, as a way to manage package versions and dependencies (see next point). If not, I would ask the authors to refer to bcbioRNASeq as an R package rather than a Bioconductor package. The authors do not specify the versions of the packages needed for their workflow to work. Although the DESCRIPTION file of the package provides such information, adding it to the manuscript would allow readers to reproduce the workflow example. This would be automatically taken care of if the package was part of a Bioconductor release. In addition, at the beginning of the paper, the authors load the packages "DESeq2" and "DEGreport". Aren't these packages in the Import: field of the DESCRIPTION file of bcbioRNASeq? S4 object. Is it really needed to store all the normalized data in the S4 object? This could lead to a huge object when the analysis is run on hundreds of samples. Since scaling normalization is very fast wouldn’t it be better to compute normalized data on the fly and only store the raw and tpm data computed by tximport and featureCounts? On a related note, wouldn’t it be better for the authors to store the featureCounts data in an additional element of the assays() slot and provide coercion methods from their object to the DESeqDataSet and tximport objects? Is it really needed to store both tximport results in the assays slot and in the bcbio slot? Overall, I have the feeling that the object is needlessly big and this could lead to a big memory footprint. Are the plots based on raw or normalized data? If the latter, which normalization / transformation is used by default? How does the user change it? Interpretation of the plots. Although the authors describe the plots in generic terms, it would be useful to explain them more specifically referring to the actual example analysis. For instance, which of the three transformation of Figure 3 is best for the example data? Is Figure 1F a typical pattern or does it uncover something unusual with the data? Same for Figure 7. A better metric to highlight the difference in distribution among samples (Figure 2) is the Relative Log Expression (RLE) plot. The authors might want to include such plot to their already excellent array of QC plots. What to do if the data fail the QC step? The authors present all their QC plots but then move on by simply stating "Once the QC is complete and the dataset looks good [...]". What if the data do not look good? It would be good to advice on what to do (just as a discussion perhaps). E.g., there could be outlying samples to be removed or batch effects to be accounted for in the model. Minor issues: The last sentence of the first paragraph of the Introduction seems to indicate that the R package actually runs the QC tools, while these are run in bcbio and the results are loaded in the R package for exploration. First chunk of R code (page 4 of the PDF version of the paper): at my first read I was wondering how to get the data to run this command. It should be made clearer that this is only meant to show the syntax and is not part of the runnable example. The link to the bcb.rda file is broken. `normalized <- counts(dds, "normalized") ` This line doesn’t work. Did the author mean normalized=TRUE? `tpm <- tpm(txi) ` This line doesn’t work. Did the author mean `tpm <- counts(dds, normalized=“tpm”)`? Please describe writeCounts(). Please consider removing the "+" in the R chunks (e.g., at page 13 of the PDF) so that readers could run the code by copying and pasting into an R session. In statistics, ICA is often used to refer to Independent Component Analysis, so the authors may want to avoid this acronym for the correlation analysis to avoid confusion. In my RStudio session the plotPCACovariates() plot did not work (I couldn't see any points but just a gray background). YAML parameters for the Functional Analysis Rmarkdown: I believe that the line "res" should be "resFile". Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Yes Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Yes Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes Competing Interests No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (0) Risso D. Peer Review Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.13084.r27759) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/6-1976/v1#referee-response-27759 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2017 Soneson C. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 20 Nov 2017 | for Version 1 Charlotte Soneson , Institute of Molecular Life Sciences, University of Zurich (UZH), Zürich, Switzerland 0 Views copyright © 2017 Soneson C. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions This article describes bcbioRNASeq, an R package for analysis of RNA-seq data for which gene expression estimation and quality assessment have been done via the bcbio pipeline. The package contains functions for generating plots and performing downstream analysis as well as Rmarkdown templates for generating stand-alone reports of the quality control, differential expression analysis and functional analysis steps. In general, I find the article well written and easy to follow. The included functionality covers the most important parts of a typical RNA-seq analysis, and the package is likely to be a useful tool for bcbio users. Also, the provided templates are easy to extend with additional analyses if necessary. Following are some suggestions for improvement of specific parts of the article. 1. It would be good to indicate any dependencies on particular versions of R/Bioconductor/specific packages, and preferably also give the session info for the session with which the manuscript was generated. I ran the code with R v3.4.2 (Bioconductor v3.6, DESeq2 v1.18.1, bcbioRNASeq v0.1.2), and while most of the code executes correctly, there are some lines that do not. In particular: > normalized tpm writeCounts()(raw, normalized, rlog, tpm) ## formatting error > resTbl plotVolcano(res) > resTbl plotPCACovariates(bcb, fdr = 0.1) does not generate Figure 7 (the asterisks are missing). > alphaSummary(dds, contrast = c(factor = "group", numerator = "day7", denominator = "normal")) does not generate the numbers in Table 1. > resPatterns topTables(resTbl, n = 5) does not generate the numbers in Table 2. 2. If I read correctly, the "tximport" slot of the bcbio object contains length-scaled TPMs, not aggregated transcript counts from tximport. This should be made clearer in the article. Is there an explicit choice in the bcbio pipeline that determines the type of count-scale abundances that are generated? 3. It would be useful to indicate in the beginning of the article where the metadata is stored in the output from bcbio. I.e., where should one look for the values available to supply to the "interestingGroups" argument of loadRNASeq()? 4. In the object description, it would be worth explaining a bit more clearly how the values contained in the slot "-tmm: trimmed mean of M-values, calculated by edgeR" were calculated. 5. In the object description, devtools::sessionInfo() should be devtools::session_info() 6. In the Use case, "Also accessible with tpm()" should presumably be under point 3. 7. Regarding the visual thresholds in the plots (warning thresholds and optimal values), how are the default values determined? Are they fixed, or do they depend on some characteristics of the data? And are they particularly suitable for data generated under specific conditions, in specific organisms or with particular protocols? I am also wondering whether the use of the word "optimal" to designate one threshold may cause confusion. For example, if the "optimal" total number of reads is ~20M, and the "optimal" mapping rate is ~90%, it may not be immediately clear how one should interpret values exceeding (and potentially far away from) these "optimal" values. Finally, is there a reason for only having one line in the exonic and intronic mapping rate plots, but both lines in the other plots? 8. In the "Model fitting" section, it is suggested that it is important to evaluate the variance stabilizing performance of different transformations before the differential expression analysis. However, the transformed data are never actually used for the DE analysis (which is performed with DESeq2). Thus, it should be clarified how the results obtained here are used to inform the downstream analysis. 9. For the QC and differential expression analysis, the article outlines the analysis steps in detail. However, the functional analysis is only described through the existence of an Rmarkdown template. It would be nice to have at least part of the functional analysis also explained and written out in the article. 10. In resPatterns[["plot"]], the x-axis labels are not centered under the respective boxplots. 11. The package is referred to as a "Bioconductor package", but as far as I can see it is not (yet) in Bioconductor. 12. It seems that zero counts are excluded from the plots in Figure 2. This could be clarified in the text. 13. In some places, "library" is used in the place of "package". 14. For the degPatterns() call, it is indicated that it is "CPU intensive". It might be useful to indicate approximately *how* CPU/time intensive, since all other steps in the workflow execute quickly. 15. In the code blocks, it would be easier if non-code characters like > and + were removed, so that the code could be directly copied into an R session. Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Partly Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Partly Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes Competing Interests No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (0) Soneson C. Peer Review Report For: bcbioRNASeq: R package for bcbio RNA-seq analysis [version 2; peer review: 1 approved, 1 approved with reservations] . F1000Research 2018, 6 :1976 ( https://doi.org/10.5256/f1000research.13084.r27761) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/6-1976/v1#referee-response-27761 Alongside their report, reviewers assign a status to the article: Approved - the paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations - A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved - fundamental flaws in the paper seriously undermine the findings and conclusions Adjust parameters to alter display View on desktop for interactive features Includes Interactive Elements View on desktop for interactive features Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Stay Updated Sign up for content alerts and receive a weekly or monthly email with all newly published articles Register with F1000Research Already registered? Sign in Not now, thanks close PLEASE NOTE If you are an AUTHOR of this article, please check that you signed in with the account associated with this article otherwise we cannot automatically identify your role as an author and your comment will be labelled as a “User Comment”. If you are a REVIEWER of this article, please check that you have signed in with the account associated with this article and then go to your account to submit your report, please do not post your review here. If you do not have access to your original account, please contact us . All commenters must hold a formal affiliation as per our Policies . The information that you give us will be displayed next to your comment. User comments must be in English, comprehensible and relevant to the article under discussion. We reserve the right to remove any comments that we consider to be inappropriate, offensive or otherwise in breach of the User Comment Terms and Conditions . Commenters must not use a comment for personal attacks. When criticisms of the article are based on unpublished data, the data should be made available. I accept the User Comment Terms and Conditions Please confirm that you accept the User Comment Terms and Conditions. Affiliation ✕ refresh Please enter your institution. Note: To add your institution or organisation, start typing the name and then select the correct name from the list. Where applicable, the name will appear in both the original language and in English. Do not paste in the name. If the name does not appear in the drop-down list, we will display the information you have entered. ✕ refresh Country/Region * USA UK Canada China France Germany Afghanistan Aland Islands Albania Algeria American Samoa Andorra Angola Anguilla Antarctica Antigua and Barbuda Argentina Armenia Aruba Australia Austria Azerbaijan Bahamas Bahrain Bangladesh Barbados Belarus Belgium Belize Benin Bermuda Bhutan Bolivia Bosnia and Herzegovina Botswana Bouvet Island Brazil British Indian Ocean Territory British Virgin Islands Brunei Bulgaria Burkina Faso Burundi Cambodia Cameroon Canada Cape Verde Cayman Islands Central African Republic Chad Chile China Christmas Island Cocos (Keeling) Islands Colombia Comoros Congo Cook Islands Costa Rica Cote d'Ivoire Croatia Cuba Cyprus Czech Republic Democratic Republic of the Congo Denmark Djibouti Dominica Dominican Republic Ecuador Egypt El Salvador Equatorial Guinea Eritrea Estonia Ethiopia Falkland Islands Faroe Islands Federated States of Micronesia Fiji Finland France French Guiana French Polynesia French Southern Territories Gabon Georgia Germany Ghana Gibraltar Greece Greenland Grenada Guadeloupe Guam Guatemala Guernsey Guinea Guinea-Bissau Guyana Haiti Heard Island and Mcdonald Islands Holy See (Vatican City State) Honduras Hong Kong Hungary Iceland India Indonesia Iran Iraq Ireland Israel Italy Jamaica Japan Jersey Jordan Kazakhstan Kenya Kiribati Kosovo (Serbia and Montenegro) Kuwait Kyrgyzstan Lao People's Democratic Republic Latvia Lebanon Lesotho Liberia Libya Liechtenstein Lithuania Luxembourg Macao Madagascar Malawi Malaysia Maldives Mali Malta Marshall Islands Martinique Mauritania Mauritius Mayotte Mexico Minor Outlying Islands of the United States Moldova Monaco Mongolia Montenegro Montserrat Morocco Mozambique Myanmar Namibia Nauru Nepal Netherlands Antilles New Caledonia New Zealand Nicaragua Niger Nigeria Niue Norfolk Island North Korea North Macedonia Northern Mariana Islands Norway Oman Pakistan Palau Palestinian Territory Panama Papua New Guinea Paraguay Peru Philippines Pitcairn Poland Portugal Puerto Rico Qatar Reunion Romania Russian Federation Rwanda Saint Helena Saint Kitts and Nevis Saint Lucia Saint Pierre and Miquelon Saint Vincent and the Grenadines Samoa San Marino Sao Tome and Principe Saudi Arabia Senegal Serbia Seychelles Sierra Leone Singapore Slovakia Slovenia Solomon Islands Somalia South Africa South Georgia and the South Sandwich Is South Korea South Sudan Spain Sri Lanka Sudan Suriname Svalbard and Jan Mayen Swaziland Sweden Switzerland Syria Taiwan Tajikistan Tanzania Thailand The Gambia The Netherlands Timor-Leste Togo Tokelau Tonga Trinidad and Tobago Tunisia Turkey Turkmenistan Turks and Caicos Islands Tuvalu UK USA Uganda Ukraine United Arab Emirates United States Virgin Islands Uruguay Uzbekistan Vanuatu Venezuela Vietnam Wallis and Futuna West Bank and Gaza Strip Western Sahara Yemen Zambia Zimbabwe Please select your country/region. You must enter a comment. Competing Interests Please disclose any competing interests that might be construed to influence your judgment of the article's or peer review report's validity or importance. Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Please state your competing interests The comment has been saved. An error has occurred. Please try again. Cancel Post var lTitle = "bcbioRNASeq: R package for bcbio RNA-seq...".replace("'", ''); var linkedInUrl = "http://www.linkedin.com/shareArticle?url=https://f1000research.com/articles/6-1976/v2" + "&title=" + encodeURIComponent(lTitle) + "&summary=" + encodeURIComponent('Read the article by '); var deliciousUrl = "https://del.icio.us/post?url=https://f1000research.com/articles/6-1976/v2&title=" + encodeURIComponent(lTitle); var redditUrl = "http://reddit.com/submit?url=https://f1000research.com/articles/6-1976/v2" + "&title=" + encodeURIComponent(lTitle); linkedInUrl += encodeURIComponent('Steinbaugh MJ et al.'); var offsetTop = /chrome/i.test( navigator.userAgent ) ? 4 : -10; var addthis_config = { ui_offset_top: offsetTop, services_compact : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_expanded : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_custom : [ { name: "LinkedIn", url: linkedInUrl, icon:"/img/icon/at_linkedin.svg" }, { name: "Mendeley", url: "http://www.mendeley.com/import/?url=https://f1000research.com/articles/6-1976/v2/mendeley", icon:"/img/icon/at_mendeley.svg" }, { name: "Reddit", url: redditUrl, icon:"/img/icon/at_reddit.svg" }, ] }; var addthis_share = { url: "https://f1000research.com/articles/6-1976", templates : { twitter : "bcbioRNASeq: R package for bcbio RNA-seq analysis. Steinbaugh MJ et al., published by " + "@F1000Research" + ", https://f1000research.com/articles/6-1976/v2" } }; if (typeof(addthis) != "undefined"){ addthis.addEventListener('addthis.ready', checkCount); addthis.addEventListener('addthis.menu.share', checkCount); } $(".f1r-shares-twitter").attr("href", "https://twitter.com/intent/tweet?text=" + addthis_share.templates.twitter); $(".f1r-shares-facebook").attr("href", "https://www.facebook.com/sharer/sharer.php?u=" + addthis_share.url); $(".f1r-shares-linkedin").attr("href", addthis_config.services_custom[0].url); $(".f1r-shares-reddit").attr("href", addthis_config.services_custom[2].url); $(".f1r-shares-mendelay").attr("href", addthis_config.services_custom[1].url); function checkCount(){ setTimeout(function(){ $(".addthis_button_expanded").each(function(){ var count = $(this).text(); if (count !== "" && count != "0") $(this).removeClass("is-hidden"); else $(this).addClass("is-hidden"); }); }, 1000); } close How to cite this report {{reportCitation}} Cancel Copy Citation Details $(function(){R.ui.buttonDropdowns('.dropdown-for-downloads');}); $(function(){R.ui.toolbarDropdowns('.toolbar-dropdown-for-downloads');}); $.get("/articles/acj/12093/16697") new F1000.Clipboard(); new F1000.ThesaurusTermsDisplay("articles", "article", "16697"); $(document).ready(function() { $( "#frame1" ).on('load', function() { var mydiv = $(this).contents().find("div"); var h = mydiv.height(); console.log(h) }); var tooltipLivingFigure = jQuery(".interactive-living-figure-label .icon-more-info"), titleLivingFigure = tooltipLivingFigure.attr("title"); tooltipLivingFigure.simpletip({ fixed: true, position: ["-115", "30"], baseClass: 'small-tooltip', content:titleLivingFigure + " " }); tooltipLivingFigure.removeAttr("title"); $("body").on("click", ".cite-living-figure", function(e) { e.preventDefault(); var ref = $(this).attr("data-ref"); $(this).closest(".living-figure-list-container").find("#" + ref).fadeIn(200); }); $("body").on("click", ".close-cite-living-figure", function(e) { e.preventDefault(); $(this).closest(".popup-window-wrapper").fadeOut(200); }); $(document).on("mouseup", function(e) { var metricsContainer = $(".article-metrics-popover-wrapper"); if (!metricsContainer.is(e.target) && metricsContainer.has(e.target).length === 0) { $(".article-metrics-close-button").click(); } }); var articleId = $('#articleId').val(); if($("#main-article-count-box").attachArticleMetrics) { $("#main-article-count-box").attachArticleMetrics(articleId, { articleMetricsView: true }); } }); var figshareWidget = $(".new_figshare_widget"); if (figshareWidget.length > 0) { window.figshare.load("f1000", function(Widget) { // Select a tag/tags defined in your page. In this tag we will place the widget. _.map(figshareWidget, function(el){ var widget = new Widget({ articleId: $(el).attr("figshare_articleId") //height:300 // this is the height of the viewer part. [Default: 550] }); widget.initialize(); // initialize the widget widget.mount(el); // mount it in a tag that's on your page // this will save the widget on the global scope for later use from // your JS scripts. This line is optional. //window.widget = widget; }); }); } close Error Close Add Reset F1000.MICROSERVICES.AFFILIATION = ''; $(document).ready(function () { $('.js-affiliations-form').each((index, form) => { new AffiliationForm({ formId: form.id, institutionErrorSelector: '.comment-enter-institution', departmentErrorSelector: '.comment-enter-department', placeSelector: '.js-add-comment-place', stateSelector: '.js-add-comment-state', zipCodeSelector: '.js-add-comment-zipcode', countrySelector: '.js-add-comment-country', countryErrorSelector: '.comment-enter-country', }); }); }); $(document).ready(function () { var reportIds = { "27760": 0, "27761": 66, "35236": 38, "35237": 53, "28587": 0, "27758": 0, "27759": 55, }; $(".referee-response-container,.js-referee-report").each(function(index, el) { var reportId = $(el).attr("data-reportid"), reportCount = reportIds[reportId] || 0; $(el).find(".comments-count-container,.js-referee-report-views").html(reportCount); }); var uuidInput = $("#article_uuid"), oldUUId = uuidInput.val(), newUUId = "f9d0e689-2082-4ec6-8e96-c346d62a14c2"; uuidInput.val(newUUId); $("a[href*='article_uuid=']").each(function(index, el) { var newHref = $(el).attr("href").replace(oldUUId, newUUId); $(el).attr("href", newHref); }); }); An innovative open access publishing platform offering rapid publication and open peer review, whilst supporting data deposition and sharing. Browse Gateways Collections How it Works Contact For Developers Cookie Notice Privacy Notice RSS Submit Your Research Follow us © 2012-2026 F1000 Research Ltd. ISSN 2046-1402 | Legal | Partner of Research4Life • CrossRef • ORCID • FAIRSharing R.templateTests.simpleTemplate = R.template(' $text $text $text $text $text '); R.templateTests.runTests(); var F1000platform = new F1000.Platform({ name: "f1000research", displayName: "F1000Research", hostName: "f1000research.com", id: "1", editorialEmail: "[email protected]", infoEmail: "[email protected]", usePmcStats: true }); $(function(){R.ui.dropdowns('.dropdown-for-authors, .dropdown-for-about, .dropdown-for-myresearch');}); // $(function(){R.ui.dropdowns('.dropdown-for-referees');}); $(document).ready(function () { if ($(".cookie-warning").is(":visible")) { $(".sticky").css("margin-bottom", "35px"); $(".devices").addClass("devices-and-cookie-warning"); } $(".cookie-warning .close-button").click(function (e) { $(".devices").removeClass("devices-and-cookie-warning"); $(".sticky").css("margin-bottom", "0"); }); $("#tweeter-feed .tweet-message").each(function (i, message) { var self = $(message); self.html(linkify(self.html())); }); $(".partner").on("mouseenter mouseleave", function() { $(this).find(".gray-scale, .colour").toggleClass("is-hidden"); }); }); Sign In Remember me Forgotten your password? Sign In Cancel Email or password not correct. Please try again Please wait... $(function(){ // Note: All the setup needs to run against a name attribute and *not* the id due the clonish // nature of facebox... $("a[id=googleSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("GOOGLE"); $("form[id=oAuthForm]").submit(); }); $("a[id=facebookSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("FACEBOOK"); $("form[id=oAuthForm]").submit(); }); $("a[id=orcidSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("ORCID"); $("form[id=oAuthForm]").submit(); }); }); If you've forgotten your password, please enter your email address below and we'll send you instructions on how to reset your password. The email address should be the one you originally registered with F1000. Email address not valid, please try again You registered with F1000 via Google, so we cannot reset your password. To sign in, please click here . If you still need help with your Google account password, please click here . You registered with F1000 via Facebook, so we cannot reset your password. To sign in, please click here . If you still need help with your Facebook account password, please click here . Code not correct, please try again Reset password Cancel Email us for further assistance. Server error, please try again. If your email address is registered with us, we will email you instructions to reset your password. If you think you should have received this email but it has not arrived, please check your spam filters and/or contact for further assistance. Please wait... Register $(document).ready(function () { signIn.createSignInAsRow($("#sign-in-form-gfb-popup")); $(".target-field").each(function () { var uris = $(this).val().split("/"); if (uris.pop() === "login") { $(this).val(uris.toString().replace(",","/")); } }); });

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. The paper's references may be in our DB but unresolved to ``paper_id`` (resolution happens at ingest when the cited DOI matches a row we already have). Run the cross-source citation reconcile pass to retry.

Source provenance

europepmc
last seen: 2026-05-19T01:45:01.086888+00:00
unpaywall
last seen: 2026-05-29T02:00:03.542394+00:00
License: CC-BY-4.0