Easy and efficient ensemble gene set testing with EGSEA

preprint OA: closed CC-BY-4.0

Abstract

Gene set enrichment analysis is a popular approach for prioritising the biological processes perturbed in genomic datasets. The Bioconductor project hosts over 80 software packages capable of gene set analysis. Most of these packages search for enriched signatures amongst differentially regulated genes to reveal higher level biological themes that may be missed when focusing only on evidence from individual genes. With so many different methods on offer, choosing the best algorithm and visualization approach can be challenging. The EGSEA package solves this problem by combining results from up to 12 prominent gene set testing algorithms to obtain a consensus ranking of biologically relevant results.This workflow demonstrates how EGSEA can extend limma-based differential expression analyses for RNA-seq and microarray data using experiments that profile 3 distinct cell populations important for studying the origins of breast cancer. Following data normalization and set-up of an appropriate linear model for differential expression analysis, EGSEA builds gene signature specific indexes that link a wide range of mouse or human gene set collections obtained from MSigDB, GeneSetDB and KEGG to the gene expression data being investigated. EGSEA is then configured and the ensemble enrichment analysis run, returning an object that can be queried using several S4 methods for ranking gene sets and visualizing results via heatmaps, KEGG pathway views, GO graphs, scatter plots and bar plots. Finally, an HTML report that combines these displays can fast-track the sharing of results with collaborators, and thus expedite downstream biological validation. EGSEA is simple to use and can be easily integrated with existing gene expression analysis pipelines for both human and mouse data.
Full text 193,781 characters · extracted from preprint-html · click to expand
Easy and efficient ensemble gene set testing with... | F1000Research "use strict";function _typeof(t){return(_typeof="function"==typeof Symbol&&"symbol"==typeof Symbol.iterator?function(t){return typeof t}:function(t){return t&&"function"==typeof Symbol&&t.constructor===Symbol&&t!==Symbol.prototype?"symbol":typeof t})(t)}!function(){var t=function(){var t,e,o=[],n=window,r=n;for(;r;){try{if(r.frames.__tcfapiLocator){t=r;break}}catch(t){}if(r===n.top)break;r=r.parent}t||(!function t(){var e=n.document,o=!!n.frames.__tcfapiLocator;if(!o)if(e.body){var r=e.createElement("iframe");r.style.cssText="display:none",r.name="__tcfapiLocator",e.body.appendChild(r)}else setTimeout(t,5);return!o}(),n.__tcfapi=function(){for(var t=arguments.length,n=new Array(t),r=0;r 3&&2===parseInt(n[1],10)&&"boolean"==typeof n[3]&&(e=n[3],"function"==typeof n[2]&&n[2]("set",!0)):"ping"===n[0]?"function"==typeof n[2]&&n[2]({gdprApplies:e,cmpLoaded:!1,cmpStatus:"stub"}):o.push(n)},n.addEventListener("message",(function(t){var e="string"==typeof t.data,o={};if(e)try{o=JSON.parse(t.data)}catch(t){}else o=t.data;var n="object"===_typeof(o)&&null!==o?o.__tcfapiCall:null;n&&window.__tcfapi(n.command,n.version,(function(o,r){var a={__tcfapiReturn:{returnValue:o,success:r,callId:n.callId}};t&&t.source&&t.source.postMessage&&t.source.postMessage(e?JSON.stringify(a):a,"*")}),n.parameter)}),!1))};"undefined"!=typeof module?module.exports=t:t()}(); dataLayer = dataLayer || []; // Standard GTM initialization - Google Consent Mode handles consent automatically (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start': new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0], j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src= 'https://www.googletagmanager.com/gtm.js?id='+i+dl+ '>m_auth=hzk0Vc3qFsQYhCrIoHz68A>m_preview=env-1>m_cookies_win=x';f.parentNode.insertBefore(j,f); })(window,document,'script','dataLayer','GTM-MWFK8L5J'); ;window.NREUM||(NREUM={});NREUM.init={distributed_tracing:{enabled:true},privacy:{cookies_enabled:true},ajax:{deny_list:["bam.nr-data.net"]}}; ;NREUM.loader_config={accountID:"438030",trustKey:"438030",agentID:"772317073",licenseKey:"97f8f67f26",applicationID:"772317073"} ;NREUM.info={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net",licenseKey:"97f8f67f26",applicationID:"772317073",sa:1} ;/*! For license information please see nr-loader-spa-1.236.0.min.js.LICENSE.txt */ (()=>{"use strict";var e,t,r={5763:(e,t,r)=>{r.d(t,{P_:()=>l,Mt:()=>g,C5:()=>s,DL:()=>v,OP:()=>T,lF:()=>D,Yu:()=>y,Dg:()=>h,CX:()=>c,GE:()=>b,sU:()=>_});var n=r(8632),i=r(9567);const o={beacon:n.ce.beacon,errorBeacon:n.ce.errorBeacon,licenseKey:void 0,applicationID:void 0,sa:void 0,queueTime:void 0,applicationTime:void 0,ttGuid:void 0,user:void 0,account:void 0,product:void 0,extra:void 0,jsAttributes:{},userAttributes:void 0,atts:void 0,transactionName:void 0,tNamePlain:void 0},a={};function s(e){if(!e)throw new Error("All info objects require an agent identifier!");if(!a[e])throw new Error("Info for ".concat(e," was never set"));return a[e]}function c(e,t){if(!e)throw new Error("All info objects require an agent identifier!");a[e]=(0,i.D)(t,o),(0,n.Qy)(e,a[e],"info")}var u=r(7056);const d=()=>{const e={blockSelector:"[data-nr-block]",maskInputOptions:{password:!0}};return{allow_bfcache:!0,privacy:{cookies_enabled:!0},ajax:{deny_list:void 0,enabled:!0,harvestTimeSeconds:10},distributed_tracing:{enabled:void 0,exclude_newrelic_header:void 0,cors_use_newrelic_header:void 0,cors_use_tracecontext_headers:void 0,allowed_origins:void 0},session:{domain:void 0,expiresMs:u.oD,inactiveMs:u.Hb},ssl:void 0,obfuscate:void 0,jserrors:{enabled:!0,harvestTimeSeconds:10},metrics:{enabled:!0},page_action:{enabled:!0,harvestTimeSeconds:30},page_view_event:{enabled:!0},page_view_timing:{enabled:!0,harvestTimeSeconds:30,long_task:!1},session_trace:{enabled:!0,harvestTimeSeconds:10},harvest:{tooManyRequestsDelay:60},session_replay:{enabled:!1,harvestTimeSeconds:60,sampleRate:.1,errorSampleRate:.1,maskTextSelector:"*",maskAllInputs:!0,get blockClass(){return"nr-block"},get ignoreClass(){return"nr-ignore"},get maskTextClass(){return"nr-mask"},get blockSelector(){return e.blockSelector},set blockSelector(t){e.blockSelector+=",".concat(t)},get maskInputOptions(){return e.maskInputOptions},set maskInputOptions(t){e.maskInputOptions={...t,password:!0}}},spa:{enabled:!0,harvestTimeSeconds:10}}},f={};function l(e){if(!e)throw new Error("All configuration objects require an agent identifier!");if(!f[e])throw new Error("Configuration for ".concat(e," was never set"));return f[e]}function h(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");f[e]=(0,i.D)(t,d()),(0,n.Qy)(e,f[e],"config")}function g(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");var r=l(e);if(r){for(var n=t.split("."),i=0;i {r.d(t,{D:()=>i});var n=r(50);function i(e,t){try{if(!e||"object"!=typeof e)return(0,n.Z)("Setting a Configurable requires an object as input");if(!t||"object"!=typeof t)return(0,n.Z)("Setting a Configurable requires a model to set its initial properties");const r=Object.create(Object.getPrototypeOf(t),Object.getOwnPropertyDescriptors(t)),o=0===Object.keys(r).length?e:r;for(let a in o)if(void 0!==e[a])try{"object"==typeof e[a]&&"object"==typeof t[a]?r[a]=i(e[a],t[a]):r[a]=e[a]}catch(e){(0,n.Z)("An error occurred while setting a property of a Configurable",e)}return r}catch(e){(0,n.Z)("An error occured while setting a Configurable",e)}}},6818:(e,t,r)=>{r.d(t,{Re:()=>i,gF:()=>o,q4:()=>n});const n="1.236.0",i="PROD",o="CDN"},385:(e,t,r)=>{r.d(t,{FN:()=>a,IF:()=>u,Nk:()=>f,Tt:()=>s,_A:()=>o,il:()=>n,pL:()=>c,v6:()=>i,w1:()=>d});const n="undefined"!=typeof window&&!!window.document,i="undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self.navigator instanceof WorkerNavigator||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis.navigator instanceof WorkerNavigator),o=n?window:"undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis),a=""+o?.location,s=/iPad|iPhone|iPod/.test(navigator.userAgent),c=s&&"undefined"==typeof SharedWorker,u=(()=>{const e=navigator.userAgent.match(/Firefox[/\s](\d+\.\d+)/);return Array.isArray(e)&&e.length>=2?+e[1]:0})(),d=Boolean(n&&window.document.documentMode),f=!!navigator.sendBeacon},1117:(e,t,r)=>{r.d(t,{w:()=>o});var n=r(50);const i={agentIdentifier:"",ee:void 0};class o{constructor(e){try{if("object"!=typeof e)return(0,n.Z)("shared context requires an object as input");this.sharedContext={},Object.assign(this.sharedContext,i),Object.entries(e).forEach((e=>{let[t,r]=e;Object.keys(i).includes(t)&&(this.sharedContext[t]=r)}))}catch(e){(0,n.Z)("An error occured while setting SharedContext",e)}}}},8e3:(e,t,r)=>{r.d(t,{L:()=>d,R:()=>c});var n=r(2177),i=r(1284),o=r(4322),a=r(3325);const s={};function c(e,t){const r={staged:!1,priority:a.p[t]||0};u(e),s[e].get(t)||s[e].set(t,r)}function u(e){e&&(s[e]||(s[e]=new Map))}function d(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:"",t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:"feature";if(u(e),!e||!s[e].get(t))return a(t);s[e].get(t).staged=!0;const r=[...s[e]];function a(t){const r=e?n.ee.get(e):n.ee,a=o.X.handlers;if(r.backlog&&a){var s=r.backlog[t],c=a[t];if(c){for(var u=0;s&&u {let[t,r]=e;return r.staged}))&&(r.sort(((e,t)=>e[1].priority-t[1].priority)),r.forEach((e=>{let[t]=e;a(t)})))}function f(e,t){var r=e[1];(0,i.D)(t[r],(function(t,r){var n=e[0];if(r[0]===n){var i=r[1],o=e[3],a=e[2];i.apply(o,a)}}))}},2177:(e,t,r)=>{r.d(t,{c:()=>f,ee:()=>u});var n=r(8632),i=r(2210),o=r(1284),a=r(5763),s="nr@context";let c=(0,n.fP)();var u;function d(){}function f(e){return(0,i.X)(e,s,l)}function l(){return new d}function h(){u.aborted=!0,u.backlog={}}c.ee?u=c.ee:(u=function e(t,r){var n={},c={},f={},g=!1;try{g=16===r.length&&(0,a.OP)(r).isolatedBacklog}catch(e){}var p={on:b,addEventListener:b,removeEventListener:y,emit:v,get:x,listeners:w,context:m,buffer:A,abort:h,aborted:!1,isBuffering:E,debugId:r,backlog:g?{}:t&&"object"==typeof t.backlog?t.backlog:{}};return p;function m(e){return e&&e instanceof d?e:e?(0,i.X)(e,s,l):l()}function v(e,r,n,i,o){if(!1!==o&&(o=!0),!u.aborted||i){t&&o&&t.emit(e,r,n);for(var a=m(n),s=w(e),d=s.length,f=0;fn,p:()=>i});var n=r(2177).ee.get("handle");function i(e,t,r,i,o){o?(o.buffer([e],i),o.emit(e,t,r)):(n.buffer([e],i),n.emit(e,t,r))}},4322:(e,t,r)=>{r.d(t,{X:()=>o});var n=r(5546);o.on=a;var i=o.handlers={};function o(e,t,r,o){a(o||n.E,i,e,t,r)}function a(e,t,r,i,o){o||(o="feature"),e||(e=n.E);var a=t[o]=t[o]||{};(a[r]=a[r]||[]).push([e,i])}},3239:(e,t,r)=>{r.d(t,{bP:()=>s,iz:()=>c,m$:()=>a});var n=r(385);let i=!1,o=!1;try{const e={get passive(){return i=!0,!1},get signal(){return o=!0,!1}};n._A.addEventListener("test",null,e),n._A.removeEventListener("test",null,e)}catch(e){}function a(e,t){return i||o?{capture:!!e,passive:i,signal:t}:!!e}function s(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;window.addEventListener(e,t,a(r,n))}function c(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;document.addEventListener(e,t,a(r,n))}},4402:(e,t,r)=>{r.d(t,{Ht:()=>u,M:()=>c,Rl:()=>a,ky:()=>s});var n=r(385);const i="xxxxxxxx-xxxx-4xxx-yxxx-xxxxxxxxxxxx";function o(e,t){return e?15&e[t]:16*Math.random()|0}function a(){const e=n._A?.crypto||n._A?.msCrypto;let t,r=0;return e&&e.getRandomValues&&(t=e.getRandomValues(new Uint8Array(31))),i.split("").map((e=>"x"===e?o(t,++r).toString(16):"y"===e?(3&o()|8).toString(16):e)).join("")}function s(e){const t=n._A?.crypto||n._A?.msCrypto;let r,i=0;t&&t.getRandomValues&&(r=t.getRandomValues(new Uint8Array(31)));const a=[];for(var s=0;s {r.d(t,{Bq:()=>n,Hb:()=>o,oD:()=>i});const n="NRBA",i=144e5,o=18e5},7894:(e,t,r)=>{function n(){return Math.round(performance.now())}r.d(t,{z:()=>n})},7243:(e,t,r)=>{r.d(t,{e:()=>o});var n=r(385),i={};function o(e){if(e in i)return i[e];if(0===(e||"").indexOf("data:"))return{protocol:"data"};let t;var r=n._A?.location,o={};if(n.il)t=document.createElement("a"),t.href=e;else try{t=new URL(e,r.href)}catch(e){return o}o.port=t.port;var a=t.href.split("://");!o.port&&a[1]&&(o.port=a[1].split("/")[0].split("@").pop().split(":")[1]),o.port&&"0"!==o.port||(o.port="https"===a[0]?"443":"80"),o.hostname=t.hostname||r.hostname,o.pathname=t.pathname,o.protocol=a[0],"/"!==o.pathname.charAt(0)&&(o.pathname="/"+o.pathname);var s=!t.protocol||":"===t.protocol||t.protocol===r.protocol,c=t.hostname===r.hostname&&t.port===r.port;return o.sameOrigin=s&&(!t.hostname||c),"/"===o.pathname&&(i[e]=o),o}},50:(e,t,r)=>{function n(e,t){"function"==typeof console.warn&&(console.warn("New Relic: ".concat(e)),t&&console.warn(t))}r.d(t,{Z:()=>n})},2587:(e,t,r)=>{r.d(t,{N:()=>c,T:()=>u});var n=r(2177),i=r(5546),o=r(8e3),a=r(3325);const s={stn:[a.D.sessionTrace],err:[a.D.jserrors,a.D.metrics],ins:[a.D.pageAction],spa:[a.D.spa],sr:[a.D.sessionReplay,a.D.sessionTrace]};function c(e,t){const r=n.ee.get(t);e&&"object"==typeof e&&(Object.entries(e).forEach((e=>{let[t,n]=e;void 0===u[t]&&(s[t]?s[t].forEach((e=>{n?(0,i.p)("feat-"+t,[],void 0,e,r):(0,i.p)("block-"+t,[],void 0,e,r),(0,i.p)("rumresp-"+t,[Boolean(n)],void 0,e,r)})):n&&(0,i.p)("feat-"+t,[],void 0,void 0,r),u[t]=Boolean(n))})),Object.keys(s).forEach((e=>{void 0===u[e]&&(s[e]?.forEach((t=>(0,i.p)("rumresp-"+e,[!1],void 0,t,r))),u[e]=!1)})),(0,o.L)(t,a.D.pageViewEvent))}const u={}},2210:(e,t,r)=>{r.d(t,{X:()=>i});var n=Object.prototype.hasOwnProperty;function i(e,t,r){if(n.call(e,t))return e[t];var i=r();if(Object.defineProperty&&Object.keys)try{return Object.defineProperty(e,t,{value:i,writable:!0,enumerable:!1}),i}catch(e){}return e[t]=i,i}},1284:(e,t,r)=>{r.d(t,{D:()=>n});const n=(e,t)=>Object.entries(e||{}).map((e=>{let[r,n]=e;return t(r,n)}))},4351:(e,t,r)=>{r.d(t,{P:()=>o});var n=r(2177);const i=()=>{const e=new WeakSet;return(t,r)=>{if("object"==typeof r&&null!==r){if(e.has(r))return;e.add(r)}return r}};function o(e){try{return JSON.stringify(e,i())}catch(e){try{n.ee.emit("internal-error",[e])}catch(e){}}}},3960:(e,t,r)=>{r.d(t,{K:()=>a,b:()=>o});var n=r(3239);function i(){return"undefined"==typeof document||"complete"===document.readyState}function o(e,t){if(i())return e();(0,n.bP)("load",e,t)}function a(e){if(i())return e();(0,n.iz)("DOMContentLoaded",e)}},8632:(e,t,r)=>{r.d(t,{EZ:()=>u,Qy:()=>c,ce:()=>o,fP:()=>a,gG:()=>d,mF:()=>s});var n=r(7894),i=r(385);const o={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net"};function a(){return i._A.NREUM||(i._A.NREUM={}),void 0===i._A.newrelic&&(i._A.newrelic=i._A.NREUM),i._A.NREUM}function s(){let e=a();return e.o||(e.o={ST:i._A.setTimeout,SI:i._A.setImmediate,CT:i._A.clearTimeout,XHR:i._A.XMLHttpRequest,REQ:i._A.Request,EV:i._A.Event,PR:i._A.Promise,MO:i._A.MutationObserver,FETCH:i._A.fetch}),e}function c(e,t,r){let i=a();const o=i.initializedAgents||{},s=o[e]||{};return Object.keys(s).length||(s.initializedAt={ms:(0,n.z)(),date:new Date}),i.initializedAgents={...o,[e]:{...s,[r]:t}},i}function u(e,t){a()[e]=t}function d(){return function(){let e=a();const t=e.info||{};e.info={beacon:o.beacon,errorBeacon:o.errorBeacon,...t}}(),function(){let e=a();const t=e.init||{};e.init={...t}}(),s(),function(){let e=a();const t=e.loader_config||{};e.loader_config={...t}}(),a()}},7956:(e,t,r)=>{r.d(t,{N:()=>i});var n=r(3239);function i(e){let t=arguments.length>1&&void 0!==arguments[1]&&arguments[1],r=arguments.length>2?arguments[2]:void 0,i=arguments.length>3?arguments[3]:void 0;return void(0,n.iz)("visibilitychange",(function(){if(t)return void("hidden"==document.visibilityState&&e());e(document.visibilityState)}),r,i)}},1214:(e,t,r)=>{r.d(t,{em:()=>v,u5:()=>N,QU:()=>S,_L:()=>I,Gm:()=>L,Lg:()=>M,gy:()=>U,BV:()=>Q,Kf:()=>ee});var n=r(2177);const i="nr@original";var o=Object.prototype.hasOwnProperty,a=!1;function s(e,t){return e||(e=n.ee),r.inPlace=function(e,t,n,i,o){n||(n="");var a,s,c,u="-"===n.charAt(0);for(c=0;c 2?n-2:0),o=2;o {r(A[T],e,w),r(E[T],e,w)})),r(l._A,"fetch",y),t.on(y+"end",(function(e,r){var n=this;if(r){var i=r.headers.get("content-length");null!==i&&(n.rxSize=i),t.emit(y+"done",[null,r],n)}else t.emit(y+"done",[e],n)})),t}const O={},j=["pushState","replaceState"];function S(e){const t=function(e){return(e||n.ee).get("history")}(e);return!l.il||O[t.debugId]++||(O[t.debugId]=1,s(t).inPlace(window.history,j,"-")),t}var P=r(3239);const C={},R=["appendChild","insertBefore","replaceChild"];function I(e){const t=function(e){return(e||n.ee).get("jsonp")}(e);if(!l.il||C[t.debugId])return t;C[t.debugId]=!0;var r=s(t),i=/[?&](?:callback|cb)=([^&#]+)/,o=/(.*)\.([^.]+)/,a=/^(\w+)(\.|$)(.*)$/;function c(e,t){var r=e.match(a),n=r[1],i=r[3];return i?c(i,t[n]):t[n]}return r.inPlace(Node.prototype,R,"dom-"),t.on("dom-start",(function(e){!function(e){if(!e||"string"!=typeof e.nodeName||"script"!==e.nodeName.toLowerCase())return;if("function"!=typeof e.addEventListener)return;var n=(a=e.src,s=a.match(i),s?s[1]:null);var a,s;if(!n)return;var u=function(e){var t=e.match(o);if(t&&t.length>=3)return{key:t[2],parent:c(t[1],window)};return{key:e,parent:window}}(n);if("function"!=typeof u.parent[u.key])return;var d={};function f(){t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}function l(){t.emit("jsonp-error",[],d),t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}r.inPlace(u.parent,[u.key],"cb-",d),e.addEventListener("load",f,(0,P.m$)(!1)),e.addEventListener("error",l,(0,P.m$)(!1)),t.emit("new-jsonp",[e.src],d)}(e[0])})),t}var k=r(5763);const H={};function L(e){const t=function(e){return(e||n.ee).get("mutation")}(e);if(!l.il||H[t.debugId])return t;H[t.debugId]=!0;var r=s(t),i=k.Yu.MO;return i&&(window.MutationObserver=function(e){return this instanceof i?new i(r(e,"fn-")):i.apply(this,arguments)},MutationObserver.prototype=i.prototype),t}const z={};function M(e){const t=function(e){return(e||n.ee).get("promise")}(e);if(z[t.debugId])return t;z[t.debugId]=!0;var r=n.c,o=s(t),a=k.Yu.PR;return a&&function(){function e(r){var n=t.context(),i=o(r,"executor-",n,null,!1);const s=Reflect.construct(a,[i],e);return t.context(s).getCtx=function(){return n},s}l._A.Promise=e,Object.defineProperty(e,"name",{value:"Promise"}),e.toString=function(){return a.toString()},Object.setPrototypeOf(e,a),["all","race"].forEach((function(r){const n=a[r];e[r]=function(e){let i=!1;[...e||[]].forEach((e=>{this.resolve(e).then(a("all"===r),a(!1))}));const o=n.apply(this,arguments);return o;function a(e){return function(){t.emit("propagate",[null,!i],o,!1,!1),i=i||!e}}}})),["resolve","reject"].forEach((function(r){const n=a[r];e[r]=function(e){const r=n.apply(this,arguments);return e!==r&&t.emit("propagate",[e,!0],r,!1,!1),r}})),e.prototype=a.prototype;const n=a.prototype.then;a.prototype.then=function(){var e=this,i=r(e);i.promise=e;for(var a=arguments.length,s=new Array(a),c=0;c e())),t};function m(e,t){i.inPlace(t,["onreadystatechange"],"fn-",E)}function b(){var e=this,t=r.context(e);e.readyState>3&&!t.resolved&&(t.resolved=!0,r.emit("xhr-resolved",[],e)),i.inPlace(e,f,"fn-",E)}if(function(e,t){for(var r in e)t[r]=e[r]}(o,p),p.prototype=o.prototype,i.inPlace(p.prototype,J,"-xhr-",E),r.on("send-xhr-start",(function(e,t){m(e,t),function(e){h.push(e),a&&(y?y.then(A):u?u(A):(w=-w,x.data=w))}(t)})),r.on("open-xhr-start",m),a){var y=c&&c.resolve();if(!u&&!c){var w=1,x=document.createTextNode(w);new a(A).observe(x,{characterData:!0})}}else t.on("fn-end",(function(e){e[0]&&e[0].type===d||A()}));function A(){for(var e=0;e {r.d(t,{t:()=>n});const n=r(3325).D.ajax},6660:(e,t,r)=>{r.d(t,{A:()=>i,t:()=>n});const n=r(3325).D.jserrors,i="nr@seenError"},3081:(e,t,r)=>{r.d(t,{gF:()=>o,mY:()=>i,t9:()=>n,vz:()=>s,xS:()=>a});const n=r(3325).D.metrics,i="sm",o="cm",a="storeSupportabilityMetrics",s="storeEventMetrics"},4649:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageAction},7633:(e,t,r)=>{r.d(t,{Dz:()=>i,OJ:()=>a,qw:()=>o,t9:()=>n});const n=r(3325).D.pageViewEvent,i="firstbyte",o="domcontent",a="windowload"},9251:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageViewTiming},3614:(e,t,r)=>{r.d(t,{BST_RESOURCE:()=>i,END:()=>s,FEATURE_NAME:()=>n,FN_END:()=>u,FN_START:()=>c,PUSH_STATE:()=>d,RESOURCE:()=>o,START:()=>a});const n=r(3325).D.sessionTrace,i="bstResource",o="resource",a="-start",s="-end",c="fn"+a,u="fn"+s,d="pushState"},7836:(e,t,r)=>{r.d(t,{BODY:()=>A,CB_END:()=>E,CB_START:()=>u,END:()=>x,FEATURE_NAME:()=>i,FETCH:()=>_,FETCH_BODY:()=>v,FETCH_DONE:()=>m,FETCH_START:()=>p,FN_END:()=>c,FN_START:()=>s,INTERACTION:()=>l,INTERACTION_API:()=>d,INTERACTION_EVENTS:()=>o,JSONP_END:()=>b,JSONP_NODE:()=>g,JS_TIME:()=>T,MAX_TIMER_BUDGET:()=>a,REMAINING:()=>f,SPA_NODE:()=>h,START:()=>w,originalSetTimeout:()=>y});var n=r(5763);const i=r(3325).D.spa,o=["click","submit","keypress","keydown","keyup","change"],a=999,s="fn-start",c="fn-end",u="cb-start",d="api-ixn-",f="remaining",l="interaction",h="spaNode",g="jsonpNode",p="fetch-start",m="fetch-done",v="fetch-body-",b="jsonp-end",y=n.Yu.ST,w="-start",x="-end",A="-body",E="cb"+x,T="jsTime",_="fetch"},5938:(e,t,r)=>{r.d(t,{W:()=>o});var n=r(5763),i=r(2177);class o{constructor(e,t,r){this.agentIdentifier=e,this.aggregator=t,this.ee=i.ee.get(e,(0,n.OP)(this.agentIdentifier).isolatedBacklog),this.featureName=r,this.blocked=!1}}},9144:(e,t,r)=>{r.d(t,{j:()=>m});var n=r(3325),i=r(5763),o=r(5546),a=r(2177),s=r(7894),c=r(8e3),u=r(3960),d=r(385),f=r(50),l=r(3081),h=r(8632);function g(){const e=(0,h.gG)();["setErrorHandler","finished","addToTrace","inlineHit","addRelease","addPageAction","setCurrentRouteName","setPageViewName","setCustomAttribute","interaction","noticeError","setUserId"].forEach((t=>{e[t]=function(){for(var r=arguments.length,n=new Array(r),i=0;i 1?r-1:0),i=1;i {e.exposed&&e.api[t]&&o.push(e.api[t](...n))})),o.length>1?o:o[0]}(t,...n)}}))}var p=r(2587);function m(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:{},m=arguments.length>2?arguments[2]:void 0,v=arguments.length>3?arguments[3]:void 0,{init:b,info:y,loader_config:w,runtime:x={loaderType:m},exposed:A=!0}=t;const E=(0,h.gG)();y||(b=E.init,y=E.info,w=E.loader_config),(0,i.Dg)(e,b||{}),(0,i.GE)(e,w||{}),(0,i.sU)(e,x),y.jsAttributes??={},d.v6&&(y.jsAttributes.isWorker=!0),(0,i.CX)(e,y),g();const T=function(e,t){t||(0,c.R)(e,"api");const h={};var g=a.ee.get(e),p=g.get("tracer"),m="api-",v=m+"ixn-";function b(t,r,n,o){const a=(0,i.C5)(e);return null===r?delete a.jsAttributes[t]:(0,i.CX)(e,{...a,jsAttributes:{...a.jsAttributes,[t]:r}}),x(m,n,!0,o||null===r?"session":void 0)(t,r)}function y(){}["setErrorHandler","finished","addToTrace","inlineHit","addRelease"].forEach((e=>h[e]=x(m,e,!0,"api"))),h.addPageAction=x(m,"addPageAction",!0,n.D.pageAction),h.setCurrentRouteName=x(m,"routeName",!0,n.D.spa),h.setPageViewName=function(t,r){if("string"==typeof t)return"/"!==t.charAt(0)&&(t="/"+t),(0,i.OP)(e).customTransaction=(r||"http://custom.transaction")+t,x(m,"setPageViewName",!0)()},h.setCustomAttribute=function(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2];if("string"==typeof e){if(["string","number"].includes(typeof t)||null===t)return b(e,t,"setCustomAttribute",r);(0,f.Z)("Failed to execute setCustomAttribute.\nNon-null value must be a string or number type, but a type of was provided."))}else(0,f.Z)("Failed to execute setCustomAttribute.\nName must be a string type, but a type of was provided."))},h.setUserId=function(e){if("string"==typeof e||null===e)return b("enduser.id",e,"setUserId",!0);(0,f.Z)("Failed to execute setUserId.\nNon-null value must be a string type, but a type of was provided."))},h.interaction=function(){return(new y).get()};var w=y.prototype={createTracer:function(e,t){var r={},i=this,a="function"==typeof t;return(0,o.p)(v+"tracer",[(0,s.z)(),e,r],i,n.D.spa,g),function(){if(p.emit((a?"":"no-")+"fn-start",[(0,s.z)(),i,a],r),a)try{return t.apply(this,arguments)}catch(e){throw p.emit("fn-err",[arguments,this,"string"==typeof e?new Error(e):e],r),e}finally{p.emit("fn-end",[(0,s.z)()],r)}}}};function x(e,t,r,i){return function(){return(0,o.p)(l.xS,["API/"+t+"/called"],void 0,n.D.metrics,g),i&&(0,o.p)(e+t,[(0,s.z)(),...arguments],r?null:this,i,g),r?void 0:this}}function A(){r.e(439).then(r.bind(r,7438)).then((t=>{let{setAPI:r}=t;r(e),(0,c.L)(e,"api")})).catch((()=>(0,f.Z)("Downloading runtime APIs failed...")))}return["actionText","setName","setAttribute","save","ignore","onEnd","getContext","end","get"].forEach((e=>{w[e]=x(v,e,void 0,n.D.spa)})),h.noticeError=function(e,t){"string"==typeof e&&(e=new Error(e)),(0,o.p)(l.xS,["API/noticeError/called"],void 0,n.D.metrics,g),(0,o.p)("err",[e,(0,s.z)(),!1,t],void 0,n.D.jserrors,g)},d.il?(0,u.b)((()=>A()),!0):A(),h}(e,v);return(0,h.Qy)(e,T,"api"),(0,h.Qy)(e,A,"exposed"),(0,h.EZ)("activatedFeatures",p.T),T}},3325:(e,t,r)=>{r.d(t,{D:()=>n,p:()=>i});const n={ajax:"ajax",jserrors:"jserrors",metrics:"metrics",pageAction:"page_action",pageViewEvent:"page_view_event",pageViewTiming:"page_view_timing",sessionReplay:"session_replay",sessionTrace:"session_trace",spa:"spa"},i={[n.pageViewEvent]:1,[n.pageViewTiming]:2,[n.metrics]:3,[n.jserrors]:4,[n.ajax]:5,[n.sessionTrace]:6,[n.pageAction]:7,[n.spa]:8,[n.sessionReplay]:9}}},n={};function i(e){var t=n[e];if(void 0!==t)return t.exports;var o=n[e]={exports:{}};return r[e](o,o.exports,i),o.exports}i.m=r,i.d=(e,t)=>{for(var r in t)i.o(t,r)&&!i.o(e,r)&&Object.defineProperty(e,r,{enumerable:!0,get:t[r]})},i.f={},i.e=e=>Promise.all(Object.keys(i.f).reduce(((t,r)=>(i.f[r](e,t),t)),[])),i.u=e=>(({78:"page_action-aggregate",147:"metrics-aggregate",242:"session-manager",317:"jserrors-aggregate",348:"page_view_timing-aggregate",412:"lazy-feature-loader",439:"async-api",538:"recorder",590:"session_replay-aggregate",675:"compressor",733:"session_trace-aggregate",786:"page_view_event-aggregate",873:"spa-aggregate",898:"ajax-aggregate"}[e]||e)+"."+{78:"ac76d497",147:"3dc53903",148:"1a20d5fe",242:"2a64278a",317:"49e41428",348:"bd6de33a",412:"2f55ce66",439:"30bd804e",538:"1b18459f",590:"cf0efb30",675:"ae9f91a8",733:"83105561",786:"06482edd",860:"03a8b7a5",873:"e6b09d52",898:"998ef92b"}[e]+"-1.236.0.min.js"),i.o=(e,t)=>Object.prototype.hasOwnProperty.call(e,t),e={},t="NRBA:",i.l=(r,n,o,a)=>{if(e[r])e[r].push(n);else{var s,c;if(void 0!==o)for(var u=document.getElementsByTagName("script"),d=0;d {s.onerror=s.onload=null,clearTimeout(h);var i=e[r];if(delete e[r],s.parentNode&&s.parentNode.removeChild(s),i&&i.forEach((e=>e(n))),t)return t(n)},h=setTimeout(l.bind(null,void 0,{type:"timeout",target:s}),12e4);s.onerror=l.bind(null,s.onerror),s.onload=l.bind(null,s.onload),c&&document.head.appendChild(s)}},i.r=e=>{"undefined"!=typeof Symbol&&Symbol.toStringTag&&Object.defineProperty(e,Symbol.toStringTag,{value:"Module"}),Object.defineProperty(e,"__esModule",{value:!0})},i.j=364,i.p="https://js-agent.newrelic.com/",(()=>{var e={364:0,953:0};i.f.j=(t,r)=>{var n=i.o(e,t)?e[t]:void 0;if(0!==n)if(n)r.push(n[2]);else{var o=new Promise(((r,i)=>n=e[t]=[r,i]));r.push(n[2]=o);var a=i.p+i.u(t),s=new Error;i.l(a,(r=>{if(i.o(e,t)&&(0!==(n=e[t])&&(e[t]=void 0),n)){var o=r&&("load"===r.type?"missing":r.type),a=r&&r.target&&r.target.src;s.message="Loading chunk "+t+" failed.\n("+o+": "+a+")",s.name="ChunkLoadError",s.type=o,s.request=a,n[1](s)}}),"chunk-"+t,t)}};var t=(t,r)=>{var n,o,[a,s,c]=r,u=0;if(a.some((t=>0!==e[t]))){for(n in s)i.o(s,n)&&(i.m[n]=s[n]);if(c)c(i)}for(t&&t(r);u {i.r(o);var e=i(3325),t=i(5763);const r=Object.values(e.D);function n(e){const n={};return r.forEach((r=>{n[r]=function(e,r){return!1!==(0,t.Mt)(r,"".concat(e,".enabled"))}(r,e)})),n}var a=i(9144);var s=i(5546),c=i(385),u=i(8e3),d=i(5938),f=i(3960),l=i(50);class h extends d.W{constructor(e,t,r){let n=!(arguments.length>3&&void 0!==arguments[3])||arguments[3];super(e,t,r),this.auto=n,this.abortHandler,this.featAggregate,this.onAggregateImported,n&&(0,u.R)(e,r)}importAggregator(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:{};if(this.featAggregate||!this.auto)return;const r=c.il&&!0===(0,t.Mt)(this.agentIdentifier,"privacy.cookies_enabled");let n;this.onAggregateImported=new Promise((e=>{n=e}));const o=async()=>{let t;try{if(r){const{setupAgentSession:e}=await Promise.all([i.e(860),i.e(242)]).then(i.bind(i,3228));t=e(this.agentIdentifier)}}catch(e){(0,l.Z)("A problem occurred when starting up session manager. This page will not start or extend any session.",e)}try{if(!this.shouldImportAgg(this.featureName,t))return void(0,u.L)(this.agentIdentifier,this.featureName);const{lazyFeatureLoader:r}=await i.e(412).then(i.bind(i,8582)),{Aggregate:o}=await r(this.featureName,"aggregate");this.featAggregate=new o(this.agentIdentifier,this.aggregator,e),n(!0)}catch(e){(0,l.Z)("Downloading and initializing ".concat(this.featureName," failed..."),e),this.abortHandler?.(),n(!1)}};c.il?(0,f.b)((()=>o()),!0):o()}shouldImportAgg(r,n){return r!==e.D.sessionReplay||!1!==(0,t.Mt)(this.agentIdentifier,"session_trace.enabled")&&(!!n?.isNew||!!n?.state.sessionReplay)}}var g=i(7633),p=i(7894);class m extends h{static featureName=g.t9;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];if(super(r,n,g.t9,i),("undefined"==typeof PerformanceNavigationTiming||c.Tt)&&"undefined"!=typeof PerformanceTiming){const n=(0,t.OP)(r);n[g.Dz]=Math.max(Date.now()-n.offset,0),(0,f.K)((()=>n[g.qw]=Math.max((0,p.z)()-n[g.Dz],0))),(0,f.b)((()=>{const t=(0,p.z)();n[g.OJ]=Math.max(t-n[g.Dz],0),(0,s.p)("timing",["load",t],void 0,e.D.pageViewTiming,this.ee)}))}this.importAggregator()}}var v=i(1117),b=i(1284);class y extends v.w{constructor(e){super(e),this.aggregatedData={}}store(e,t,r,n,i){var o=this.getBucket(e,t,r,i);return o.metrics=function(e,t){t||(t={count:0});return t.count+=1,(0,b.D)(e,(function(e,r){t[e]=w(r,t[e])})),t}(n,o.metrics),o}merge(e,t,r,n,i){var o=this.getBucket(e,t,n,i);if(o.metrics){var a=o.metrics;a.count+=r.count,(0,b.D)(r,(function(e,t){if("count"!==e){var n=a[e],i=r[e];i&&!i.c?a[e]=w(i.t,n):a[e]=function(e,t){if(!t)return e;t.c||(t=x(t.t));return t.min=Math.min(e.min,t.min),t.max=Math.max(e.max,t.max),t.t+=e.t,t.sos+=e.sos,t.c+=e.c,t}(i,a[e])}}))}else o.metrics=r}storeMetric(e,t,r,n){var i=this.getBucket(e,t,r);return i.stats=w(n,i.stats),i}getBucket(e,t,r,n){this.aggregatedData[e]||(this.aggregatedData[e]={});var i=this.aggregatedData[e][t];return i||(i=this.aggregatedData[e][t]={params:r||{}},n&&(i.custom=n)),i}get(e,t){return t?this.aggregatedData[e]&&this.aggregatedData[e][t]:this.aggregatedData[e]}take(e){for(var t={},r="",n=!1,i=0;i t.max&&(t.max=e),e 2&&void 0!==arguments[2])||arguments[2];super(e,r,j.t,n),c.il&&((0,t.OP)(e).initHidden=Boolean("hidden"===document.visibilityState),(0,N.N)((()=>(0,s.p)("docHidden",[(0,p.z)()],void 0,j.t,this.ee)),!0),(0,O.bP)("pagehide",(()=>(0,s.p)("winPagehide",[(0,p.z)()],void 0,j.t,this.ee))),this.importAggregator())}}var P=i(3081);class C extends h{static featureName=P.t9;constructor(e,t){let r=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(e,t,P.t9,r),this.importAggregator()}}var R,I=i(2210),k=i(1214),H=i(2177),L={};try{R=localStorage.getItem("__nr_flags").split(","),console&&"function"==typeof console.log&&(L.console=!0,-1!==R.indexOf("dev")&&(L.dev=!0),-1!==R.indexOf("nr_dev")&&(L.nrDev=!0))}catch(e){}function z(e){try{L.console&&z(e)}catch(e){}}L.nrDev&&H.ee.on("internal-error",(function(e){z(e.stack)})),L.dev&&H.ee.on("fn-err",(function(e,t,r){z(r.stack)})),L.dev&&(z("NR AGENT IN DEVELOPMENT MODE"),z("flags: "+(0,b.D)(L,(function(e,t){return e})).join(", ")));var M=i(6660);class B extends h{static featureName=M.t;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(r,n,M.t,i),this.skipNext=0;try{this.removeOnAbort=new AbortController}catch(e){}const o=this;o.ee.on("fn-start",(function(e,t,r){o.abortHandler&&(o.skipNext+=1)})),o.ee.on("fn-err",(function(t,r,n){o.abortHandler&&!n[M.A]&&((0,I.X)(n,M.A,(function(){return!0})),this.thrown=!0,(0,s.p)("err",[n,(0,p.z)()],void 0,e.D.jserrors,o.ee))})),o.ee.on("fn-end",(function(){o.abortHandler&&!this.thrown&&o.skipNext>0&&(o.skipNext-=1)})),o.ee.on("internal-error",(function(t){(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,o.ee)})),this.origOnerror=c._A.onerror,c._A.onerror=this.onerrorHandler.bind(this),c._A.addEventListener("unhandledrejection",(t=>{const r=function(e){let t="Unhandled Promise Rejection: ";if(e instanceof Error)try{return e.message=t+e.message,e}catch(t){return e}if(void 0===e)return new Error(t);try{return new Error(t+(0,D.P)(e))}catch(e){return new Error(t)}}(t.reason);(0,s.p)("err",[r,(0,p.z)(),!1,{unhandledPromiseRejection:1}],void 0,e.D.jserrors,this.ee)}),(0,O.m$)(!1,this.removeOnAbort?.signal)),(0,k.gy)(this.ee),(0,k.BV)(this.ee),(0,k.em)(this.ee),(0,t.OP)(r).xhrWrappable&&(0,k.Kf)(this.ee),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}onerrorHandler(t,r,n,i,o){"function"==typeof this.origOnerror&&this.origOnerror(...arguments);try{this.skipNext?this.skipNext-=1:(0,s.p)("err",[o||new F(t,r,n),(0,p.z)()],void 0,e.D.jserrors,this.ee)}catch(t){try{(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,this.ee)}catch(e){}}return!1}}function F(e,t,r){this.message=e||"Uncaught error with no additional information",this.sourceURL=t,this.line=r}let U=1;const q="nr@id";function G(e){const t=typeof e;return!e||"object"!==t&&"function"!==t?-1:e===c._A?0:(0,I.X)(e,q,(function(){return U++}))}function V(e){if("string"==typeof e&&e.length)return e.length;if("object"==typeof e){if("undefined"!=typeof ArrayBuffer&&e instanceof ArrayBuffer&&e.byteLength)return e.byteLength;if("undefined"!=typeof Blob&&e instanceof Blob&&e.size)return e.size;if(!("undefined"!=typeof FormData&&e instanceof FormData))try{return(0,D.P)(e).length}catch(e){return}}}var X=i(7243);class W{constructor(e){this.agentIdentifier=e,this.generateTracePayload=this.generateTracePayload.bind(this),this.shouldGenerateTrace=this.shouldGenerateTrace.bind(this)}generateTracePayload(e){if(!this.shouldGenerateTrace(e))return null;var r=(0,t.DL)(this.agentIdentifier);if(!r)return null;var n=(r.accountID||"").toString()||null,i=(r.agentID||"").toString()||null,o=(r.trustKey||"").toString()||null;if(!n||!i)return null;var a=(0,_.M)(),s=(0,_.Ht)(),c=Date.now(),u={spanId:a,traceId:s,timestamp:c};return(e.sameOrigin||this.isAllowedOrigin(e)&&this.useTraceContextHeadersForCors())&&(u.traceContextParentHeader=this.generateTraceContextParentHeader(a,s),u.traceContextStateHeader=this.generateTraceContextStateHeader(a,c,n,i,o)),(e.sameOrigin&&!this.excludeNewrelicHeader()||!e.sameOrigin&&this.isAllowedOrigin(e)&&this.useNewrelicHeaderForCors())&&(u.newrelicHeader=this.generateTraceHeader(a,s,c,n,i,o)),u}generateTraceContextParentHeader(e,t){return"00-"+t+"-"+e+"-01"}generateTraceContextStateHeader(e,t,r,n,i){return i+"@nr=0-1-"+r+"-"+n+"-"+e+"----"+t}generateTraceHeader(e,t,r,n,i,o){if(!("function"==typeof c._A?.btoa))return null;var a={v:[0,1],d:{ty:"Browser",ac:n,ap:i,id:e,tr:t,ti:r}};return o&&n!==o&&(a.d.tk=o),btoa((0,D.P)(a))}shouldGenerateTrace(e){return this.isDtEnabled()&&this.isAllowedOrigin(e)}isAllowedOrigin(e){var r=!1,n={};if((0,t.Mt)(this.agentIdentifier,"distributed_tracing")&&(n=(0,t.P_)(this.agentIdentifier).distributed_tracing),e.sameOrigin)r=!0;else if(n.allowed_origins instanceof Array)for(var i=0;i 2&&void 0!==arguments[2])||arguments[2];super(r,n,Z.t,i),(0,t.OP)(r).xhrWrappable&&(this.dt=new W(r),this.handler=(e,t,r,n)=>(0,s.p)(e,t,r,n,this.ee),(0,k.u5)(this.ee),(0,k.Kf)(this.ee),function(r,n,i,o){function a(e){var t=this;t.totalCbs=0,t.called=0,t.cbTime=0,t.end=E,t.ended=!1,t.xhrGuids={},t.lastSize=null,t.loadCaptureCalled=!1,t.params=this.params||{},t.metrics=this.metrics||{},e.addEventListener("load",(function(r){_(t,e)}),(0,O.m$)(!1)),c.IF||e.addEventListener("progress",(function(e){t.lastSize=e.loaded}),(0,O.m$)(!1))}function s(e){this.params={method:e[0]},T(this,e[1]),this.metrics={}}function u(e,n){var i=(0,t.DL)(r);i.xpid&&this.sameOrigin&&n.setRequestHeader("X-NewRelic-ID",i.xpid);var a=o.generateTracePayload(this.parsedOrigin);if(a){var s=!1;a.newrelicHeader&&(n.setRequestHeader("newrelic",a.newrelicHeader),s=!0),a.traceContextParentHeader&&(n.setRequestHeader("traceparent",a.traceContextParentHeader),a.traceContextStateHeader&&n.setRequestHeader("tracestate",a.traceContextStateHeader),s=!0),s&&(this.dt=a)}}function d(e,t){var r=this.metrics,i=e[0],o=this;if(r&&i){var a=V(i);a&&(r.txSize=a)}this.startTime=(0,p.z)(),this.listener=function(e){try{"abort"!==e.type||o.loadCaptureCalled||(o.params.aborted=!0),("load"!==e.type||o.called===o.totalCbs&&(o.onloadCalled||"function"!=typeof t.onload)&&"function"==typeof o.end)&&o.end(t)}catch(e){try{n.emit("internal-error",[e])}catch(e){}}};for(var s=0;s 1?e[1]=i:e.push(i)}else e[0]&&e[0].headers&&s(e[0].headers,n)&&(this.dt=n);function s(e,t){var r=!1;return t.newrelicHeader&&(e.set("newrelic",t.newrelicHeader),r=!0),t.traceContextParentHeader&&(e.set("traceparent",t.traceContextParentHeader),t.traceContextStateHeader&&e.set("tracestate",t.traceContextStateHeader),r=!0),r}}function x(e,t){this.params={},this.metrics={},this.startTime=(0,p.z)(),this.dt=t,e.length>=1&&(this.target=e[0]),e.length>=2&&(this.opts=e[1]);var r,n=this.opts||{},i=this.target;"string"==typeof i?r=i:"object"==typeof i&&i instanceof Y?r=i.url:c._A?.URL&&"object"==typeof i&&i instanceof URL&&(r=i.href),T(this,r);var o=(""+(i&&i instanceof Y&&i.method||n.method||"GET")).toUpperCase();this.params.method=o,this.txSize=V(n.body)||0}function A(t,r){var n;this.endTime=(0,p.z)(),this.params||(this.params={}),this.params.status=r?r.status:0,"string"==typeof this.rxSize&&this.rxSize.length>0&&(n=+this.rxSize);var o={txSize:this.txSize,rxSize:n,duration:(0,p.z)()-this.startTime};i("xhr",[this.params,o,this.startTime,this.endTime,"fetch"],this,e.D.ajax)}function E(t){var r=this.params,n=this.metrics;if(!this.ended){this.ended=!0;for(var o=0;o 2&&void 0!==arguments[2])||arguments[2];super(e,t,we.t,r),this.importAggregator()}}new class{constructor(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:(0,_.ky)(16);c._A?(this.agentIdentifier=t,this.sharedAggregator=new y({agentIdentifier:this.agentIdentifier}),this.features={},this.desiredFeatures=new Set(e.features||[]),this.desiredFeatures.add(m),Object.assign(this,(0,a.j)(this.agentIdentifier,e,e.loaderType||"agent")),this.start()):(0,l.Z)("Failed to initial the agent. Could not determine the runtime environment.")}get config(){return{info:(0,t.C5)(this.agentIdentifier),init:(0,t.P_)(this.agentIdentifier),loader_config:(0,t.DL)(this.agentIdentifier),runtime:(0,t.OP)(this.agentIdentifier)}}start(){const t="features";try{const r=n(this.agentIdentifier),i=[...this.desiredFeatures];i.sort(((t,r)=>e.p[t.featureName]-e.p[r.featureName])),i.forEach((t=>{if(r[t.featureName]||t.featureName===e.D.pageViewEvent){const n=function(t){switch(t){case e.D.ajax:return[e.D.jserrors];case e.D.sessionTrace:return[e.D.ajax,e.D.pageViewEvent];case e.D.sessionReplay:return[e.D.sessionTrace];case e.D.pageViewTiming:return[e.D.pageViewEvent];default:return[]}}(t.featureName);n.every((e=>r[e]))||(0,l.Z)("".concat(t.featureName," is enabled but one or more dependent features has been disabled (").concat((0,D.P)(n),"). This may cause unintended consequences or missing data...")),this.features[t.featureName]=new t(this.agentIdentifier,this.sharedAggregator)}})),(0,T.Qy)(this.agentIdentifier,this.features,t)}catch(e){(0,l.Z)("Failed to initialize all enabled instrument classes (agent aborted) -",e);for(const e in this.features)this.features[e].abortHandler?.();const r=(0,T.fP)();return delete r.initializedAgents[this.agentIdentifier]?.api,delete r.initializedAgents[this.agentIdentifier]?.[t],delete this.sharedAggregator,r.ee?.abort(),delete r.ee?.get(this.agentIdentifier),!1}}}({features:[J,m,S,class extends h{static featureName=oe;constructor(t,r){if(super(t,r,oe,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;const n=this.ee;let i;(0,k.QU)(n),this.eventsEE=(0,k.em)(n),this.eventsEE.on(se,(function(e,t){this.bstStart=(0,p.z)()})),this.eventsEE.on(ae,(function(t,r){(0,s.p)("bst",[t[0],r,this.bstStart,(0,p.z)()],void 0,e.D.sessionTrace,n)})),n.on(ce+ne,(function(e){this.time=(0,p.z)(),this.startPath=location.pathname+location.hash})),n.on(ce+ie,(function(t){(0,s.p)("bstHist",[location.pathname+location.hash,this.startPath,this.time],void 0,e.D.sessionTrace,n)}));try{i=new PerformanceObserver((t=>{const r=t.getEntries();(0,s.p)(te,[r],void 0,e.D.sessionTrace,n)})),i.observe({type:re,buffered:!0})}catch(e){}this.importAggregator({resourceObserver:i})}},C,xe,B,class extends h{static featureName=de;constructor(e,r){if(super(e,r,de,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;if(!(0,t.OP)(e).xhrWrappable)return;try{this.removeOnAbort=new AbortController}catch(e){}let n,i=0;const o=this.ee.get("tracer"),a=(0,k._L)(this.ee),s=(0,k.Lg)(this.ee),u=(0,k.BV)(this.ee),d=(0,k.Kf)(this.ee),f=this.ee.get("events"),l=(0,k.u5)(this.ee),h=(0,k.QU)(this.ee),g=(0,k.Gm)(this.ee);function m(e,t){h.emit("newURL",[""+window.location,t])}function v(){i++,n=window.location.hash,this[ve]=(0,p.z)()}function b(){i--,window.location.hash!==n&&m(0,!0);var e=(0,p.z)();this[pe]=~~this[pe]+e-this[ve],this[ye]=e}function y(e,t){e.on(t,(function(){this[t]=(0,p.z)()}))}this.ee.on(ve,v),s.on(be,v),a.on(be,v),this.ee.on(ye,b),s.on(ge,b),a.on(ge,b),this.ee.buffer([ve,ye,"xhr-resolved"],this.featureName),f.buffer([ve],this.featureName),u.buffer(["setTimeout"+le,"clearTimeout"+fe,ve],this.featureName),d.buffer([ve,"new-xhr","send-xhr"+fe],this.featureName),l.buffer([me+fe,me+"-done",me+he+fe,me+he+le],this.featureName),h.buffer(["newURL"],this.featureName),g.buffer([ve],this.featureName),s.buffer(["propagate",be,ge,"executor-err","resolve"+fe],this.featureName),o.buffer([ve,"no-"+ve],this.featureName),a.buffer(["new-jsonp","cb-start","jsonp-error","jsonp-end"],this.featureName),y(l,me+fe),y(l,me+"-done"),y(a,"new-jsonp"),y(a,"jsonp-end"),y(a,"cb-start"),h.on("pushState-end",m),h.on("replaceState-end",m),window.addEventListener("hashchange",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("load",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("popstate",(function(){m(0,i>1)}),(0,O.m$)(!0,this.removeOnAbort?.signal)),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}}],loaderType:"spa"})})(),window.NRBA=o})(); window.jQuery || document.write(' ') CKEDITOR_BASEPATH='https://f1000research.com/js/vendor/ckeditor/' window.reactTheme = 'research'; window.MathJax = { CommonHTML: { linebreaks: { automatic: true } }, 'HTML-CSS': { linebreaks: { automatic: true } }, SVG: { linebreaks: { automatic: true } }, AuthorInit: function() { MathJax.Hub.Register.MessageHook('End Process', function () { let timeout = false; // holder for timeout id const delay = 250; // delay after event is "complete" to run callback const reflowMath = function() { const dispFormulas = document.querySelectorAll('.disp-formula.panel'); if (!dispFormulas) { return; } for (const dispFormula of dispFormulas) { const child = dispFormula.querySelector('.MathJax_Preview').nextSibling.firstChild; const isMultiline = MathJax.Hub.getAllJax(dispFormula)[0].root.isMultiline; if (dispFormula.offsetWidth < child.offsetWidth || isMultiline) { MathJax.Hub.Queue(['Rerender', MathJax.Hub, dispFormula]); } } }; window.addEventListener('resize', function() { clearTimeout(timeout); // clear the timeout timeout = setTimeout(reflowMath, delay); // start timing for event "completion" }); }); }, }; if (window.location.hash == '#_=_'){ window.location = window.location.href.split('#')[0] } !function(f,b,e,v,n,t,s){if(f.fbq)return;n=f.fbq=function() {n.callMethod? n.callMethod.apply(n,arguments):n.queue.push(arguments)} ;if(!f._fbq)f._fbq=n; n.push=n;n.loaded=!0;n.version='2.0';n.queue=[];t=b.createElement(e);t.async=!0; t.src=v;s=b.getElementsByTagName(e)[0];s.parentNode.insertBefore(t,s)}(window, document,'script','https://connect.facebook.net/en_US/fbevents.js'); fbq('init', '1641728616063202'); fbq('track', "PixelInitialized", {}); (function(h,o,t,j,a,r){ h.hj=h.hj||function(){(h.hj.q=h.hj.q||[]).push(arguments)}; h._hjSettings={hjid:2318163,hjsv:6}; a=o.getElementsByTagName('head')[0]; r=o.createElement('script');r.async=1; r.src=t+h._hjSettings.hjid+j+h._hjSettings.hjsv; a.appendChild(r); })(window,document,'https://static.hotjar.com/c/hotjar-','.js?sv='); search file_upload Submit your research search menu close search Browse Gateways & Collections How to Publish Submit your Research My Submissions Article Guidelines Article Guidelines (New Versions) Open Data, Software and Code Guidelines Open Data and Accessible Source Materials Guidelines (HSS) Open Data, Software and Code Guidelines (PSE) Prepublication Checks Production Process Posters and Slides Guidelines Document Guidelines Article Processing Charges Peer Review Finding Article Reviewers About How it Works For Reviewers Our Advisors Policies Glossary FAQs For Developers Newsroom Contact My Research Submissions Content and Tracking Alerts My Details Sign In file_upload Submit your research { "@context": "https://schema.org", "@type": "ScholarlyArticle", "mainEntityOfPage": { "@type": "WebPage", "@id": "https://f1000research.com/articles/6-2010" }, "headline": "Easy and efficient ensemble gene set testing with EGSEA", "datePublished": "2017-11-14T11:56:34", "dateModified": "2017-11-14T11:56:34", "author": [ { "@type": "Person", "name": "Monther Alhamdoosh" }, { "@type": "Person", "name": "Charity W. Law" }, { "@type": "Person", "name": "Luyi Tian" }, { "@type": "Person", "name": "Julie M. Sheridan" }, { "@type": "Person", "name": "Milica Ng" }, { "@type": "Person", "name": "Matthew E. Ritchie" } ], "publisher": { "@type": "Organization", "name": "F1000Research", "logo": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 480, "width": 60 } }, "image": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 1200, "width": 150 }, "description": "Gene set enrichment analysis is a popular approach for prioritising the biological processes perturbed in genomic datasets. The Bioconductor project hosts over 80 software packages capable of gene set analysis. Most of these packages search for enriched signatures amongst differentially regulated genes to reveal higher level biological themes that may be missed when focusing only on evidence from individual genes. With so many different methods on offer, choosing the best algorithm and visualization approach can be challenging. The EGSEA package solves this problem by combining results from up to 12 prominent gene set testing algorithms to obtain a consensus ranking of biologically relevant results.This workflow demonstrates how EGSEA can extend limma-based differential expression analyses for RNA-seq and microarray data using experiments that profile 3 distinct cell populations important for studying the origins of breast cancer. Following data normalization and set-up of an appropriate linear model for differential expression analysis, EGSEA builds gene signature specific indexes that link a wide range of mouse or human gene set collections obtained from MSigDB, GeneSetDB and KEGG to the gene expression data being investigated. EGSEA is then configured and the ensemble enrichment analysis run, returning an object that can be queried using several S4 methods for ranking gene sets and visualizing results via heatmaps, KEGG pathway views, GO graphs, scatter plots and bar plots. Finally, an HTML report that combines these displays can fast-track the sharing of results with collaborators, and thus expedite downstream biological validation. EGSEA is simple to use and can be easily integrated with existing gene expression analysis pipelines for both human and mouse data." } { "@context": "http://schema.org", "@type": "BreadcrumbList", "itemListElement": [ { "@type": "ListItem", "position": "1", "item": { "@id": "https://f1000research.com/", "name": "Home" } }, { "@type": "ListItem", "position": "2", "item": { "@id": "https://f1000research.com/browse/articles", "name": "Browse" } }, { "@type": "ListItem", "position": "3", "item": { "@id": "https://f1000research.com/articles/6-2010/v1", "name": "Easy and efficient ensemble gene set testing with EGSEA" } } ] } Home Browse Easy and efficient ensemble gene set testing with EGSEA ALL Metrics - Views Downloads Get PDF Get XML Cite How to cite this article Alhamdoosh M, Law CW, Tian L et al. Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.12688/f1000research.12544.1 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. Close Copy Citation Details Export Export Citation Sciwheel EndNote Ref. Manager Bibtex ProCite Sente EXPORT Select a format first Track Share ▬ ✚ Software Tool Article Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] Monther Alhamdoosh https://orcid.org/0000-0002-2411-1325 1 , Charity W. Law https://orcid.org/0000-0001-6082-6814 2,3 , Luyi Tian https://orcid.org/0000-0003-3420-3685 2,3 , Julie M. Sheridan https://orcid.org/0000-0002-1478-3662 2,4 , Milica Ng 1 , Matthew E. Ritchie https://orcid.org/0000-0002-7383-0609 2,3,5 Monther Alhamdoosh https://orcid.org/0000-0002-2411-1325 1 , Charity W. Law https://orcid.org/0000-0001-6082-6814 2,3 , [...] Luyi Tian https://orcid.org/0000-0003-3420-3685 2,3 , Julie M. Sheridan https://orcid.org/0000-0002-1478-3662 2,4 , Milica Ng 1 , Matthew E. Ritchie https://orcid.org/0000-0002-7383-0609 2,3,5 PUBLISHED 14 Nov 2017 Author details Author details 1 CSL Limited, Bio21 Institute, Parkville, Victoria, Australia 2 Department of Medical Biology, The University of Melbourne, Parkville, Victoria, Australia 3 Molecular Medicine Division, The Walter and Eliza Hall Institute of Medical Research, Parkville, Victoria, Australia 4 Molecular Genetics of Cancer Division, The Walter and Eliza Hall Institute of Medical Research, Parkville, Victoria, Australia 5 School of Mathematics and Statistics, The University of Melbourne, Parkville, Victoria, Australia Monther Alhamdoosh Roles: Conceptualization, Software, Visualization, Writing – Original Draft Preparation, Writing – Review & Editing Charity W. Law Roles: Software, Writing – Original Draft Preparation Luyi Tian Roles: Software, Writing – Review & Editing Julie M. Sheridan Roles: Investigation, Validation, Writing – Original Draft Preparation Milica Ng Roles: Software, Supervision, Writing – Review & Editing Matthew E. Ritchie Roles: Conceptualization, Software, Supervision, Writing – Original Draft Preparation, Writing – Review & Editing OPEN PEER REVIEW DETAILS REVIEWER STATUS This article is included in the Bioinformatics gateway. This article is included in the Bioconductor gateway. Abstract Gene set enrichment analysis is a popular approach for prioritising the biological processes perturbed in genomic datasets. The Bioconductor project hosts over 80 software packages capable of gene set analysis. Most of these packages search for enriched signatures amongst differentially regulated genes to reveal higher level biological themes that may be missed when focusing only on evidence from individual genes. With so many different methods on offer, choosing the best algorithm and visualization approach can be challenging. The EGSEA package solves this problem by combining results from up to 12 prominent gene set testing algorithms to obtain a consensus ranking of biologically relevant results.This workflow demonstrates how EGSEA can extend limma-based differential expression analyses for RNA-seq and microarray data using experiments that profile 3 distinct cell populations important for studying the origins of breast cancer. Following data normalization and set-up of an appropriate linear model for differential expression analysis, EGSEA builds gene signature specific indexes that link a wide range of mouse or human gene set collections obtained from MSigDB, GeneSetDB and KEGG to the gene expression data being investigated. EGSEA is then configured and the ensemble enrichment analysis run, returning an object that can be queried using several S4 methods for ranking gene sets and visualizing results via heatmaps, KEGG pathway views, GO graphs, scatter plots and bar plots. Finally, an HTML report that combines these displays can fast-track the sharing of results with collaborators, and thus expedite downstream biological validation. EGSEA is simple to use and can be easily integrated with existing gene expression analysis pipelines for both human and mouse data. READ ALL READ LESS Keywords gene expression, RNA-sequencing, microarrays, gene set enrichment, Bioconductor Corresponding Author(s) Monther Alhamdoosh ( [email protected] ) Matthew E. Ritchie ( [email protected] ) Close Corresponding authors: Monther Alhamdoosh, Matthew E. Ritchie Competing interests: MA and MN are employees of CSL Limited. Grant information: This work was funded by a National Health and Medical Research Council (NHMRC) Fellowship to MER (GNT1104924), Victorian State Government Operational Infrastructure Support and Australian Government NHMRC IRIISS. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript. Copyright: © 2017 Alhamdoosh M et al . This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. How to cite: Alhamdoosh M, Law CW, Tian L et al. Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.12688/f1000research.12544.1 ) First published: 14 Nov 2017, 6 :2010 ( https://doi.org/10.12688/f1000research.12544.1 ) Latest published: 14 Nov 2017, 6 :2010 ( https://doi.org/10.12688/f1000research.12544.1 ) Introduction Gene set enrichment analysis allows researchers to efficiently extract biological insights from long lists of differentially expressed genes by interrogating them at a systems level. In recent years, there has been a proliferation of gene set enrichment (GSE) analysis methods released through the Bioconductor project 1 together with a steady increase in the number of gene set collections available through online databases such as MSigDB 2 , GeneSetDB 3 and KEGG 4 . In an effort to unify these computational methods and knowledge-bases, the EGSEA R/Bioconductor package was developed. EGSEA, which stands for Ensemble of Gene Set Enrichment Analyses 5 combines the results from multiple algorithms to arrive at a consensus gene set ranking to identify biological themes and pathways perturbed in an experiment. EGSEA calculates seven statistics to combine the individual gene set statistics of base GSE methods to rank biologically relevant gene sets. The current version of the EGSEA package 6 utilizes the analysis results of up to twelve prominent GSE algorithms that include: ora 7 , globaltest 8 , plage 9 , safe 10 , zscore 11 , gage 12 , ssgsea 13 , padog 14 , gsva 15 , camera 16 , roast 17 and fry 17 . The ora , gage , camera and gsva methods depend on a competitive null hypothesis which assumes the genes in a set do not have a stronger association with the experimental condition compared to randomly chosen genes outside the set. The remaining eight methods are based on a self-contained null hypothesis that only considers genes within a set and again assumes that they have no association with the experimental condition. EGSEA provides access to a diverse range of gene signature collections through the EGSEAdata package that includes more than 25,000 gene sets for human and mouse organised according to their database sources ( Table 1 ). For example, MSigDB 2 includes a number of collections (Hallmark (h) and c1–c7) that explore different biological themes ranging from very broad (h, c2, c5) through to more specialised ones focusing on cancer (c4, c6) and immunology (c7). The other main sources are GeneSetDB 3 and KEGG 4 which have similar collections focusing on different biological characteristics ( Table 1 ). The choice of collection/s in any given analysis should of course be guided by the biological question of interest. The MSigDB c2 and c5 collections are the most widely used in our own analysis practice, spanning a wide range of biological processes and can often reveal new biological insights when applied to a given dataset. Table 1. Summary of the gene set collections available in the EGSEAdata package. Database Collection Description MSigDB h Hallmarks c1 Positional c2 Curated c3 Motif c4 Computational c5 GO c6 Oncogenic c7 Immunologic Gene sets representing well-defined biological states or processes that have coherent expression. Gene sets by chromosome and cytogenetic band. Gene sets obtained from a variety of sources, including online pathway databases and the biomedical literature. Gene sets of potential targets regulated by transcription factors or microRNAs. Gene sets defined computationally by mining large collections of cancer-oriented microarray data. Gene sets annotated by Gene Ontology (GO) terms. Gene sets of the major cellular pathways disrupted in cancer. Gene sets representing different cell types and stimulations relevant to the immune system. KEGG Signalling Disease Metabolic Gene sets obtained from the KEGG database. GeneSetDB Pathway Disease Drug Regulation GO Terms Gene sets obtained from various online databases. The purpose of this article is to demonstrate the gene set testing workflow available in EGSEA on both RNA-seq and microarray data. Each analysis involves four major steps that are summarized in Figure 1 : (1) selecting appropriate gene set collections for analysis and building an index that maps between the members of each set and the expression matrix; (2) choosing the base GSE methods to combine and the ranking options; (3) running the EGSEA test and (4) reporting results in various ways to share with collaborators. The EGSEA functions involved in each of these steps are introduced with code examples to demonstrate how they can be deployed as part of a limma differential expression analysis to help with the interpretation of results. Figure 1. The main steps in an EGSEA analysis and the functions that perform each task. Gene expression profiling of the mouse mammary gland The first experiment analysed in this workflow is an RNA-seq dataset from Sheridan et al. (2015) 18 that consists of 3 cell populations (Basal, Luminal Progenitor (LP) and Mature Luminal (ML)) sorted from the mammary glands of female virgin mice. Triplicate RNA samples from each population were obtained in 3 batches and sequenced on an Illumina HiSeq 2000 using a 100 base-pair single-ended protocol. Raw sequence reads from the fastq files were aligned to the mouse reference genome (mm10) using the Rsubread package 19 . Next, gene-level counts were obtained using featureCounts 20 based on Rsubread’s built-in mm10 RefSeq-based annotation. The raw data along with further information on experimental design and sample preparation can be downloaded from the Gene Expression Omnibus (GEO, www.ncbi.nlm.nih.gov/geo/ ) using GEO Series accession number GSE63310 and will be preprocessed according to the RNA-seq workflow published by Law et al. (2016) 21 . The second experiment analysed in this workflow comes from Lim et al. (2010) 22 and is the microarray equivalent of the RNA-seq dataset mentioned above. The same 3 populations (Basal (also referred to as “MaSC-enriched”), LP and ML) were sorted from mouse mammary glands via flow cytometry. Total RNA from 5 replicates of each cell population were hybridised onto 3 Illumina MouseWG-6 v2 BeadChips. The intensity files and chip annotation file available in Illumina’s proprietary formats (IDAT and BGX respectively) can be downloaded from http://bioinf.wehi.edu.au/EGSEA/arraydata.zip . The raw data from this experiment is also available from GEO under Series accession number GSE19446. Analysis of RNA-seq data with EGSEA Our RNA-seq analysis follows on directly from the workflow of Law et al. (2016) which performs a differential gene expression analysis on this data set using the Bioconductor packages edgeR 23 , limma 24 and Glimma 25 with gene annotation from the Mus.musculus package 26 . The limma package offers a well-developed suite of statistical methods for dealing with differential expression for both microarray and RNA-seq datasets and will be used in the analyses of both datasets presented in this workflow. Reading, preprocessing and normalisation of RNA-seq data To get started with this analysis, download the R data file from http://bioinf.wehi.edu.au/EGSEA/mam.rnaseq.rdata . The code below loads the preprocessed count matrix from Law et al. (2016), performs TMM normalisation 27 on the raw counts, and calculates voom weights for use in comparisons of gene expression between Basal and LP, Basal and ML, and LP and ML populations. > library(limma) > library(edgeR) > load("mam.rnaseq.rdata") > names(mam.rnaseq.data) [1] "samples" "counts" "genes" > dim(mam.rnaseq.data) [1] 14165 9 > x = calcNormFactors(mam.rnaseq.data, method = "TMM") > design = model.matrix(~0+x$samples$group+x$samples$lane) > colnames(design) = gsub("x\\$samples\\$group", "", colnames(design)) > colnames(design) = gsub("x\\$samples\\$lane", "", colnames(design)) > head(design) Basal LP ML L006 L008 1 0 1 0 0 0 2 0 0 1 0 0 3 1 0 0 0 0 4 1 0 0 1 0 5 0 0 1 1 0 6 0 1 0 1 0 > contr.matrix = makeContrasts( + BasalvsLP = Basal-LP, + BasalvsML = Basal - ML, + LPvsML = LP - ML, + levels = colnames(design)) > head(contr.matrix) Contrasts Levels BasalvsLP BasalvsML LPvsML Basal 1 1 0 LP -1 0 1 ML 0 -1 -1 L006 0 0 0 L008 0 0 0 The voom function 28 from the limma package converts counts to log-counts-per-million (log-cpm) and calculates observation-level precision weights. The voom object ( v ) contains normalized log-cpm values and gene information used by all of the methods in the EGSEA analysis below. The precision weights stored within v are also used by the camera , roast and fry gene set testing methods. > v = voom(x, design, plot=FALSE) > names(v) [1] "genes" "targets" "E" "weights" "design" For further information on preprocessing see Law et al. (2016), as a detailed explanation of these steps is beyond the scope of this article. Gene set testing The EGSEA algorithm makes use of the voom object ( v ), a design matrix ( design ) and an optional contrasts matrix ( contr.matrix ). The design matrix describes how the samples in the experiment relate to the coefficients estimated by the linear model 29 . The contrasts matrix then compares two or more of these coefficients to allow relative assessment of differential expression. Base methods that utilize linear models such as those from limma and GSVA ( gsva , plage , zscore and ssgsea ) make use of the design and contrasts matrices directly. For methods that do not support linear models, these two matrices are used to extract the group information for each comparison. 1. Exploring, selecting and indexing gene set collections The package EGSEAdata includes more than 25,000 gene sets organized in collections depending on their database sources. Summary information about the gene set collections available in EGSEAdata can be displayed as follows: > library(EGSEAdata) > egsea.data("mouse") The following databases are available in EGSEAdata for Mus musculus: Database name: KEGG Pathways Version: NA Download/update date: 07 March 2017 Data source: gage::kegg.gsets() Supported species: human, mouse, rat Gene set collections: Signaling, Metabolism, Disease Related data objects: kegg.pathways Number of gene sets in each collection for Mus musculus : Signaling: 132 Metabolism: 89 Disease: 67 Database name: Molecular Signatures Database (MSigDB) Version: 5.2 Download/update date: 07 March 2017 Data source: http://software.broadinstitute.org/gsea Supported species: human, mouse Gene set collections: h, c1, c2, c3, c4, c5, c6, c7 Related data objects: msigdb, Mm.H, Mm.c2, Mm.c3, Mm.c4, Mm.c5, Mm.c6, Mm.c7 Number of gene sets in each collection for Mus musculus : h Hallmark Signatures: 50 c2 Curated Gene Sets: 4729 c3 Motif Gene Sets: 836 c4 Computational Gene Sets: 858 c5 GO Gene Sets: 6166 c6 Oncogenic Signatures: 189 c7 Immunologic Signatures: 4872 Database name: GeneSetDB Database Version: NA Download/update date: 15 January 2016 Data source: http://www.genesetdb.auckland.ac.nz/ Supported species: human, mouse, rat Gene set collections: gsdbdis, gsdbgo, gsdbdrug, gsdbpath, gsdbreg Related data objects: gsetdb.human, gsetdb.mouse, gsetdb.rat Number of gene sets in each collection for Mus musculus : GeneSetDB Drug/Chemical: 6019 GeneSetDB Disease/Phenotype: 5077 GeneSetDB Gene Ontology: 2202 GeneSetDB Pathway: 1444 GeneSetDB Gene Regulation: 201 Type ? to get a specific information about it, e.g., ?kegg.pathways. As the output above suggests, users can obtain help on any of the collections using the standard R help ( ? ) command, for instance ?Mm.c2 will return more information on the mouse version of the c2 collection from MSigDB. The above information can be returned as a list: > info = egsea.data("mouse", returnInfo = TRUE) > names(info) [1] "kegg" "msigdb" "gsetdb" > info$msigdb$info$collections [1] "h" "c1" "c2" "c3" "c4" "c5" "c6" "c7" To highlight the capabilities of the EGSEA package, the KEGG pathways, c2 (curated gene sets) and c5 (Gene Ontology gene sets) collections from the MSigDB database are selected. Next, an index is built for each gene set collection using the EGSEA indexing functions to link the genes in the different gene set collections to the rows of our RNA-seq gene expression matrix. Indexes for the c2 and c5 collections from MSigDB and for the KEGG pathways are built using the buildIdx function which relies on Entrez gene IDs as its key. In the EGSEAdata gene set collections, Entrez IDs are used as they are widely adopted by the different source databases and tend to be more consistent and robust since there is one identifier per gene in a gene set. It is also relatively easy to convert other gene IDs into Entrez IDs. > library(EGSEA) > gs.annots = buildIdx(entrezIDs=v$genes$ENTREZID, species="mouse", + msigdb.gsets=c("c2", "c5"), go.part = TRUE) [1] "Loading MSigDB Gene Sets ... " [1] "Loaded gene sets for the collection c2 ..." [1] "Indexed the collection c2 ..." [1] "Created annotation for the collection c2 ..." [1] "Loaded gene sets for the collection c5 ..." [1] "Indexed the collection c5 ..." [1] "Created annotation for the collection c5 ..." MSigDB c5 gene set collection has been partitioned into c5BP, c5CC, c5MF [1] "Building KEGG pathways annotation object ... " > names(gs.annots) [1] "c2" "c5BP" "c5CC" "c5MF" "kegg" To obtain additional information on the gene set collection indexes, including the total number of gene sets, the version number and date of last revision, the methods summary , show and getSetByName (or getSetByID ) can be invoked on an object of class GSCollectionIndex , which stores all of the relevant gene set information, as follows: > class(gs.annots$c2) [1] "GSCollectionIndex" attr(,"package") [1] "EGSEA" > summary(gs.annots$c2) c2 Curated Gene Sets (c2): 4726 gene sets - Version: 5.2, Update date: 07 March 2017 > show(gs.annots$c2) An object of class "GSCollectionIndex" Number of gene sets: 4726 Annotation columns: ID, GeneSet, BroadUrl, Description, PubMedID, NumGenes, Contributor Total number of indexing genes: 14165 Species: Mus musculus Collection name: c2 Curated Gene Sets Collection unique label: c2 Database version: 5.2 Database update date: 07 March 2017 > s = getSetByName(gs.annots$c2, "SMID_BREAST_CANCER_LUMINAL_A_DN") ID: M13072 GeneSet: SMID_BREAST_CANCER_LUMINAL_A_DN BroadUrl: http://www.broadinstitute.org/gsea/msigdb/cards/SMID_BREAST_CANCER_LUMINAL_A_DN.html Description: Genes down-regulated in the luminal A subtype of breast cancer. PubMedID: 18451135 NumGenes: 23/24 Contributor: Jessica Robertson > class(s) [1] "list" > names(s) [1] "SMID_BREAST_CANCER_LUMINAL_A_DN" > names(s$SMID_BREAST_CANCER_LUMINAL_A_DN) [1] "ID" "GeneSet" "BroadUrl" "Description" "PubMedID" [6] "NumGenes" "Contributor" Objects of class GSCollectionIndex store for each gene set the Entrez gene IDs in the slot original , the indexes in the slot idx and additional annotation for each set in the slot anno . > slotNames(gs.annots$c2) [1] "original" "idx" "anno" "featureIDs" "species" [6] "name" "label" "version" "date" Other EGSEA functions such as buildCustomIdx , buildGMTIdx , buildKEGGIdx , buildMSigDBIdx and buildGeneSetDBIdx can be also used to build gene set collection indexes. The functions buildCustomIdx and buildGMTIdx were written to allow users to run EGSEA on gene set collections that may have been curated within a lab or downloaded from public databases and allow use of gene identifiers other than Entrez IDs. Example databases include, ENCODE Gene Set Hub (available from https://sourceforge.net/projects/encodegenesethub/ ), which is a growing resource of gene sets derived from high quality ENCODE profiling experiments encompassing hundreds of DNase hypersensitivity, histone modification and transcription factor binding experiments 30 . Other resources include PathwayCommons ( http://www.pathwaycommons.org/ ) 31 and the KEGGREST 32 package that provides access to up-to-date KEGG pathways across many species. 2. Configuring EGSEA Before an EGSEA test is carried out, a few parameters need to be specified. First, a mapping between Entrez IDs and Gene Symbols is created for use by the visualization procedures. This mapping can be extracted from the genes data.frame of the voom object as follows: > colnames(v$genes) [1] "ENTREZID" "SYMBOL" "CHR" > symbolsMap = v$genes[, c(1, 2)] > colnames(symbolsMap) = c("FeatureID", "Symbols") > symbolsMap[, "Symbols"] = as.character(symbolsMap[, "Symbols"]) Another important parameter in EGSEA is the list of base GSE methods ( baseMethods in the code below), which determines the individual algorithms that are used in the ensemble testing. The supported base methods can be listed using the function egsea.base as follows: > egsea.base() [1] "camera" "roast" "safe" "gage" "padog" "plage" [7] "zscore" "gsva" "ssgsea" "globaltest" "ora" "fry" The plage , zscore and ssgsea algorithms are available in the GSVA package and camera , fry and roast are implemented in the limma package 24 . The ora method is implemented using the phyper function from the stats package 33 , which estimates the hypergeometric distribution for a 2 × 2 contingency table. The remaining algorithms are implemented in Bioconductor packages of the same name. A wrapper function is provided for each individual GSE method to utilize this existing R code and create a universal interface for all methods. Eleven base methods are selected for our EGSEA analysis: camera , safe , gage , padog , plage , zscore , gsva , ssgsea , globaltest , ora and fry . Fry is a fast approximation of roast that assumes equal gene-wise variances across samples to produce similar p -values to a roast analysis run with an infinite number of rotations, and is selected here to save time. > baseMethods = egsea.base()[-2] > baseMethods [1] "camera" "safe" "gage" "padog" "plage" "zscore" [7] "gsva" "ssgsea" "globaltest" "ora" "fry" Although, different combinations of base methods might produce different results, it has been found via simulation that including more methods gives better performance 5 . Since each base method generates different p -values, EGSEA supports six different methods from the metap package 34 for combining individual p -values ( Wilkinson 35 is default), which can be listed as follows: > egsea.combine() [1] "fisher" "wilkinson" "average" "logitp" "sump" "sumz" [7] "votep" "median" Finally, the sorting of EGSEA results plays an essential role in identifying relevant gene sets. Any of EGSEA’s combined scores or the rankings from individual base methods can be used for sorting the results. > egsea.sort() [1] "p.value" "p.adj" "vote.rank" "avg.rank" "med.rank" [6] "min.pvalue" "min.rank" "avg.logfc" "avg.logfc.dir" "direction" [11] "significance" "camera" "roast" "safe" "gage" [16] "padog" "plage" "zscore" "gsva" "ssgsea" [21] "globaltest" "ora" "fry" Although p.adj is the default option for sorting EGSEA results for convenience, we recommend the use of either med.rank or vote.rank because they efficiently utilize the rankings of individual methods and tend to produce fewer false positives 5 . 3. Ensemble testing with EGSEA Next, the EGSEA analysis is performed using the egsea function that takes a voom object, a contrasts matrix, collections of gene sets and other run parameters as follows: > gsa = egsea(voom.results=v, contrasts=contr.matrix, + gs.annots=gs.annots, symbolsMap=symbolsMap, + baseGSEAs=baseMethods, sort.by="med.rank", + num.threads = 8, report = FALSE) EGSEA analysis has started ##------ Fri Jun 16 09:49:11 2017 ------## Log fold changes are estimated using limma package ... limma DE analysis is carried out ... Number of used cores has changed to 3 in order to avoid CPU overloading. EGSEA is running on the provided data and c2 collection EGSEA is running on the provided data and c5BP collection EGSEA is running on the provided data and c5CC collection EGSEA is running on the provided data and c5MF collection EGSEA is running on the provided data and kegg collection ##------ Fri Jun 16 09:57:56 2017 ------## EGSEA analysis took 525.812 seconds. EGSEA analysis has completed In situations where the design matrix includes an intercept, a vector of integers that specify the columns of the design matrix to test using EGSEA can be passed to the contrasts argument. If this parameter is NULL , all pairwise comparisons based on v$targets$group are created, assuming that group is the primary factor in the design matrix. Likewise, all the coefficients of the primary factor are used if the design matrix has an intercept. EGSEA is implemented with parallel computing features enabled using the parallel package 33 at both the method-level and experimental contrast-level. The running time of the EGSEA test depends on the base methods selected and whether report generation is enabled or not. The latter significantly increases the run time, particularly if the argument display.top is assigned a large value (> 20) and/or a large number of gene set collections are selected. EGSEA reporting functionality generates set-level plots for the top gene sets as well as collection-level plots. The EGSEA package also has a function named egsea.cnt , that can perform the EGSEA test using an RNA-seq count matrix rather than a voom object, a function named egsea.ora , that can perform over-representation analysis with EGSEA reporting capabilities using only a vector of gene IDs, and the egsea.ma function that can perform EGSEA testing using a microarray expression matrix as shown later in the workflow. Classes used to manage the results. The output of the functions egsea , egsea.cnt , egsea.ora and egsea.ma is an S4 object of class EGSEAResults . Several S4 methods can be invoked to query this object. For example, an overview of the EGSEA analysis can be displayed using the show method as follows: > show(gsa) An object of class "EGSEAResults" Total number of genes: 14165 Total number of samples: 9 Contrasts: BasalvsLP, BasalvsML, LPvsML Base GSE methods: camera (limma:3.32.2), safe (safe:3.16.0), gage (gage:2.26.0), padog (PADOG:1.18.0), plage (GSVA:1.24.1), zscore (GSVA:1.24.1), gsva (GSVA:1.24.1), ssgsea (GSVA:1.24.1), P-values combining method: wilkinson Sorting statistic: med.rank Organism: Mus musculus HTML report generated: No Tested gene set collections: c2 Curated Gene Sets (c2): 4726 gene sets - Version: 5.2, Update date: 07 March 2017 c5 GO Gene Sets (BP) (c5BP): 4653 gene sets - Version: 5.2, Update date: 07 March 2017 c5 GO Gene Sets (CC) (c5CC): 584 gene sets - Version: 5.2, Update date: 07 March 2017 c5 GO Gene Sets (MF) (c5MF): 928 gene sets - Version: 5.2, Update date: 07 March 2017 KEGG Pathways (kegg): 287 gene sets - Version: NA, Update date: 07 March 2017 EGSEA version: 1.5.2 EGSEAdata version: 1.4.0 Use summary(object) and topSets(object, ...) to explore this object. This command displays the number of genes and samples that were included in the analysis, the experimental contrasts, base GSE methods, the method used to combine the p -values derived from different GSE algorithms, the sorting statistic used and the size of each gene set collection. Note that the gene set collections are identified using the labels that appear in parentheses (e.g. c2 ) in the output of show . 4. Reporting EGSEA results Getting top ranked gene sets. A summary of the top 10 gene sets in each collection for each contrast in addition to the EGSEA comparative analysis can be displayed using the S4 method summary as follows: > summary(gsa) **** Top 10 gene sets in the c2 Curated Gene Sets collection **** ** Contrast BasalvsLP ** LIM_MAMMARY_STEM_CELL_DN | LIM_MAMMARY_LUMINAL_PROGENITOR_UP MONTERO_THYROID_CANCER_POOR_SURVIVAL_UP | SMID_BREAST_CANCER_LUMINAL_A_DN NAKAYAMA_SOFT_TISSUE_TUMORS_PCA2_UP | REACTOME_LATENT_INFECTION_OF_HOMO_SAPIENS... REACTOME_TRANSFERRIN_ENDOCYTOSIS_AND_RECYCLING | FARMER_BREAST_CANCER_CLUSTER_2 KEGG_EPITHELIAL_CELL_SIGNALING_... | LANDIS_BREAST_CANCER_PROGRESSION_UP ** Contrast BasalvsML ** LIM_MAMMARY_STEM_CELL_DN | LIM_MAMMARY_STEM_CELL_UP LIM_MAMMARY_LUMINAL_MATURE_DN | PAPASPYRIDONOS_UNSTABLE_ATEROSCLEROTIC_PLAQUE_DN NAKAYAMA_SOFT_TISSUE_TUMORS_PCA2_UP | LIM_MAMMARY_LUMINAL_MATURE_UP CHARAFE_BREAST_CANCER_LUMINAL_VS_MESENCHYMAL_UP | RICKMAN_HEAD_AND_NECK_CANCER_A YAGUE_PRETUMOR_DRUG_RESISTANCE_DN | BERTUCCI_MEDULLARY_VS_DUCTAL_BREAST_CANCER_DN ** Contrast LPvsML ** LIM_MAMMARY_LUMINAL_MATURE_UP | LIM_MAMMARY_LUMINAL_MATURE_DN PHONG_TNF_RESPONSE_VIA_P38_PARTIAL | WOTTON_RUNX_TARGETS_UP WANG_MLL_TARGETS | PHONG_TNF_TARGETS_DN REACTOME_PEPTIDE_LIGAND_BINDING_RECEPTORS | CHIANG_LIVER_CANCER_SUBCLASS_CTNNB1_DN GERHOLD_RESPONSE_TO_TZD_DN | DURAND_STROMA_S_UP ** Comparison analysis ** LIM_MAMMARY_LUMINAL_MATURE_DN | LIM_MAMMARY_STEM_CELL_DN NAKAYAMA_SOFT_TISSUE_TUMORS_PCA2_UP | LIM_MAMMARY_LUMINAL_MATURE_UP COLDREN_GEFITINIB_RESISTANCE_DN | LIM_MAMMARY_STEM_CELL_UP CHARAFE_BREAST_CANCER_LUMINAL_VS_MESENCHYMAL_UP | LIM_MAMMARY_LUMINAL_PROGENITOR_UP BERTUCCI_MEDULLARY_VS_DUCTAL_BREAST_CANCER_DN | MIKKELSEN_IPS_WITH_HCP_H3K27ME3 **** Top 10 gene sets in the c5 GO Gene Sets (BP) collection **** ** Contrast BasalvsLP ** GO_SYNAPSE_ORGANIZATION | GO_IRON_ION_TRANSPORT GO_CALCIUM_INDEPENDENT_CELL_CELL_ADHESION_VIA_PLASMA_MEMBRANE_CELL_ADHESION_MOLECULES | GO_PH_REDUCTION GO_HOMOPHILIC_CELL_ADHESION_VIA_PLASMA_MEMBRANE_ADHESION_MOLECULES | GO_VACUOLAR_ACIDIFICATION GO_FERRIC_IRON_TRANSPORT | GO_TRIVALENT_INORGANIC_CATION_TRANSPORT GO_NEURON_PROJECTION_GUIDANCE | GO_MESONEPHROS_DEVELOPMENT ** Contrast BasalvsML ** GO_FERRIC_IRON_TRANSPORT | GO_TRIVALENT_INORGANIC_CATION_TRANSPORT GO_IRON_ION_TRANSPORT | GO_NEURON_PROJECTION_GUIDANCE GO_GLIAL_CELL_MIGRATION | GO_SPINAL_CORD_DEVELOPMENT GO_REGULATION_OF_SYNAPSE_ORGANIZATION | GO_ACTION_POTENTIAL GO_MESONEPHROS_DEVELOPMENT | GO_NEGATIVE_REGULATION_OF_SMOOTH_MUSCLE_CELL_MIGRATION ** Contrast LPvsML ** GO_NEGATIVE_REGULATION_OF_NECROTIC_CELL_DEATH | GO_PARTURITION GO_RESPONSE_TO_VITAMIN_D | GO_GPI_ANCHOR_METABOLIC_PROCESS GO_REGULATION_OF_BLOOD_PRESSURE | GO_DETECTION_OF_MOLECULE_OF_BACTERIAL_ORIGIN GO_CELL_SUBSTRATE_ADHESION | GO_PROTEIN_TRANSPORT_ALONG_MICROTUBULE GO_INTRACILIARY_TRANSPORT | GO_CELLULAR_RESPONSE_TO_VITAMIN ** Comparison analysis ** GO_IRON_ION_TRANSPORT | GO_FERRIC_IRON_TRANSPORT GO_TRIVALENT_INORGANIC_CATION_TRANSPORT | GO_NEURON_PROJECTION_GUIDANCE GO_MESONEPHROS_DEVELOPMENT | GO_SYNAPSE_ORGANIZATION GO_REGULATION_OF_SYNAPSE_ORGANIZATION | GO_MEMBRANE_DEPOLARIZATION_DURING_CARDIAC_MUSCLE_CELL_ACTION_POTENTIAL GO_HOMOPHILIC_CELL_ADHESION_VIA_PLASMA_MEMBRANE_ADHESION_MOLECULES | GO_NEGATIVE_REGULATION_OF_SMOOTH_MUSCLE_CELL_MIGRATION **** Top 10 gene sets in the c5 GO Gene Sets (CC) collection **** ** Contrast BasalvsLP ** GO_PROTON_TRANSPORTING_V_TYPE_ATPASE_COMPLEX | GO_VACUOLAR_PROTON_TRANSPORTING_V_TYPE_ATPASE_COMPLEX GO_MICROTUBULE_END | GO_MICROTUBULE_PLUS_END GO_ACTIN_FILAMENT_BUNDLE | GO_CELL_CELL_ADHERENS_JUNCTION GO_NEUROMUSCULAR_JUNCTION | GO_AP_TYPE_MEMBRANE_COAT_ADAPTOR_COMPLEX GO_INTERMEDIATE_FILAMENT | GO_CONDENSED_NUCLEAR_CHROMOSOME_CENTROMERIC_REGION ** Contrast BasalvsML ** GO_FILOPODIUM_MEMBRANE | GO_LATE_ENDOSOME_MEMBRANE GO_PROTON_TRANSPORTING_V_TYPE_ATPASE_COMPLEX | GO_NEUROMUSCULAR_JUNCTION GO_COATED_MEMBRANE | GO_ACTIN_FILAMENT_BUNDLE GO_CLATHRIN_COAT | GO_AP_TYPE_MEMBRANE_COAT_ADAPTOR_COMPLEX GO_CLATHRIN_ADAPTOR_COMPLEX | GO_CONTRACTILE_FIBER ** Contrast LPvsML ** GO_CILIARY_TRANSITION_ZONE | GO_TCTN_B9D_COMPLEX GO_NUCLEAR_NUCLEOSOME | GO_INTRINSIC_COMPONENT_OF_ORGANELLE_MEMBRANE GO_ENDOPLASMIC_RETICULUM_QUALITY_CONTROL_COMPARTMENT | GO_KERATIN_FILAMENT GO_PROTEASOME_COMPLEX | GO_CILIARY_BASAL_BODY GO_PROTEASOME_CORE_COMPLEX | GO_CORNIFIED_ENVELOPE ** Comparison analysis ** GO_PROTON_TRANSPORTING_V_TYPE_ATPASE_COMPLEX | GO_ACTIN_FILAMENT_BUNDLE GO_NEUROMUSCULAR_JUNCTION | GO_AP_TYPE_MEMBRANE_COAT_ADAPTOR_COMPLEX GO_CONTRACTILE_FIBER | GO_INTERMEDIATE_FILAMENT GO_LATE_ENDOSOME_MEMBRANE | GO_CLATHRIN_VESICLE_COAT GO_ENDOPLASMIC_RETICULUM_QUALITY_CONTROL_COMPARTMENT | GO_MICROTUBULE_END **** Top 10 gene sets in the c5 GO Gene Sets (MF) collection **** ** Contrast BasalvsLP ** GO_HYDROGEN_EXPORTING_ATPASE_ACTIVITY | GO_SIGNALING_PATTERN_RECOGNITION_RECEPTOR_ACTIVITY GO_LIPID_TRANSPORTER_ACTIVITY | GO_TRIGLYCERIDE_LIPASE_ACTIVITY GO_AMINE_BINDING | GO_STRUCTURAL_CONSTITUENT_OF_MUSCLE GO_NEUROPEPTIDE_RECEPTOR_ACTIVITY | GO_WIDE_PORE_CHANNEL_ACTIVITY GO_CATION_TRANSPORTING_ATPASE_ACTIVITY | GO_LIPASE_ACTIVITY ** Contrast BasalvsML ** GO_G_PROTEIN_COUPLED_RECEPTOR_ACTIVITY | GO_TRANSMEMBRANE_RECEPTOR_PROTEIN_KINASE_ACTIVITY GO_STRUCTURAL_CONSTITUENT_OF_MUSCLE | GO_VOLTAGE_GATED_SODIUM_CHANNEL_ACTIVITY GO_CORECEPTOR_ACTIVITY | GO_TRANSMEMBRANE_RECEPTOR_PROTEIN_TYROSINE_KINASE_ACTIVITY GO_LIPID_TRANSPORTER_ACTIVITY | GO_SULFOTRANSFERASE_ACTIVITY GO_CATION_TRANSPORTING_ATPASE_ACTIVITY | GO_PEPTIDE_RECEPTOR_ACTIVITY ** Contrast LPvsML ** GO_MANNOSE_BINDING | GO_PHOSPHORIC_DIESTER_HYDROLASE_ACTIVITY GO_BETA_1_3_GALACTOSYLTRANSFERASE_ACTIVITY | GO_COMPLEMENT_BINDING GO_ALDEHYDE_DEHYDROGENASE_NAD_ACTIVITY | GO_MANNOSIDASE_ACTIVITY GO_LIGASE_ACTIVITY_FORMING_CARBON_NITROGEN_BONDS | GO_CARBOHYDRATE_PHOSPHATASE_ACTIVITY GO_LIPASE_ACTIVITY | GO_PEPTIDE_RECEPTOR_ACTIVITY ** Comparison analysis ** GO_STRUCTURAL_CONSTITUENT_OF_MUSCLE | GO_LIPID_TRANSPORTER_ACTIVITY GO_CATION_TRANSPORTING_ATPASE_ACTIVITY | GO_CHEMOREPELLENT_ACTIVITY GO_HEPARAN_SULFATE_PROTEOGLYCAN_BINDING | GO_TRANSMEMBRANE_RECEPTOR_PROTEIN_TYROSINE_KINASE_ACTIVITY GO_LIPASE_ACTIVITY | GO_PEPTIDE_RECEPTOR_ACTIVITY GO_CORECEPTOR_ACTIVITY | GO_TRANSMEMBRANE_RECEPTOR_PROTEIN_KINASE_ACTIVITY **** Top 10 gene sets in the KEGG Pathways collection **** ** Contrast BasalvsLP ** Collecting duct acid secretion | alpha-Linolenic acid metabolism Synaptic vesicle cycle | Hepatitis C Vascular smooth muscle contraction | Rheumatoid arthritis cGMP-PKG signaling pathway | Axon guidance Progesterone-mediated oocyte maturation | Arrhythmogenic right ventricular cardiomyopathy (ARVC) ** Contrast BasalvsML ** Collecting duct acid secretion | Synaptic vesicle cycle Other glycan degradation | Axon guidance Arrhythmogenic right ventricular cardiomyopathy (ARVC) | Glycerophospholipid metabolism Lysosome | Vascular smooth muscle contraction Protein digestion and absorption | Oxytocin signaling pathway ** Contrast LPvsML ** Glycosylphosphatidylinositol(GPI)-anchor biosynthesis | Histidine metabolism Drug metabolism - cytochrome P450 | PI3K-Akt signaling pathway Proteasome | Sulfur metabolism Renin-angiotensin system | Nitrogen metabolism Tyrosine metabolism | Systemic lupus erythematosus ** Comparison analysis ** Collecting duct acid secretion | Synaptic vesicle cycle Vascular smooth muscle contraction | Axon guidance Arrhythmogenic right ventricular cardiomyopathy (ARVC) | Oxytocin signaling pathway Lysosome | Adrenergic signaling in cardiomyocytes Linoleic acid metabolism | cGMP-PKG signaling pathway EGSEA’s comparative analysis allows researchers to estimate the significance of a gene set across multiple experimental contrasts. This analysis helps in the identification of biological processes that are perturbed in multiple experimental conditions simultaneously. This experiment is the RNA-seq equivalent of Lim et al. (2010) 22 , who used Illumina microarrays to study the same cell populations (see later), so it is reassuring to observe the LIM gene signatures derived from this experiment amongst the top ranked c2 gene signatures in both the individual contrasts and comparative results. Another way of exploring the EGSEA results is to retrieve the top ranked N sets in each collection and contrast using the method topSets . For example, the top 10 gene sets in the c2 collection for the comparative analysis can be retrieved as follows: > topSets(gsa, gs.label="c2", contrast = "comparison", names.only=TRUE) Extracting the top gene sets of the collection c2 Curated Gene Sets for the contrast comparison Sorted by med.rank [1] "LIM_MAMMARY_LUMINAL_MATURE_DN" [2] "LIM_MAMMARY_STEM_CELL_DN" [3] "NAKAYAMA_SOFT_TISSUE_TUMORS_PCA2_UP" [4] "LIM_MAMMARY_LUMINAL_MATURE_UP" [5] "COLDREN_GEFITINIB_RESISTANCE_DN" [6] "LIM_MAMMARY_STEM_CELL_UP" [7] "CHARAFE_BREAST_CANCER_LUMINAL_VS_MESENCHYMAL_UP" [8] "LIM_MAMMARY_LUMINAL_PROGENITOR_UP" [9] "BERTUCCI_MEDULLARY_VS_DUCTAL_BREAST_CANCER_DN" [10] "MIKKELSEN_IPS_WITH_HCP_H3K27ME3" The gene sets are ordered based on their med.rank as selected when egsea was invoked above. When the argument names.only is set to FALSE , additional information is displayed for each gene set including gene set annotation, the EGSEA scores and the individual rankings by each base method. As expected, gene sets retrieved by EGSEA included the LIM gene sets 22 that were derived from microarray profiles of analagous mammary cell populations (sets 1, 2, 4, 6 and 8) as well as those derived from populations with similar origin (sets 7 and 9) and behaviour or characteristics (sets 5 and 10). Next, topSets can be used to search for gene sets of interest based on different EGSEA scores as well as the rankings of individual methods. For example, the ranking of the six LIM gene sets from the c2 collection can be displayed based on the med.rank as follows: > t = topSets(gsa, contrast = "comparison", + names.only=FALSE, number = Inf, verbose = FALSE) > t[grep("LIM_", rownames(t)), c("p.adj", "Rank", "med.rank", "vote.rank")] p.adj Rank med.rank vote.rank LIM_MAMMARY_LUMINAL_MATURE_DN 1.646053e-29 1 36 5 LIM_MAMMARY_STEM_CELL_DN 6.082053e-43 2 37 5 LIM_MAMMARY_LUMINAL_MATURE_UP 2.469061e-22 4 92 5 LIM_MAMMARY_STEM_CELL_UP 3.154132e-103 6 134 5 LIM_MAMMARY_LUMINAL_PROGENITOR_UP 3.871536e-30 8 180 5 LIM_MAMMARY_LUMINAL_PROGENITOR_DN 2.033005e-06 178 636 115 While five of the LIM gene sets are ranked in the top 10 by EGSEA, the values shown in the median rank ( med.rank ) column indicate that individual methods can assign much lower ranks to these sets. EGSEA’s prioritisation of these gene sets demonstrates the benefit of an ensemble approach. Similarly, we can find the top 10 pathways in the KEGG collection from the ensemble analysis for the Basal versus LP contrast and the comparative analysis as follows: > topSets(gsa, gs.label="kegg", contrast="BasalvsLP", sort.by="med.rank") Extracting the top gene sets of the collection KEGG Pathways for the contrast BasalvsLP Sorted by med.rank [1] "Collecting duct acid secretion" "alpha-Linolenic acid metabolism" [3] "Synaptic vesicle cycle" "Hepatitis C" [5] "Vascular smooth muscle contraction" "Rheumatoid arthritis" [7] "cGMP-PKG signaling pathway" "Axon guidance" [9] "Progesterone-mediated oocyte maturation" "Arrhythmogenic right ventricular cardiomyopathy (ARVC)" > topSets(gsa, gs.label="kegg", contrast="comparison", sort.by="med.rank") Extracting the top gene sets of the collection KEGG Pathways for the contrast comparison Sorted by med.rank [1] "Collecting duct acid secretion" "Synaptic vesicle cycle" [3] "Vascular smooth muscle contraction" "Axon guidance" [5] "Arrhythmogenic right ventricular cardiomyopathy (ARVC)" "Oxytocin signaling pathway" [7] "Lysosome" "Adrenergic signaling in cardiomyocytes" [9] "Linoleic acid metabolism" "cGMP-PKG signaling pathway" EGSEA highlights many pathways with known importance in the mammary gland such as those associated with distinct roles in lactation like basal cell contraction ( Vascular smooth muscle contraction and Oxytocin signalling pathway ) and milk production and secretion from luminal lineage cells ( Collecting duct acid secretion, Synaptic vesicle cycle and Lysosome ). Visualizing results at the gene set level. Graphical representation of gene expression patterns within and between gene sets is an essential part of communicating the results of an analysis to collaborators and other researchers. EGSEA enables users to explore the elements of a gene set via a heatmap using the plotHeatmap method. Figure 2 shows examples for the LIM_MAMMARY_STEM_CELL_UP and LIM_MAMMARY_STEM_CELL_DN signatures which can be visualized across all contrasts using the code below. Figure 2. Heatmaps of log-fold-changes for genes in the LIM_MAMMARY_STEM_CELL_UP and LIM_MAMMARY_STEM_CELL_DN gene sets across the three experimental comparisons (Basal vs LP, Basal vs ML and LP vs ML). > plotHeatmap(gsa, gene.set="LIM_MAMMARY_STEM_CELL_UP", gs.label="c2", + contrast = "comparison", file.name = "hm_cmp_LIM_MAMMARY_STEM_CELL_UP") Generating heatmap for LIM_MAMMARY_STEM_CELL_UP from the collection c2 Curated Gene Sets and for the contrast comparison > plotHeatmap(gsa, gene.set="LIM_MAMMARY_STEM_CELL_DN", gs.label="c2", + contrast = "comparison", file.name = "hm_cmp_LIM_MAMMARY_STEM_CELL_DN") Generating heatmap for LIM_MAMMARY_STEM_CELL_DN from the collection c2 Curated Gene Sets and for the contrast comparison When using plotHeatmap , the gene.set value must match the name returned from the topSets method. The rows of the heatmap represent the genes in the set and the columns represent the experimental contrasts. The heatmap colour-scale ranges from down-regulated (blue) to up-regulated (red) while the row labels (Gene symbols) are coloured in green when the genes are statistically significant in the DE analysis (i.e. FDR ≤ 0.05 in at least one contrast). Heatmaps can be generated for individual comparisons by changing the contrast argument of plotHeatmap . The plotHeatmap method also generates a CSV file that includes the DE analysis results from limma::topTable for all expressed genes in the selected gene set and for each contrast (in the case of contrast = "comparison" ). This file can be used to create customised plots using other R/Bioconductor packages. In addition to heatmaps, pathway maps can be generated for the KEGG gene sets using the plotPathway method which uses functionality from the pathview package 36 . For example, the third KEGG signalling pathway retrieved for the contrast BasalvsLP is Vascular smooth muscle contraction and can be visualized as follows: > plotPathway(gsa, gene.set = "Vascular smooth muscle contraction", + contrast = "BasalvsLP", gs.label = "kegg", + file.name = "Vascular_smooth_muscle_contraction") Generating pathway map for Vascular smooth muscle contraction from the collection KEGG Pathways and for the contrast BasalvsLP Pathway components are coloured based on the gene-specific log-fold-changes as calculated in the limma DE analysis ( Figure 3 ). Similarly, a comparative map can be generated for a given pathway across all contrasts. Figure 3. Pathway map for Vascular smooth muscle contraction (KEGG pathway mmu04270) with log-fold-changes from the Basal vs LP contrast. > plotPathway(gsa, gene.set = "Vascular smooth muscle contraction", + contrast = "comparison", gs.label = "kegg", + file.name = "Vascular_smooth_muscle_contraction_cmp") Generating pathway map for Vascular smooth muscle contraction from the collection KEGG Pathways and for the contrast comparison The comparative pathway map shows the log-fold-changes for each gene in each contrast by dividing the gene nodes on the map into multiple columns, one for each contrast ( Figure 4 ). Figure 4. Pathway map for Vascular smooth muscle contraction (KEGG pathway mmu04270) with log-fold-changes across three experimental contrasts shown for each gene in the same order left to right that they appear in the contrasts matrix (i.e. Basal vs LP, Basal vs ML and LP vs ML). Visualizing results at the experiment level. Since EGSEA combines the results from multiple gene set testing methods, it can be interesting to compare how different base methods rank a given gene set collection for a selected contrast. The plotMethods command generates a multi-dimensional scaling (MDS) plot for the ranking of gene sets across all the base methods used ( Figure 5 ). Methods that rank gene sets similarly will appear closer together in this plot and we see that certain methods consistently cluster together across different gene set collections. The clustering of methods does not necessarily follow the style of null hypothesis tested though (i.e. self-contained versus competitive ). > plotMethods(gsa, gs.label = "c2", contrast = "BasalvsLP", + file.name = "mds_c2_BasalvsLP") Generating MDS plot for the collection c2 Curated Gene Sets and for the contrast BasalvsLP > plotMethods(gsa, gs.label = "c5BP", contrast = "BasalvsLP", + file.name = "mds_c5_BasalvsLP") Generating MDS plot for the collection c5BP GO Gene Sets and for the contrast BasalvsLP Figure 5. Multi-dimensional scaling (MDS) plot showing the relationship between different gene set testing methods based on the rankings of the c2 ( a ) and c5 ( b ) gene sets on the Basal vs LP contrast. The significance of each gene set in a given collection for a selected contrast can be visualized using EGSEA’s plotSummary method. > plotSummary(gsa, gs.label = 3, contrast = 3, + file.name = "summary_kegg_LPvsML") Generating Summary plots for the collection KEGG Pathways and for the contrast LPvsML The summary plot visualizes the gene sets as bubbles based on the − log 10 ( p - value ) (X-axis) and the average absolute log fold-change of the set genes (Y-axis). The sets that appear towards the top-right corner of this plot are most likely to be biologically relevant. EGSEA generates two types of summary plots: the directional summary plot ( Figure 6a ), which colours the bubbles based on the regulation direction of the gene set (the direction of the majority of genes), and the ranking summary plot ( Figure 6b ), which colours the bubbles based on the gene set ranking in a given collection (according to the sort.by argument). The bubble size is based on the EGSEA significance score in the former plot and the gene set size in the latter. For example, the summary plots of the KEGG pathways for the LP vs ML contrast show few significant pathways ( Figure 6 ). The blue colour labels on the ranking plot represents gene sets that do not appear in the top 10 gene sets that are selected based on the sort.by argument, yet their EGSEA significance scores are among the top 5 in the entire collection based on the significance score . This is used to identify gene sets with high significance scores that were not captured by the sort.by score. The gene set IDs and more information about each set can be found in the EGSEA HTML report generated later. Figure 6. Summary plots of the significance of all gene sets in the KEGG collection for the LP vs ML contrast. By default, plotSummary uses a gene set’s p.adj score for the X-axis. This behaviour can be easily modified by assigning any of the available sort.by scores into the parameter x.axis , for example, med.rank can be used to create an EGSEA summary plot ( Figure 7a ) as follows: > plotSummary(gsa, gs.label = 1, contrast = 3, + file.name = "summary_c2_LPvsML", + x.axis = "med.rank") Generating Summary plots for the collection c2 Curated Gene Sets and for the contrast LPvsML Figure 7. Summary plots of the significance of selected gene sets in the c2 collection for the LP vs ML contrast. The x-axis in each plot is the med.rank. A cut-off of 300 was used to select significant gene sets in the filtered plot ( b ). The summary plot tends to become cluttered when the size of the gene set collection is very large as in Figure 7a . The parameter x.cutoff can be used to focus in on the significant gene sets rather than plotting the entire gene set collection, for example ( Figure 7b ): > plotSummary(gsa, gs.label = 1, contrast = 3, + file.name = "summary_sig_c2_LPvsML", + x.axis = "med.rank", x.cutoff=300) Generating Summary plots for the collection c2 Curated Gene Sets and for the contrast LPvsML Comparative summary plots can be also generated to compare the significance of gene sets between two contrasts, for example, the comparison between Basal vs LP and Basal vs ML ( Figure 8a ) shows that most of the KEGG pathways are regulated in the same direction with relatively few pathways regulated in opposite directions (purple coloured bubbles in Figure 8a ). Such figures can be generated using the plotSummary method as follows: > plotSummary(gsa, gs.label = "kegg", contrast = c(1,2), + file.name = "summary_kegg_1vs2") Generating Summary plots for the collection KEGG Pathways and for the comparison BasalvsLP vs BasalvsML Figure 8. Comparative summary plots of the significance of all gene sets in the KEGG collection for the comparison of the contrasts: Basal vs LP and Basal vs ML. The plotSummary method has two useful parameters: (i) use.names that can be used to display gene set names instead of gene set IDs and (ii) interactive that can be used to generate an interactive version of this plot. The c5 collection of MSigDB and the Gene Ontology collection of GeneSetDB contain Gene Ontology (GO) terms. These collections are meant to be non-redundant, containing only a small subset of the entire GO and visualizing how these terms are related to each other can be informative. EGSEA utilizes functionality from the topGO package 37 to generate GO graphs for the significant biological processes (BPs), cellular compartments (CCs) and molecular functions (MFs). The plotGOGraph method can generate such a display ( Figure 9 ) as follows: > plotGOGraph(gsa, gs.label="c5BP", contrast = 1, file.name="BasalvsLP-c5BP-top-") Generating GO Graphs for the collection c5 GO Gene Sets (BP) and for the contrast BasalvsLP based on the med.rank > plotGOGraph(gsa, gs.label="c5CC", contrast = 1, file.name="BasalvsLP-c5CC-top-") Generating GO Graphs for the collection c5 GO Gene Sets (CC) and for the contrast BasalvsLP based on the med.rank Figure 9. GO graphs of the top significant GO terms from the c5 gene set collection for the contrast Basal vs LP. The GO graphs are coloured based on the values of the argument sort.by , which in this instance was taken as med.rank by default since this was selected when EGSEA was invoked. The top five most significant GO terms are highlighted by default in each GO category (MF, CC or BP). More terms can be displayed by changing the value of the parameter noSig . However, this might generate very complicated and unresolved graphs. The colour of the nodes varies between red (most significant) and yellow (least significant). The values of the sort.by scoring function are scaled between 0 and 1 to generate these graphs. Another way to visualize results at the experiment level is via a summary bar plot . The method plotBars can be used to generate a bar plot for the top N gene sets in an individual collection for a particular contrast or from a comparative analysis across multiple contrasts. For example, the top 20 gene sets of the comparative analysis carried out on the c2 collection of MSigDB can be visualized in a bar plot ( Figure 10 ) as follows: > plotBars(gsa, gs.label = "c2", contrast = "comparison", file.name="comparison-c2-bars") Generating a bar plot for the collection c2 Curated Gene Sets and the contrast comparison Figure 10. Bar plot of the -log10(p-value) of the top 20 gene sets from the comparative analysis of the c2 collection. The colour of the bars is based on the regulation direction of the gene sets, i.e., red for up-regulated, blue for down-regulated and purple for neutral regulation (in the case of the comparative analysis on experimental contrasts that show opposite behaviours). By default, the − log 10 ( p . ad j ) values are plotted for the top 20 gene sets selected and ordered based on the sort.by parameter. The parameters bar.vals , number and sort.by of plotBars can be changed to customize the bar plot . When changes over multiple conditions are of interest, a summary heatmap can be a useful visualization. The method plotSummaryHeatmaps generates a heatmap of the top N gene sets in the comparative analysis across all experimental conditions ( Figure 11 ). By default, 20 gene sets are selected based on the sort.by parameter and the values plotted are the average log-fold changes at the set level for the genes regulated in the same direction as the set regulation direction, i.e. avg.logfc.dir . The parameters number, sort.by and hm.vals of the plotSummaryHeatmaps can be used to customize the summary heatmap. Additionally, the parameter show.vals can be used to display the values of a specific EGSEA score on the heatmap cells. An example summary heatmap can be generated for the MSigDB c2 collection with the following code: > plotSummaryHeatmap(gsa, gs.label="c2", hm.vals = "avg.logfc.dir", + file.name="summary_heatmaps_c2") Generating summary heatmap for the collection c2 Curated Gene Sets sort.by: med.rank, hm.vals: avg.logfc.dir, show.vals: > plotSummaryHeatmap(gsa, gs.label="kegg", hm.vals = "avg.logfc.dir", + file.name="summary_heatmaps_kegg") Generating summary heatmap for the collection KEGG Pathways sort.by: med.rank, hm.vals: avg.logfc.dir, show.vals: Figure 11. Summary heatmaps for the top 20 gene sets from the c2 ( a ) and KEGG ( b ) collections obtained from the EGSEA comparative analysis. We find the heatmap view at both the gene set and summary level and the summary level bar plots to be useful summaries to include in publications to highlight the gene set testing results. The top differentially expressed genes from each contrast can be accessed from the EGSEAResults object using the limmaTopTable method. > t = limmaTopTable(gsa, contrast=1) > head(t) ENTREZID SYMBOL CHR logFC AveExpr t P.Value adj.P.Val B 19253 19253 Ptpn18 1 -5.63 4.13 -34.5 5.87e-10 9.62e-07 13.2 16324 16324 Inhbb 1 -4.79 6.46 -33.2 7.99e-10 9.62e-07 13.3 53624 53624 Cldn7 11 -5.51 6.30 -40.2 1.75e-10 9.62e-07 14.5 218518 218518 Marveld2 13 -5.14 4.93 -34.8 5.56e-10 9.62e-07 13.5 12759 12759 Clu 14 -5.44 8.86 -41.0 1.52e-10 9.62e-07 14.7 70350 70350 Basp1 15 -6.07 5.25 -34.3 6.22e-10 9.62e-07 13.3 Creating an HTML report of the results. To generate an EGSEA HTML report for this dataset, you can either set report=TRUE when you invoke egsea or use the S4 method generateReport as follows: > generateReport(gsa, number = 20, report.dir="./mam-rnaseq-egsea-report") EGSEA HTML report is being generated ... The EGSEA report generated for this dataset is available online at http://bioinf.wehi.edu.au/EGSEA/mam-rnaseq-egsea-report/index.html ( Figure 12 ). The HTML report is a convenient means of organising all of the results generated up to now, from the individual tables to the gene set level heatmaps, pathway maps and summary level plots. It can easily be shared with collaborators to allow them to explore their results more fully. Interactive tables of results via the DT package ( https://CRAN.R-project.org/package=DT ) and summary plots from plotly ( https://CRAN.R-project.org/package=plotly ) are integrated into the report using htmlwidgets ( https://CRAN.R-project.org/package=htmlwidgets ) and can be added by setting interactive = TRUE in the command above. This option significantly increases both the run time and size of the final report due to the large number of gene sets in most collections. Figure 12. The EGSEA HTML report main page. This summary page details the analysis parameters (methods combined and ranking options selected) and organises the gene set analysis results by contrast, with further separation by gene set collection. The final section on this page presents results from the comparative analysis. For each contrast and gene set collection analysed, links to tables of results and plots are provided. This example completes our overview of EGSEA’s gene set testing and plotting capabilities for RNA-seq data. Readers can refer to the EGSEA vignette or individual help pages for further details on each of the above methods and classes. Analysis of microarray data with EGSEA The second dataset analysed in this workflow comes from Lim et al. (2010) 22 and is the microarray equivalent of the RNA-seq data analysed above. Support for microarray data is a new feature in EGSEA, and in this example, we show an express route for analysis according to the steps shown in Figure 1 , from selecting gene sets and building indexes, to configuring EGSEA, testing and reporting the results. First, the data must be appropriately preprocessed for an EGSEA analysis and to do this we make use of functions available in limma. Reading, preprocessing and normalisation of microarray data To analyse this dataset, we begin by unzipping the files downloaded from http://bioinf.wehi.edu.au/EGSEA/arraydata.zip into the current working directory. Illumina BeadArray data can be read in directly using the readIDAT and readBGX functions from the illuminaio package 38 . However, a more convenient way is via the read.idat function in limma which uses these illuminaio functions and outputs the data as an EListRaw object for further processing. > library(limma) > targets = read.delim("targets.txt", header=TRUE, sep=" ") > data = read.idat(as.character(targets$File), + bgxfile="GPL6887_MouseWG-6_V2_0_R0_11278593_A.bgx", + annotation=c("Entrez_Gene_ID","Symbol", "Chromosome")) Reading manifest file GPL6887_MouseWG-6_V2_0_R0_11278593_A.bgx ... Done 4481850214_B_Grn.idat ... Done 4481850214_C_Grn.idat ... Done 4481850214_D_Grn.idat ... Done 4481850214_F_Grn.idat ... Done 4481850187_A_Grn.idat ... Done 4481850187_B_Grn.idat ... Done 4481850187_D_Grn.idat ... Done 4481850187_E_Grn.idat ... Done 4481850187_F_Grn.idat ... Done 4466975058_A_Grn.idat ... Done 4466975058_B_Grn.idat ... Done 4466975058_C_Grn.idat ... Done 4466975058_D_Grn.idat ... Done 4466975058_E_Grn.idat ... Done 4466975058_F_Grn.idat ... Done Finished reading data. > data$other$Detection = detectionPValues(data) > data$targets = targets > colnames(data) = targets$Sample Next the neqc function in limma is used to carry out normexp background correction and quantile normalisation on the raw intensity values using negative control probes 39 . This is followed by log 2 -transformation of the normalised intensity values and removal of the control probes. > data = neqc(data) We then filter out probes that are consistently non-expressed or lowly expressed throughout all samples as they are uninformative in downstream analysis. Our threshold for expression requires probes to have a detection p -value of less than 0.05 in at least 5 samples (the number of samples within each group). We next remove genes without a valid Entrez ID and in cases where there are multiple probes targeting different isoforms of the same gene, select the probe with highest average expression as the representative one to use in the EGSEA analysis. This leaves 7,123 probes for further analysis. > table(targets$Celltype) Basal LP ML 5 5 5 > keep.exprs = rowSums(data$other$Detection=5 > table(keep.exprs) keep.exprs FALSE TRUE 23638 21643 > data = data[keep.exprs,] > dim(data) [1] 21643 15 > head(data$genes) Probe_Id Array_Address_Id Entrez_Gene_ID Symbol Chromosome 3 ILMN_1219601 2030280 C920011N12Rik 4 ILMN_1252621 1980164 101142 2700050P07Rik 6 6 ILMN_3162407 6220026 Zfp36 7 ILMN_2514723 2030072 1110067B18Rik 8 ILMN_2692952 6040743 329831 4833436C18Rik 4 9 ILMN_1257952 7160091 B930060K05Rik > sum(is.na(data$genes$Entrez_Gene_ID)) [1] 11535 > data1 = data[!is.na(data$genes$Entrez_Gene_ID), ] > dim(data1) [1] 10108 15 > ord = order(lmFit(data1)$Amean, decreasing=TRUE) > ids2keep = data1$genes$Array_Address_Id[ord][!duplicated(data1$genes$Entrez_Gene_ID[ord])] > data1 = data1[match(ids2keep, data1$genes$Array_Address_Id),] > dim(data1) [1] 7123 15 > expr = data1$E > group = as.factor(data1$targets$Celltype) > probe.annot = data1$genes[, 2:4] > head(probe.annot) > head(probe.annot) Array_Address_Id Entrez_Gene_ID Symbol 39513 4120224 20102 Rps4x 9062 2260576 22143 Tuba1b 15308 5720202 12192 Zfp36l1 39894 1470600 11947 Atp5b 24709 2710477 20088 Rps24 9872 1580471 228033 Atp5g3 Setting up the linear model for EGSEA testing As before, we need to set up an appropriate linear model 29 and contrasts matrix to look for differences between the Basal and LP, Basal and ML and LP and ML populations. A batch term is included in the linear model to account for differences in expression that are attributable to the day the experiment was run. > head(data1$targets) File Sample Celltype Time Experiment 2-2 4481850214_B_Grn.idat 2-2 ML At1 1 3-3 4481850214_C_Grn.idat 3-3 LP At1 1 4-4 4481850214_D_Grn.idat 4-4 Basal At1 1 6-7 4481850214_F_Grn.idat 6-7 ML At2 1 7-8 4481850187_A_Grn.idat 7-8 LP At2 1 8-9 4481850187_B_Grn.idat 8-9 Basal At2 1 > experiment = as.character(data1$targets$Experiment) > design = model.matrix(~0 + group + experiment) > colnames(design) = gsub("group", "", colnames(design)) > design Basal LP ML experiment2 1 0 0 1 0 2 0 1 0 0 3 1 0 0 0 4 0 0 1 0 5 0 1 0 0 6 1 0 0 0 7 0 0 1 0 8 0 1 0 0 9 1 0 0 0 10 0 0 1 1 11 0 1 0 1 12 1 0 0 1 13 1 0 0 1 14 0 0 1 1 15 0 1 0 1 attr(,"assign") [1] 1 1 1 2 attr(,"contrasts") attr(,"contrasts")$group [1] "contr.treatment" attr(,"contrasts")$experiment [1] "contr.treatment" > contr.matrix = makeContrasts( + BasalvsLP = Basal-LP, + BasalvsML = Basal-ML, + LPvsML = LP-ML, + levels = colnames(design)) > contr.matrix Contrasts Levels BasalvsLP BasalvsML LPvsML Basal 1 1 0 LP -1 0 1 ML 0 -1 -1 experiment2 0 0 0 1. Creating gene set collection indexes We next extract the mouse c2, c5 and KEGG gene signature collections from the EGSEAdata package and build indexes based on Entrez IDs that link between the genes in each signature and the rows of our expression matrix. > library(EGSEA) > library(EGSEAdata) > gs.annots = buildIdx(entrezIDs=probe.annot[, 2], + species="mouse", + msigdb.gsets=c("c2", "c5"), go.part = TRUE) [1] "Loading MSigDB Gene Sets ... " [1] "Loaded gene sets for the collection c2 ..." [1] "Indexed the collection c2 ..." [1] "Created annotation for the collection c2 ..." [1] "Loaded gene sets for the collection c5 ..." [1] "Indexed the collection c5 ..." [1] "Created annotation for the collection c5 ..." MSigDB c5 gene set collection has been partitioned into c5BP, c5CC, c5MF [1] "Building KEGG pathways annotation object ... " > names(gs.annots) [1] "c2" "c5BP" "c5CC" "c5MF" "kegg" 2. Configuring and 3. Testing with EGSEA The same 11 base methods used previously in the RNA-seq analysis were selected for the ensemble testing of the microarray data using the function egsea.ma. Gene sets were again prioritised by their median rank across the 11 methods. > baseMethods = egsea.base()[-2] > baseMethods [1] "camera" "safe" "gage" "padog" "plage" "zscore" [7] "gsva" "ssgsea" "globaltest" "ora" "fry" > > gsam = egsea.ma(expr=expr, group=group, + probe.annot = probe.annot, + design = design, + contrasts=contr.matrix, + gs.annots=gs.annots, + baseGSEAs=baseMethods, sort.by="med.rank", + num.threads = 8, report = FALSE) EGSEA analysis has started ##------ Tue Jun 20 14:27:32 2017 ------## Log fold changes are estimated using limma package ... limma DE analysis is carried out ... Number of used cores has changed to 3 in order to avoid CPU overloading. EGSEA is running on the provided data and c2 collection EGSEA is running on the provided data and c5BP collection EGSEA is running on the provided data and c5CC collection EGSEA is running on the provided data and c5MF collection EGSEA is running on the provided data and kegg collection ##------ Tue Jun 20 14:33:37 2017 ------## EGSEA analysis took 365.359 seconds. EGSEA analysis has completed 4. Reporting EGSEA results An HTML report that includes each of the gene set level and summary level plots shown individually for the RNA-seq analysis was then created using the generateReport function. We complete our analysis by displaying the top ranked sets for the c2 collection from a comparative analysis across all contrasts. > generateReport(gsam, number = 20, report.dir="./mam-ma-egsea-report") EGSEA HTML report is being generated ... > topSets(gsam, gs.label="c2", contrast = "comparison", names.only=TRUE, number=5) Sorted by med.rank [1] "LIM_MAMMARY_STEM_CELL_UP" [2] "LIM_MAMMARY_LUMINAL_MATURE_DN" [3] "LIM_MAMMARY_STEM_CELL_DN" [4] "CHARAFE_BREAST_CANCER_LUMINAL_VS_MESENCHYMAL_DN" [5] "LIU_PROSTATE_CANCER_DN" The EGSEA report generated for this dataset is available online at http://bioinf.wehi.edu.au/EGSEA/mam-ma-egsea-report/index.html . Reanalysis of this data retrieves similar c2 gene sets to those identified by analysis of RNA-seq data. These included the LIM gene signatures (sets 1, 2 and 3) as well as those derived from populations with similar cellular origin (set 4). Discussion In this workflow article, we have demonstrated how to use the EGSEA package to combine the results obtained from different gene signature databases across multiple GSE methods to find an ensemble solution. A key benefit of an EGSEA analysis is the detailed and comprehensive HTML report that can be shared with collaborators to help them interpret their data. This report includes tables prioritising gene signatures according to the user specified analysis options, and both gene set specific and summary graphics, each of which can be generated individually using specific R commands. The approach taken by EGSEA is facilitated by the diverse range of gene set testing algorithms and plotting capabilities available within Bioconductor. EGSEA has been tailored to suit a limma-based differential expression analysis which continues to be a very popular and flexible platform for transcriptomic data. Analysts who choose an individual GSE algorithm to prioritise their results rather than an ensemble solution can still benefit from EGSEA’s comprehensive reporting capability. Software availability Code to perform this analysis can be found in the EGSEA123 workflow package available from Bioconductor: https://www.bioconductor.org/help/workflows/EGSEA123 . Latest source code is available at: https://github.com/mritchie/EGSEA123 . Archived source code as at the time of publication is available at: https://doi.org/10.5281/zenodo.1043436 40 . Software license: Artistic License 2.0. Competing interests MA and MN are employees of CSL Limited. Grant information This work was funded by a National Health and Medical Research Council (NHMRC) Fellowship to MER (GNT1104924), Victorian State Government Operational Infrastructure Support and Australian Government NHMRC IRIISS. Acknowledgements This material was first trialled in a workshop at the BioC 2017 conference at the Dana Farber Cancer Institute (Boston, MA) on 28 July 2017. We thank the participants at this workshop for their feedback. The authors also thank Dr Alexandra Garnham (The Walter and Eliza Hall Institute of Medical Research) for feedback on this workflow article. Faculty Opinions recommended References 1. Huber W, Carey VJ, Gentleman R, et al. : Orchestrating high-throughput genomic analysis with Bioconductor. Nat Methods. 2015; 12 (2): 115–21. PubMed Abstract | Publisher Full Text | Free Full Text 2. Subramanian A, Tamayo P, Mootha VK, et al. : Gene set enrichment analysis: a knowledge-based approach for interpreting genome-wide expression profiles. Proc Natl Acad Sci U S A. 2005; 102 (43): 15545–50. PubMed Abstract | Publisher Full Text | Free Full Text 3. Araki H, Knapp C, Tsai P, et al. : GeneSetDB: A comprehensive meta-database, statistical and visualisation framework for gene set analysis. FEBS Open Bio. 2012; 2 : 76–82. PubMed Abstract | Publisher Full Text | Free Full Text 4. Kanehisa M, Goto S: KEGG: kyoto encyclopedia of genes and genomes. Nucleic Acids Res. 2000; 28 (1): 27–30. PubMed Abstract | Publisher Full Text | Free Full Text 5. Alhamdoosh M, Ng M, Wilson NJ, et al. : Combining multiple tools outperforms individual methods in gene set enrichment analyses. Bioinformatics. 2017; 33 (3): 414–424. PubMed Abstract | Publisher Full Text | Free Full Text 6. Alhamdoosh M, Ng M, Ritchie ME: EGSEA: Ensemble of Gene Set Enrichment Analyses. R package version 1.5.2. 2017. 7. Tavazoie S, Hughes JD, Campbell MJ, et al. : Systematic determination of genetic network architecture. Nat Genet. 1999; 22 (3): 281–5. PubMed Abstract | Publisher Full Text 8. Goeman JJ, van de Geer SA, de Kort F, et al. : A global test for groups of genes: testing association with a clinical outcome. Bioinformatics. 2004; 20 (1): 93–9. PubMed Abstract | Publisher Full Text 9. Tomfohr J, Lu J, Kepler TB: Pathway level analysis of gene expression using singular value decomposition. BMC Bioinformatics. 2005; 6 : 225. PubMed Abstract | Publisher Full Text | Free Full Text 10. Barry WT, Nobel AB, Wright FA: Significance analysis of functional categories in gene expression studies: a structured permutation approach. Bioinformatics. 2005; 21 (9): 1943–9. PubMed Abstract | Publisher Full Text 11. Lee E, Chuang HY, Kim JW, et al. : Inferring pathway activity toward precise disease classification. PLoS Comput Biol. 2008; 4 (11): e1000217. PubMed Abstract | Publisher Full Text | Free Full Text 12. Luo W, Friedman MS, Shedden K, et al. : GAGE: generally applicable gene set enrichment for pathway analysis. BMC Bioinformatics. 2009; 10 : 161. PubMed Abstract | Publisher Full Text | Free Full Text 13. Barbie DA, Tamayo P, Boehm JS, et al. : Systematic RNA interference reveals that oncogenic KRAS -driven cancers require TBK1. Nature. 2009; 462 (7269): 108–12. PubMed Abstract | Publisher Full Text | Free Full Text 14. Tarca AL, Draghici S, Bhatti G, et al. : Down-weighting overlapping genes improves gene set analysis. BMC Bioinformatics. 2012; 13 : 136. PubMed Abstract | Publisher Full Text | Free Full Text 15. Hänzelmann S, Castelo R, Guinney J: GSVA: gene set variation analysis for microarray and RNA-seq data. BMC Bioinformatics. 2013; 14 : 7. PubMed Abstract | Publisher Full Text | Free Full Text 16. Wu D, Smyth GK: Camera: a competitive gene set test accounting for inter-gene correlation. Nucleic Acids Res. 2012; 40 (17): e133. PubMed Abstract | Publisher Full Text | Free Full Text 17. Wu D, Lim E, Vaillant F, et al. : ROAST: rotation gene set tests for complex microarray experiments. Bioinformatics. 2010; 26 (17): 2176–82. PubMed Abstract | Publisher Full Text | Free Full Text 18. Sheridan JM, Ritchie ME, Best SA, et al. : A pooled shRNA screen for regulators of primary mammary stem and progenitor cells identifies roles for Asap1 and Prox1 . BMC Cancer. 2015; 15 (1): 221. PubMed Abstract | Publisher Full Text | Free Full Text 19. Liao Y, Smyth GK, Shi W: The Subread aligner: fast, accurate and scalable read mapping by seed-and-vote. Nucleic Acids Res. 2013; 41 (10): e108. PubMed Abstract | Publisher Full Text | Free Full Text 20. Liao Y, Smyth GK, Shi W: featureCounts: an efficient general purpose program for assigning sequence reads to genomic features. Bioinformatics. 2014; 30 (7): 923–30. PubMed Abstract | Publisher Full Text 21. Law CW, Alhamdoosh M, Su S, et al. : RNA-seq analysis is easy as 1-2-3 with limma, Glimma and edgeR [version 2; referees: 3 approved]. F1000Res. 2016; 5 : 1408. PubMed Abstract | Publisher Full Text | Free Full Text 22. Lim E, Wu D, Pal B, et al. : Transcriptome analyses of mouse and human mammary cell subpopulations reveal multiple conserved genes and pathways. Breast Cancer Res. 2010; 12 (2): R21. PubMed Abstract | Publisher Full Text | Free Full Text 23. Robinson MD, McCarthy DJ, Smyth GK: edgeR: a Bioconductor package for differential expression analysis of digital gene expression data. Bioinformatics. 2010; 26 (1): 139–40. PubMed Abstract | Publisher Full Text | Free Full Text 24. Ritchie ME, Phipson B, Wu D, et al. : limma powers differential expression analyses for RNA-sequencing and microarray studies. Nucleic Acids Res. 2015; 43 (7): e47. PubMed Abstract | Publisher Full Text | Free Full Text 25. Su S, Law CW, Ah-Cann C, et al. : Glimma: interactive graphics for gene expression analysis. Bioinformatics. 2017; 33 (13): 2050–2. PubMed Abstract | Publisher Full Text 26. Bioconductor Core Team: Mus.musculus: Annotation package for the Mus.musculus object. R package version 1.3.1. 2015. Publisher Full Text 27. Robinson MD, Oshlack A: A scaling normalization method for differential expression analysis of RNA-seq data. Genome Biol. 2010; 11 (3): R25. PubMed Abstract | Publisher Full Text | Free Full Text 28. Law CW, Chen Y, Shi W, et al. : Voom: precision weights unlock linear model analysis tools for RNA-seq read counts. Genome Biol. 2014; 15 (2): R29. PubMed Abstract | Publisher Full Text | Free Full Text 29. Smyth GK: Linear models and empirical Bayes methods for assessing differential expression in microarray experiments. Stat Appl Genet Mol Biol. 2004; 3 (1): Article 3. PubMed Abstract | Publisher Full Text 30. Ziemann M, Kaspi A, Rafehi H, et al. : The ENCODE Gene Set Hub. Lorne Genome Conference. 2017. Publisher Full Text 31. Cerami EG, Gross BE, Demir E, et al. : Pathway commons, a web resource for biological pathway data. Nucleic Acids Res. 2011; 39 (Database issue): D685–D690. PubMed Abstract | Publisher Full Text | Free Full Text 32. Tenenbaum D: KEGGREST: Client-side REST access to KEGG. R package version 1.16.0. 2017. 33. R Core Team: R: A Language and Environment for Statistical Computing. R Foundation for Statistical Computing, Vienna, Austria, 2017. Reference Source 34. Dewey M: metap: meta-analysis of significance values. R package version 0.8. 2017. Reference Source 35. Wilkinson B: A statistical consideration in psychological research. Psychol Bull. 1951; 48 (3): 156–8. PubMed Abstract | Publisher Full Text 36. Luo W, Brouwer C: Pathview: an R/Bioconductor package for pathway-based data integration and visualization. Bioinformatics. 2013; 29 (14): 1830–1. PubMed Abstract | Publisher Full Text | Free Full Text 37. Alexa A, Rahnenfuhrer J: topGO: Enrichment Analysis for Gene Ontology. R package version. 2016. Publisher Full Text 38. Smith ML, Baggerly KA, Bengtsson H, et al. : illuminaio : an open source idat parsing tool for Illumina microarrays [version 1; referees: 2 approved]. F1000Res. 2013; 2 : 264. PubMed Abstract | Publisher Full Text | Free Full Text 39. Shi W, Oshlack A, Smyth GK: Optimizing the noise versus bias trade-off for Illumina Whole Genome Expression Beadchips. Nucleic Acids Res. 2010; 38 (22): e204. PubMed Abstract | Publisher Full Text | Free Full Text 40. mritchie: mritchie/EGSEA123: F1000 Research article version 1 (Version v1). Zenodo. 2017. Data Source Comments on this article Comments (0) Version 1 VERSION 1 PUBLISHED 14 Nov 2017 ADD YOUR COMMENT Comment Author details Author details 1 CSL Limited, Bio21 Institute, Parkville, Victoria, Australia 2 Department of Medical Biology, The University of Melbourne, Parkville, Victoria, Australia 3 Molecular Medicine Division, The Walter and Eliza Hall Institute of Medical Research, Parkville, Victoria, Australia 4 Molecular Genetics of Cancer Division, The Walter and Eliza Hall Institute of Medical Research, Parkville, Victoria, Australia 5 School of Mathematics and Statistics, The University of Melbourne, Parkville, Victoria, Australia Monther Alhamdoosh Roles: Conceptualization, Software, Visualization, Writing – Original Draft Preparation, Writing – Review & Editing Charity W. Law Roles: Software, Writing – Original Draft Preparation Luyi Tian Roles: Software, Writing – Review & Editing Julie M. Sheridan Roles: Investigation, Validation, Writing – Original Draft Preparation Milica Ng Roles: Software, Supervision, Writing – Review & Editing Matthew E. Ritchie Roles: Conceptualization, Software, Supervision, Writing – Original Draft Preparation, Writing – Review & Editing Competing interests MA and MN are employees of CSL Limited. Grant information This work was funded by a National Health and Medical Research Council (NHMRC) Fellowship to MER (GNT1104924), Victorian State Government Operational Infrastructure Support and Australian Government NHMRC IRIISS. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript. Article Versions (1) version 1 Published: 14 Nov 2017, 6:2010 https://doi.org/10.12688/f1000research.12544.1 Copyright © 2017 Alhamdoosh M et al . This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. Download Export To Sciwheel Bibtex EndNote ProCite Ref. Manager (RIS) Sente metrics Views Downloads F1000Research - - PubMed Central info_outline Data from PMC are received and updated monthly. - - Citations open_in_new 0 open_in_new 0 open_in_new SEE MORE DETAILS CITE how to cite this article Alhamdoosh M, Law CW, Tian L et al. Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.12688/f1000research.12544.1 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS track receive updates on this article Track an article to receive email alerts on any updates to this article. TRACK THIS ARTICLE Share Open Peer Review Current Reviewer Status: ? Key to Reviewer Statuses VIEW HIDE Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Version 1 VERSION 1 PUBLISHED 14 Nov 2017 Views 0 Cite How to cite this report: Kohonen P and Grafström R. Reviewer Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27967 ) The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27967 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 20 Dec 2017 Pekka Kohonen , Institute of Environmental Medicine, Karolinska Institutet, Stockholm, Sweden Roland Grafström , IMM Institute of Environmental Medicine, Karolinska Institutet, Stockholm, Sweden Approved VIEWS 0 https://doi.org/10.5256/f1000research.13583.r27967 GSEA analysis methods do not all produce same results. Score-based gene set analysis methods like the Broad Institute GSEA tool are considered to perform better than normal Fisher’s exact test (overrepresentation analysis). But analysts often use methods they know to ... Continue reading READ ALL GSEA analysis methods do not all produce same results. Score-based gene set analysis methods like the Broad Institute GSEA tool are considered to perform better than normal Fisher’s exact test (overrepresentation analysis). But analysts often use methods they know to be less than ideal in order to reduce complexity and save time. So it is good to have a unified interface for GSEA analyses with R – it helps save programming time and reduces complexity. In addition EGSEA is a unique method that combines up to 12 gene set analysis methods into a single score. Independent test also corroborate that the tool using the 12 has more specificity and good sensitivity compared to using some of the tests alone. The EGSEA 1-2-3 workflow is easy to use and generate good-quality figures with the ggplot2 R package. So9me of the figures are novel compared to other packages e.g., scatter plots designed to compare different contrasts. It is also very useful that the tool can be applied to multiple contrasts at a time, although if there are too many contrasts then the number of plots becomes unwieldly (increases combinatorically). Some more technical comments: The results object is very complicated for retrieving individual method analysis results (although summaries are readily available). I quite like the "biobroom" Bioconductor package that does "tidy" data frames from limma results objects. All in all a very useful package both for automating the running of lots of methods at the same time and of course for the "ensemble" method. It is recommended to be considered to be part of a standard bioinformatics workflow. Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Yes Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Yes Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes Competing Interests: No competing interests were disclosed. We confirm that we have read this submission and believe that we have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Kohonen P and Grafström R. Reviewer Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27967 ) The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27967 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Views 0 Cite How to cite this report: Luo W. Reviewer Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27970 ) The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27970 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 13 Dec 2017 Weijun Luo , Department of Bioinformatics and Genomics, UNC Charlotte (University of North Carolina at Charlotte), Charlotte, NC, USA Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.13583.r27970 EGSEA is a new gene set analysis tool that combines results from multiple individual tools in R as to yield better results. The authors have published EGSEA methodology previously. This paper focuses on the practical analysis workflow based on EGSEA ... Continue reading READ ALL EGSEA is a new gene set analysis tool that combines results from multiple individual tools in R as to yield better results. The authors have published EGSEA methodology previously. This paper focuses on the practical analysis workflow based on EGSEA with specific examples. As EGSEA is a compound and complicated analysis procedure, this work serves as a valuable guidance for the users to make full use of this tool. I’ve gone through the workflow line by line, it seems to work well. However, authors can improve their work by addressing the following issues. There should be an R code script which includes all source code and concise comments like the one in company with the vignette in any Bioconductor package. It would be much easy for the users/reviewers to try the example code. It is not convenient to follow the code in this manuscript, the code need to be edit to remove the prompt symbols (> or +) at each line when copying/pasting. It takes too long to run the egsea analysis example on modest machine. It is advisable to show a lesser example in the workflow with only one gene set collection like kegg and just a few base methods like: gsa = egsea(voom.results=v, contrasts=contr.matrix, gs.annots=gs.annots$kegg, symbolsMap=symbolsMap, baseGSEAs=baseMethods[1:4], sort.by="med.rank", num.threads = 3, report = FALSE) The rank of the gsa results shown following the t = topSets(..) line is confusing. The p.adj for the top 1 gene set is not the smallest, actually much bigger than top 2, 6 and 8. Presumably, the gene sets are ranked by med.rank instead of p.adj here. However, the opposite was described in the text above near the egsea.sort() line: “Although p.adj is the default option for sorting EGSEA results for convenience, ...” In addition, there is big difference between the final rank and med.rank (e.g. 1 vs 36). This may suggest inconsistent results came from different base methods. This may also be due to the large number of gene sets being tested. Again, using a smaller gene set collection and a few base methods could make the ranking more consistent. All visualization functions, i.e. plotHeatmap, plotPathway, plotGOGraph, plotMethods, plotSummary and plotBars share largely the same set of arguments, they can have a unified wrapper function like plot.gsa() with an extra argument type to specify the plot type. Functions plotPathway, plotGOGraph are wrapper functions for those in the pathview and topGO package as the author noted in the text. It would be good to explicit show some message like “calling plotting function from pathview or topGO package etc”, just like the message when running egsea(). HTML report of the results is a very valuable feature for the users. However, the code can run a long time, it would be helpful to add some progress reminder message to generateReport() function like egsea(). BTW, the KEGG Pathway graphs are not shown properly in the report example at http://bioinf.wehi.edu.au/EGSEA/mam-rnaseq-egsea-report/index.html . Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Yes Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Partly Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes Competing Interests: No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Luo W. Reviewer Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27970 ) The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27970 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Views 0 Cite How to cite this report: Drnevich J. Reviewer Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27974 ) The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27974 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 05 Dec 2017 Jenny Drnevich , Roy J. Carver Biotechnology Center, University of Illinois at Urbana–Champaign, Urbana, IL, USA Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.13583.r27974 This F1000 software tool article describes the EGSEA package that incorporates many different gene set testing methods from various packages and also allows access to a wide array of gene sets from different databases through the accompanying EGSEAdata package. These ... Continue reading READ ALL This F1000 software tool article describes the EGSEA package that incorporates many different gene set testing methods from various packages and also allows access to a wide array of gene sets from different databases through the accompanying EGSEAdata package. These packages will enable researchers to conveniently test many different methods and incorporate their results to get more robust biological insights 1 , and this article gives a well-written walk-through of how to use the packages. The biggest limitation I see it that EGSEA is focused only on human and mouse data (and rat? The article does not list rat but the help page for buildIdx() lists rat as one of the species). I understand that many of the gene set collections like MSigDB and GeneSetDB are only available for human/mouse, but KEGG currently lists 429 Eukaryotic organisms ( http://www.genome.jp/kegg/catalog/org_list.html ) and GO terms are readily available for 19 species using BioC's pre-built OrgDB packages and hundreds of other through AnnotationHub. It is unclear whether EGSEA functions buildCustomIdx and buildGMTIdx that were "written to allow users to run EGSEA on gene set collections that may have been curated within a lab or downloaded from public databases and allow use of gene identifiers other than Entrez IDs" can be used to run EGSEA on additional species. If so, this should be clearly stated in both the Abstract and in the body of the article, plus an example given on how to use buildCustomIdx for another species. If there is some reason that EGSEA cannot currently extend to other species, this should be acknowledged as a limitation and future versions should strive to allow this (although not required before approval). Other issues to address before approval: I am unable to create the html report on my Windows machine, getting the following error: Build GO DAG topology .......... There are no adj nodes for node: GO:0061857 Error in switch(type, isa = 0, partof = 1, -1) : EXPR must be a length 1 vector However, I reported this error to the support site ( https://support.bioconductor.org/p/103640/#103748 ) and got a speedy reply from the author. It hopefully will be resolved soon, although there is a concern of why the error was not found on another Windows machine. I am concerned that as demonstrated in this paper, EGSEA seems to take the place of standard limma differential expression analysis, in that the model fitting takes place within the egsea() function. Certain gene set testing functions do need the individual expression values and not just the fitted values in an MArrayLM object but given the computational time (8 min as shown in the article code block and 19 min on my own computer) you should never run egsea() without first assessing the model fit on your own! Ideally the egsea function could be written to accept MArrayLM, or at least the article should clearly state that users should have first assessed the validity of the model fit through the usual workflow of Law et al. (2016) 2 prior to running EGSEA. I also wonder why there are different interfaces for voom-based analysis and microarray data given that both use EList objects. I understand that the voom weights need to be used internally, but limma's lmFit function handles both without trouble, although it was originally coded for microarray data and the voom functionality came later. Even if there needs to be a separate function egsea.ma() for non-voom, non-count data, it should still accept an EList object so that the user does not have to pull out the expression data and the grouping info. Back to the computational time required, there are several vague references to removing the roast method "to save time" and that the report generation "significantly" increases run time. it would be nice to have an example of the time required to run roast and the report generation for the computational architecture that created the article. Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Partly Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Partly Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes References 1. Alhamdoosh M, Ng M, Wilson N, Sheridan J, et al.: Combining multiple tools outperforms individual methods in gene set enrichment analyses. Bioinformatics . 2016. Publisher Full Text 2. Law CW, Alhamdoosh M, Su S, Smyth GK, et al.: RNA-seq analysis is easy as 1-2-3 with limma, Glimma and edgeR. F1000Res . 2016; 5 : 1408 PubMed Abstract | Publisher Full Text Competing Interests: No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Drnevich J. Reviewer Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27974 ) The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27974 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Views 0 Cite How to cite this report: Castelo R. Reviewer Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27972 ) The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27972 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 27 Nov 2017 Robert Castelo , Department of Experimental and Health Sciences, Universitat Pompeu Fabra, Barcelona, Spain Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.13583.r27972 This article describes a gene set enrichment analysis (GSEA) workflow for the "Ensembl of GSEA" (EGSEA) R/Bioconductor software package . EGSEA is an ensemble-like method recently published 1 by the authors of this workflow that allows the user to simultaneously apply different GSEA algorithms on ... Continue reading READ ALL This article describes a gene set enrichment analysis (GSEA) workflow for the "Ensembl of GSEA" (EGSEA) R/Bioconductor software package . EGSEA is an ensemble-like method recently published 1 by the authors of this workflow that allows the user to simultaneously apply different GSEA algorithms on a high-throughput molecular profiling data set, by combining p-values associated with each algorithm using classical meta-analysis approaches such as the Fisher's method. Because the statistical methodology is already described in detail in the corresponding publication, the present software tool article focuses on showing a step-by-step workflow with EGSEA. However, the vignette of the software package already provides a very detailed description about how to use EGSEA through its 39 pages. Therefore, it would be useful for the interested reader to find upfront when he/she should be consulting the vignette and when he/she should be consulting this workflow. Besides this introductory aspects, the following issues should be addressed before approval: The code given in the article breaks, at least in my computer, more concretely, at this line: gsa = egsea(voom.results=v, contrasts=contr.matrix, gs.annots=gs.annots, symbolsMap=symbolsMap, baseGSEAs=baseMethods, sort.by="med.rank", num.threads = 8, report = FALSE) EGSEA analysis has started ##------ Mon Nov 27 12:37:42 2017 ------## Log fold changes are estimated using limma package ... limma DE analysis is carried out ... Number of used cores has changed to 4 in order to avoid CPU overloading. EGSEA is running on the provided data and c2 collection .......camera*....safe*...gage*.padog*....gsva*..fry*...plage*...globaltest*...zscore*...ora*...ssgsea* Error in temp.results[[baseGSEA]][[i]][names(gs.annot@idx), ] : incorrect number of dimensions while running it with the latest release version 1.6.0. This is strange since the package builds and runs the vignette without problems. So, this might be related to the different sample data sets. A possible hint may come from the fact that the 'buildIdx()' call is not returning the expected class of object, according to the workflow: class(gs.annots$s2) ## [1] "NULL" summary(gs.annots$s2) ## Length Class Mode ## 0 NULL NULL The workflow contains a rather high amount of code, often with a non-trivial use of externally instantiated objects and nested calls to functions. It would be helpful for the interested reader to be able to easily copy and paste the instructions, but the fact that R commands are given with the R shell '>' and '+' symbols makes it less easy. A non-expert user may even copy those characters and get an error. I would recommend removing those characters from the illustrated code, just as it happens with the vignette. The workflow assumes that the user has a 'DGEList' object with gene metadata including the mapping between Entrez identifiers' and HGNC symbols. This is a rather unrealistic assumption and I would recommend that the workflow starts building that object from scratch and showing how to build that table of gene metadata. Below I also describe other issues that I would recommend to be considered in future versions of the software but which I do not consider them to be required for approval of this article: The so-called "summary plot" shows the -log10 p-value on the x-axis and average absolute log fold-change of the set genes on the y-axis. Because this is in a way analogous to a rotated volcano plot, I would suggest to use the same arrangement of axes as in the volcano plot, which is a rather standardized display of significance and magnitude of the effects of interest. One of the key features of the Bioconductor project, to which the EGSEA package is contributing to, is enabling software interoperability through sharing the use of common data structures across different software packages. Using specialized data structures, where analogous ones have been already designed by the Bioconductor core team or by a wider community of developers, locks the user into that package and limits the possibilities of using it as a building block in other more complex workflows. I'm making this comment because I have the impression that the EGSEA package would benefit of using the infrastructure provided by the Bioconductor GSEABase package , in which data structures are defined to store and access gene sets and collections of gene sets of different kinds. A salient feature of that infrastructure is the possibility to seamlessly map gene identifiers of different kinds. This would simplify and improve the user experience of EGSEA since mapping between genes coded with a particular kind of identifier, and gene sets defined with another kind, is one of the most common tasks in a GSEA-like analysis. Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Yes Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Yes Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes References 1. Alhamdoosh M, Ng M, Wilson N, Sheridan J, et al.: Combining multiple tools outperforms individual methods in gene set enrichment analyses. Bioinformatics . 2016. Publisher Full Text Competing Interests: No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Castelo R. Reviewer Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27972 ) The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27972 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Comments on this article Comments (0) Version 1 VERSION 1 PUBLISHED 14 Nov 2017 ADD YOUR COMMENT Comment keyboard_arrow_left keyboard_arrow_right Open Peer Review Reviewer Status info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Reviewer Reports Invited Reviewers 1 2 3 4 Version 1 14 Nov 17 read read read read Robert Castelo , Universitat Pompeu Fabra, Barcelona, Spain Jenny Drnevich , University of Illinois at Urbana–Champaign, Urbana, USA Weijun Luo , UNC Charlotte (University of North Carolina at Charlotte), Charlotte, USA Pekka Kohonen , Karolinska Institutet, Stockholm, Sweden Roland Grafström , Karolinska Institutet, Stockholm, Sweden Comments on this article All Comments (0) Add a comment Sign up for content alerts Sign Up You are now signed up to receive this alert Browse by related subjects keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2017 Kohonen P et al. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 20 Dec 2017 | for Version 1 Pekka Kohonen , Institute of Environmental Medicine, Karolinska Institutet, Stockholm, Sweden Roland Grafström , IMM Institute of Environmental Medicine, Karolinska Institutet, Stockholm, Sweden 0 Views copyright © 2017 Kohonen P et al. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions GSEA analysis methods do not all produce same results. Score-based gene set analysis methods like the Broad Institute GSEA tool are considered to perform better than normal Fisher’s exact test (overrepresentation analysis). But analysts often use methods they know to be less than ideal in order to reduce complexity and save time. So it is good to have a unified interface for GSEA analyses with R – it helps save programming time and reduces complexity. In addition EGSEA is a unique method that combines up to 12 gene set analysis methods into a single score. Independent test also corroborate that the tool using the 12 has more specificity and good sensitivity compared to using some of the tests alone. The EGSEA 1-2-3 workflow is easy to use and generate good-quality figures with the ggplot2 R package. So9me of the figures are novel compared to other packages e.g., scatter plots designed to compare different contrasts. It is also very useful that the tool can be applied to multiple contrasts at a time, although if there are too many contrasts then the number of plots becomes unwieldly (increases combinatorically). Some more technical comments: The results object is very complicated for retrieving individual method analysis results (although summaries are readily available). I quite like the "biobroom" Bioconductor package that does "tidy" data frames from limma results objects. All in all a very useful package both for automating the running of lots of methods at the same time and of course for the "ensemble" method. It is recommended to be considered to be part of a standard bioinformatics workflow. Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Yes Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Yes Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes Competing Interests No competing interests were disclosed. We confirm that we have read this submission and believe that we have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. reply Respond to this report Responses (0) Kohonen P and Grafström R. Peer Review Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27967) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27967 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2017 Luo W. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 13 Dec 2017 | for Version 1 Weijun Luo , Department of Bioinformatics and Genomics, UNC Charlotte (University of North Carolina at Charlotte), Charlotte, NC, USA 0 Views copyright © 2017 Luo W. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions EGSEA is a new gene set analysis tool that combines results from multiple individual tools in R as to yield better results. The authors have published EGSEA methodology previously. This paper focuses on the practical analysis workflow based on EGSEA with specific examples. As EGSEA is a compound and complicated analysis procedure, this work serves as a valuable guidance for the users to make full use of this tool. I’ve gone through the workflow line by line, it seems to work well. However, authors can improve their work by addressing the following issues. There should be an R code script which includes all source code and concise comments like the one in company with the vignette in any Bioconductor package. It would be much easy for the users/reviewers to try the example code. It is not convenient to follow the code in this manuscript, the code need to be edit to remove the prompt symbols (> or +) at each line when copying/pasting. It takes too long to run the egsea analysis example on modest machine. It is advisable to show a lesser example in the workflow with only one gene set collection like kegg and just a few base methods like: gsa = egsea(voom.results=v, contrasts=contr.matrix, gs.annots=gs.annots$kegg, symbolsMap=symbolsMap, baseGSEAs=baseMethods[1:4], sort.by="med.rank", num.threads = 3, report = FALSE) The rank of the gsa results shown following the t = topSets(..) line is confusing. The p.adj for the top 1 gene set is not the smallest, actually much bigger than top 2, 6 and 8. Presumably, the gene sets are ranked by med.rank instead of p.adj here. However, the opposite was described in the text above near the egsea.sort() line: “Although p.adj is the default option for sorting EGSEA results for convenience, ...” In addition, there is big difference between the final rank and med.rank (e.g. 1 vs 36). This may suggest inconsistent results came from different base methods. This may also be due to the large number of gene sets being tested. Again, using a smaller gene set collection and a few base methods could make the ranking more consistent. All visualization functions, i.e. plotHeatmap, plotPathway, plotGOGraph, plotMethods, plotSummary and plotBars share largely the same set of arguments, they can have a unified wrapper function like plot.gsa() with an extra argument type to specify the plot type. Functions plotPathway, plotGOGraph are wrapper functions for those in the pathview and topGO package as the author noted in the text. It would be good to explicit show some message like “calling plotting function from pathview or topGO package etc”, just like the message when running egsea(). HTML report of the results is a very valuable feature for the users. However, the code can run a long time, it would be helpful to add some progress reminder message to generateReport() function like egsea(). BTW, the KEGG Pathway graphs are not shown properly in the report example at http://bioinf.wehi.edu.au/EGSEA/mam-rnaseq-egsea-report/index.html . Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Yes Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Partly Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes Competing Interests No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (0) Luo W. Peer Review Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27970) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27970 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2017 Drnevich J. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 05 Dec 2017 | for Version 1 Jenny Drnevich , Roy J. Carver Biotechnology Center, University of Illinois at Urbana–Champaign, Urbana, IL, USA 0 Views copyright © 2017 Drnevich J. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions This F1000 software tool article describes the EGSEA package that incorporates many different gene set testing methods from various packages and also allows access to a wide array of gene sets from different databases through the accompanying EGSEAdata package. These packages will enable researchers to conveniently test many different methods and incorporate their results to get more robust biological insights 1 , and this article gives a well-written walk-through of how to use the packages. The biggest limitation I see it that EGSEA is focused only on human and mouse data (and rat? The article does not list rat but the help page for buildIdx() lists rat as one of the species). I understand that many of the gene set collections like MSigDB and GeneSetDB are only available for human/mouse, but KEGG currently lists 429 Eukaryotic organisms ( http://www.genome.jp/kegg/catalog/org_list.html ) and GO terms are readily available for 19 species using BioC's pre-built OrgDB packages and hundreds of other through AnnotationHub. It is unclear whether EGSEA functions buildCustomIdx and buildGMTIdx that were "written to allow users to run EGSEA on gene set collections that may have been curated within a lab or downloaded from public databases and allow use of gene identifiers other than Entrez IDs" can be used to run EGSEA on additional species. If so, this should be clearly stated in both the Abstract and in the body of the article, plus an example given on how to use buildCustomIdx for another species. If there is some reason that EGSEA cannot currently extend to other species, this should be acknowledged as a limitation and future versions should strive to allow this (although not required before approval). Other issues to address before approval: I am unable to create the html report on my Windows machine, getting the following error: Build GO DAG topology .......... There are no adj nodes for node: GO:0061857 Error in switch(type, isa = 0, partof = 1, -1) : EXPR must be a length 1 vector However, I reported this error to the support site ( https://support.bioconductor.org/p/103640/#103748 ) and got a speedy reply from the author. It hopefully will be resolved soon, although there is a concern of why the error was not found on another Windows machine. I am concerned that as demonstrated in this paper, EGSEA seems to take the place of standard limma differential expression analysis, in that the model fitting takes place within the egsea() function. Certain gene set testing functions do need the individual expression values and not just the fitted values in an MArrayLM object but given the computational time (8 min as shown in the article code block and 19 min on my own computer) you should never run egsea() without first assessing the model fit on your own! Ideally the egsea function could be written to accept MArrayLM, or at least the article should clearly state that users should have first assessed the validity of the model fit through the usual workflow of Law et al. (2016) 2 prior to running EGSEA. I also wonder why there are different interfaces for voom-based analysis and microarray data given that both use EList objects. I understand that the voom weights need to be used internally, but limma's lmFit function handles both without trouble, although it was originally coded for microarray data and the voom functionality came later. Even if there needs to be a separate function egsea.ma() for non-voom, non-count data, it should still accept an EList object so that the user does not have to pull out the expression data and the grouping info. Back to the computational time required, there are several vague references to removing the roast method "to save time" and that the report generation "significantly" increases run time. it would be nice to have an example of the time required to run roast and the report generation for the computational architecture that created the article. Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Partly Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Partly Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes References 1. Alhamdoosh M, Ng M, Wilson N, Sheridan J, et al.: Combining multiple tools outperforms individual methods in gene set enrichment analyses. Bioinformatics . 2016. Publisher Full Text 2. Law CW, Alhamdoosh M, Su S, Smyth GK, et al.: RNA-seq analysis is easy as 1-2-3 with limma, Glimma and edgeR. F1000Res . 2016; 5 : 1408 PubMed Abstract | Publisher Full Text Competing Interests No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (0) Drnevich J. Peer Review Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27974) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27974 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2017 Castelo R. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 27 Nov 2017 | for Version 1 Robert Castelo , Department of Experimental and Health Sciences, Universitat Pompeu Fabra, Barcelona, Spain 0 Views copyright © 2017 Castelo R. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions This article describes a gene set enrichment analysis (GSEA) workflow for the "Ensembl of GSEA" (EGSEA) R/Bioconductor software package . EGSEA is an ensemble-like method recently published 1 by the authors of this workflow that allows the user to simultaneously apply different GSEA algorithms on a high-throughput molecular profiling data set, by combining p-values associated with each algorithm using classical meta-analysis approaches such as the Fisher's method. Because the statistical methodology is already described in detail in the corresponding publication, the present software tool article focuses on showing a step-by-step workflow with EGSEA. However, the vignette of the software package already provides a very detailed description about how to use EGSEA through its 39 pages. Therefore, it would be useful for the interested reader to find upfront when he/she should be consulting the vignette and when he/she should be consulting this workflow. Besides this introductory aspects, the following issues should be addressed before approval: The code given in the article breaks, at least in my computer, more concretely, at this line: gsa = egsea(voom.results=v, contrasts=contr.matrix, gs.annots=gs.annots, symbolsMap=symbolsMap, baseGSEAs=baseMethods, sort.by="med.rank", num.threads = 8, report = FALSE) EGSEA analysis has started ##------ Mon Nov 27 12:37:42 2017 ------## Log fold changes are estimated using limma package ... limma DE analysis is carried out ... Number of used cores has changed to 4 in order to avoid CPU overloading. EGSEA is running on the provided data and c2 collection .......camera*....safe*...gage*.padog*....gsva*..fry*...plage*...globaltest*...zscore*...ora*...ssgsea* Error in temp.results[[baseGSEA]][[i]][names(gs.annot@idx), ] : incorrect number of dimensions while running it with the latest release version 1.6.0. This is strange since the package builds and runs the vignette without problems. So, this might be related to the different sample data sets. A possible hint may come from the fact that the 'buildIdx()' call is not returning the expected class of object, according to the workflow: class(gs.annots$s2) ## [1] "NULL" summary(gs.annots$s2) ## Length Class Mode ## 0 NULL NULL The workflow contains a rather high amount of code, often with a non-trivial use of externally instantiated objects and nested calls to functions. It would be helpful for the interested reader to be able to easily copy and paste the instructions, but the fact that R commands are given with the R shell '>' and '+' symbols makes it less easy. A non-expert user may even copy those characters and get an error. I would recommend removing those characters from the illustrated code, just as it happens with the vignette. The workflow assumes that the user has a 'DGEList' object with gene metadata including the mapping between Entrez identifiers' and HGNC symbols. This is a rather unrealistic assumption and I would recommend that the workflow starts building that object from scratch and showing how to build that table of gene metadata. Below I also describe other issues that I would recommend to be considered in future versions of the software but which I do not consider them to be required for approval of this article: The so-called "summary plot" shows the -log10 p-value on the x-axis and average absolute log fold-change of the set genes on the y-axis. Because this is in a way analogous to a rotated volcano plot, I would suggest to use the same arrangement of axes as in the volcano plot, which is a rather standardized display of significance and magnitude of the effects of interest. One of the key features of the Bioconductor project, to which the EGSEA package is contributing to, is enabling software interoperability through sharing the use of common data structures across different software packages. Using specialized data structures, where analogous ones have been already designed by the Bioconductor core team or by a wider community of developers, locks the user into that package and limits the possibilities of using it as a building block in other more complex workflows. I'm making this comment because I have the impression that the EGSEA package would benefit of using the infrastructure provided by the Bioconductor GSEABase package , in which data structures are defined to store and access gene sets and collections of gene sets of different kinds. A salient feature of that infrastructure is the possibility to seamlessly map gene identifiers of different kinds. This would simplify and improve the user experience of EGSEA since mapping between genes coded with a particular kind of identifier, and gene sets defined with another kind, is one of the most common tasks in a GSEA-like analysis. Is the rationale for developing the new software tool clearly explained? Yes Is the description of the software tool technically sound? Yes Are sufficient details of the code, methods and analysis (if applicable) provided to allow replication of the software development and its use by others? Yes Is sufficient information provided to allow interpretation of the expected output datasets and any results generated using the tool? Yes Are the conclusions about the tool and its performance adequately supported by the findings presented in the article? Yes References 1. Alhamdoosh M, Ng M, Wilson N, Sheridan J, et al.: Combining multiple tools outperforms individual methods in gene set enrichment analyses. Bioinformatics . 2016. Publisher Full Text Competing Interests No competing interests were disclosed. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (0) Castelo R. Peer Review Report For: Easy and efficient ensemble gene set testing with EGSEA [version 1; peer review: 1 approved, 3 approved with reservations] . F1000Research 2017, 6 :2010 ( https://doi.org/10.5256/f1000research.13583.r27972) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/6-2010/v1#referee-response-27972 Alongside their report, reviewers assign a status to the article: Approved - the paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations - A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved - fundamental flaws in the paper seriously undermine the findings and conclusions Adjust parameters to alter display View on desktop for interactive features Includes Interactive Elements View on desktop for interactive features Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Stay Updated Sign up for content alerts and receive a weekly or monthly email with all newly published articles Register with F1000Research Already registered? Sign in Not now, thanks close PLEASE NOTE If you are an AUTHOR of this article, please check that you signed in with the account associated with this article otherwise we cannot automatically identify your role as an author and your comment will be labelled as a “User Comment”. If you are a REVIEWER of this article, please check that you have signed in with the account associated with this article and then go to your account to submit your report, please do not post your review here. If you do not have access to your original account, please contact us . All commenters must hold a formal affiliation as per our Policies . The information that you give us will be displayed next to your comment. User comments must be in English, comprehensible and relevant to the article under discussion. We reserve the right to remove any comments that we consider to be inappropriate, offensive or otherwise in breach of the User Comment Terms and Conditions . Commenters must not use a comment for personal attacks. When criticisms of the article are based on unpublished data, the data should be made available. I accept the User Comment Terms and Conditions Please confirm that you accept the User Comment Terms and Conditions. Affiliation ✕ refresh Please enter your institution. Note: To add your institution or organisation, start typing the name and then select the correct name from the list. Where applicable, the name will appear in both the original language and in English. Do not paste in the name. If the name does not appear in the drop-down list, we will display the information you have entered. ✕ refresh Country/Region * USA UK Canada China France Germany Afghanistan Aland Islands Albania Algeria American Samoa Andorra Angola Anguilla Antarctica Antigua and Barbuda Argentina Armenia Aruba Australia Austria Azerbaijan Bahamas Bahrain Bangladesh Barbados Belarus Belgium Belize Benin Bermuda Bhutan Bolivia Bosnia and Herzegovina Botswana Bouvet Island Brazil British Indian Ocean Territory British Virgin Islands Brunei Bulgaria Burkina Faso Burundi Cambodia Cameroon Canada Cape Verde Cayman Islands Central African Republic Chad Chile China Christmas Island Cocos (Keeling) Islands Colombia Comoros Congo Cook Islands Costa Rica Cote d'Ivoire Croatia Cuba Cyprus Czech Republic Democratic Republic of the Congo Denmark Djibouti Dominica Dominican Republic Ecuador Egypt El Salvador Equatorial Guinea Eritrea Estonia Ethiopia Falkland Islands Faroe Islands Federated States of Micronesia Fiji Finland France French Guiana French Polynesia French Southern Territories Gabon Georgia Germany Ghana Gibraltar Greece Greenland Grenada Guadeloupe Guam Guatemala Guernsey Guinea Guinea-Bissau Guyana Haiti Heard Island and Mcdonald Islands Holy See (Vatican City State) Honduras Hong Kong Hungary Iceland India Indonesia Iran Iraq Ireland Israel Italy Jamaica Japan Jersey Jordan Kazakhstan Kenya Kiribati Kosovo (Serbia and Montenegro) Kuwait Kyrgyzstan Lao People's Democratic Republic Latvia Lebanon Lesotho Liberia Libya Liechtenstein Lithuania Luxembourg Macao Madagascar Malawi Malaysia Maldives Mali Malta Marshall Islands Martinique Mauritania Mauritius Mayotte Mexico Minor Outlying Islands of the United States Moldova Monaco Mongolia Montenegro Montserrat Morocco Mozambique Myanmar Namibia Nauru Nepal Netherlands Antilles New Caledonia New Zealand Nicaragua Niger Nigeria Niue Norfolk Island North Korea North Macedonia Northern Mariana Islands Norway Oman Pakistan Palau Palestinian Territory Panama Papua New Guinea Paraguay Peru Philippines Pitcairn Poland Portugal Puerto Rico Qatar Reunion Romania Russian Federation Rwanda Saint Helena Saint Kitts and Nevis Saint Lucia Saint Pierre and Miquelon Saint Vincent and the Grenadines Samoa San Marino Sao Tome and Principe Saudi Arabia Senegal Serbia Seychelles Sierra Leone Singapore Slovakia Slovenia Solomon Islands Somalia South Africa South Georgia and the South Sandwich Is South Korea South Sudan Spain Sri Lanka Sudan Suriname Svalbard and Jan Mayen Swaziland Sweden Switzerland Syria Taiwan Tajikistan Tanzania Thailand The Gambia The Netherlands Timor-Leste Togo Tokelau Tonga Trinidad and Tobago Tunisia Turkey Turkmenistan Turks and Caicos Islands Tuvalu UK USA Uganda Ukraine United Arab Emirates United States Virgin Islands Uruguay Uzbekistan Vanuatu Venezuela Vietnam Wallis and Futuna West Bank and Gaza Strip Western Sahara Yemen Zambia Zimbabwe Please select your country/region. You must enter a comment. Competing Interests Please disclose any competing interests that might be construed to influence your judgment of the article's or peer review report's validity or importance. Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Please state your competing interests The comment has been saved. An error has occurred. Please try again. Cancel Post var lTitle = "Easy and efficient ensemble gene set testing...".replace("'", ''); var linkedInUrl = "http://www.linkedin.com/shareArticle?url=https://f1000research.com/articles/6-2010/v1" + "&title=" + encodeURIComponent(lTitle) + "&summary=" + encodeURIComponent('Read the article by '); var deliciousUrl = "https://del.icio.us/post?url=https://f1000research.com/articles/6-2010/v1&title=" + encodeURIComponent(lTitle); var redditUrl = "http://reddit.com/submit?url=https://f1000research.com/articles/6-2010/v1" + "&title=" + encodeURIComponent(lTitle); linkedInUrl += encodeURIComponent('Alhamdoosh M et al.'); var offsetTop = /chrome/i.test( navigator.userAgent ) ? 4 : -10; var addthis_config = { ui_offset_top: offsetTop, services_compact : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_expanded : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_custom : [ { name: "LinkedIn", url: linkedInUrl, icon:"/img/icon/at_linkedin.svg" }, { name: "Mendeley", url: "http://www.mendeley.com/import/?url=https://f1000research.com/articles/6-2010/v1/mendeley", icon:"/img/icon/at_mendeley.svg" }, { name: "Reddit", url: redditUrl, icon:"/img/icon/at_reddit.svg" }, ] }; var addthis_share = { url: "https://f1000research.com/articles/6-2010", templates : { twitter : "Easy and efficient ensemble gene set testing with EGSEA. Alhamdoosh M et al., published by " + "@F1000Research" + ", https://f1000research.com/articles/6-2010/v1" } }; if (typeof(addthis) != "undefined"){ addthis.addEventListener('addthis.ready', checkCount); addthis.addEventListener('addthis.menu.share', checkCount); } $(".f1r-shares-twitter").attr("href", "https://twitter.com/intent/tweet?text=" + addthis_share.templates.twitter); $(".f1r-shares-facebook").attr("href", "https://www.facebook.com/sharer/sharer.php?u=" + addthis_share.url); $(".f1r-shares-linkedin").attr("href", addthis_config.services_custom[0].url); $(".f1r-shares-reddit").attr("href", addthis_config.services_custom[2].url); $(".f1r-shares-mendelay").attr("href", addthis_config.services_custom[1].url); function checkCount(){ setTimeout(function(){ $(".addthis_button_expanded").each(function(){ var count = $(this).text(); if (count !== "" && count != "0") $(this).removeClass("is-hidden"); else $(this).addClass("is-hidden"); }); }, 1000); } close How to cite this report {{reportCitation}} Cancel Copy Citation Details $(function(){R.ui.buttonDropdowns('.dropdown-for-downloads');}); $(function(){R.ui.toolbarDropdowns('.toolbar-dropdown-for-downloads');}); $.get("/articles/acj/12544/13583") new F1000.Clipboard(); new F1000.ThesaurusTermsDisplay("articles", "article", "13583"); $(document).ready(function() { $( "#frame1" ).on('load', function() { var mydiv = $(this).contents().find("div"); var h = mydiv.height(); console.log(h) }); var tooltipLivingFigure = jQuery(".interactive-living-figure-label .icon-more-info"), titleLivingFigure = tooltipLivingFigure.attr("title"); tooltipLivingFigure.simpletip({ fixed: true, position: ["-115", "30"], baseClass: 'small-tooltip', content:titleLivingFigure + " " }); tooltipLivingFigure.removeAttr("title"); $("body").on("click", ".cite-living-figure", function(e) { e.preventDefault(); var ref = $(this).attr("data-ref"); $(this).closest(".living-figure-list-container").find("#" + ref).fadeIn(200); }); $("body").on("click", ".close-cite-living-figure", function(e) { e.preventDefault(); $(this).closest(".popup-window-wrapper").fadeOut(200); }); $(document).on("mouseup", function(e) { var metricsContainer = $(".article-metrics-popover-wrapper"); if (!metricsContainer.is(e.target) && metricsContainer.has(e.target).length === 0) { $(".article-metrics-close-button").click(); } }); var articleId = $('#articleId').val(); if($("#main-article-count-box").attachArticleMetrics) { $("#main-article-count-box").attachArticleMetrics(articleId, { articleMetricsView: true }); } }); var figshareWidget = $(".new_figshare_widget"); if (figshareWidget.length > 0) { window.figshare.load("f1000", function(Widget) { // Select a tag/tags defined in your page. In this tag we will place the widget. _.map(figshareWidget, function(el){ var widget = new Widget({ articleId: $(el).attr("figshare_articleId") //height:300 // this is the height of the viewer part. [Default: 550] }); widget.initialize(); // initialize the widget widget.mount(el); // mount it in a tag that's on your page // this will save the widget on the global scope for later use from // your JS scripts. This line is optional. //window.widget = widget; }); }); } close Error Close Add Reset F1000.MICROSERVICES.AFFILIATION = ''; $(document).ready(function () { $('.js-affiliations-form').each((index, form) => { new AffiliationForm({ formId: form.id, institutionErrorSelector: '.comment-enter-institution', departmentErrorSelector: '.comment-enter-department', placeSelector: '.js-add-comment-place', stateSelector: '.js-add-comment-state', zipCodeSelector: '.js-add-comment-zipcode', countrySelector: '.js-add-comment-country', countryErrorSelector: '.comment-enter-country', }); }); }); $(document).ready(function () { var reportIds = { "27968": 0, "27970": 80, "27972": 86, "27973": 0, "27974": 67, "27975": 0, "27967": 89, }; $(".referee-response-container,.js-referee-report").each(function(index, el) { var reportId = $(el).attr("data-reportid"), reportCount = reportIds[reportId] || 0; $(el).find(".comments-count-container,.js-referee-report-views").html(reportCount); }); var uuidInput = $("#article_uuid"), oldUUId = uuidInput.val(), newUUId = "476f2f3f-0800-4651-9fde-823d17be37cb"; uuidInput.val(newUUId); $("a[href*='article_uuid=']").each(function(index, el) { var newHref = $(el).attr("href").replace(oldUUId, newUUId); $(el).attr("href", newHref); }); }); An innovative open access publishing platform offering rapid publication and open peer review, whilst supporting data deposition and sharing. Browse Gateways Collections How it Works Contact For Developers Cookie Notice Privacy Notice RSS Submit Your Research Follow us © 2012-2026 F1000 Research Ltd. ISSN 2046-1402 | Legal | Partner of Research4Life • CrossRef • ORCID • FAIRSharing R.templateTests.simpleTemplate = R.template(' $text $text $text $text $text '); R.templateTests.runTests(); var F1000platform = new F1000.Platform({ name: "f1000research", displayName: "F1000Research", hostName: "f1000research.com", id: "1", editorialEmail: "[email protected]", infoEmail: "[email protected]", usePmcStats: true }); $(function(){R.ui.dropdowns('.dropdown-for-authors, .dropdown-for-about, .dropdown-for-myresearch');}); // $(function(){R.ui.dropdowns('.dropdown-for-referees');}); $(document).ready(function () { if ($(".cookie-warning").is(":visible")) { $(".sticky").css("margin-bottom", "35px"); $(".devices").addClass("devices-and-cookie-warning"); } $(".cookie-warning .close-button").click(function (e) { $(".devices").removeClass("devices-and-cookie-warning"); $(".sticky").css("margin-bottom", "0"); }); $("#tweeter-feed .tweet-message").each(function (i, message) { var self = $(message); self.html(linkify(self.html())); }); $(".partner").on("mouseenter mouseleave", function() { $(this).find(".gray-scale, .colour").toggleClass("is-hidden"); }); }); Sign In Remember me Forgotten your password? Sign In Cancel Email or password not correct. Please try again Please wait... $(function(){ // Note: All the setup needs to run against a name attribute and *not* the id due the clonish // nature of facebox... $("a[id=googleSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("GOOGLE"); $("form[id=oAuthForm]").submit(); }); $("a[id=facebookSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("FACEBOOK"); $("form[id=oAuthForm]").submit(); }); $("a[id=orcidSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("ORCID"); $("form[id=oAuthForm]").submit(); }); }); If you've forgotten your password, please enter your email address below and we'll send you instructions on how to reset your password. The email address should be the one you originally registered with F1000. Email address not valid, please try again You registered with F1000 via Google, so we cannot reset your password. To sign in, please click here . If you still need help with your Google account password, please click here . You registered with F1000 via Facebook, so we cannot reset your password. To sign in, please click here . If you still need help with your Facebook account password, please click here . Code not correct, please try again Reset password Cancel Email us for further assistance. Server error, please try again. If your email address is registered with us, we will email you instructions to reset your password. If you think you should have received this email but it has not arrived, please check your spam filters and/or contact for further assistance. Please wait... Register $(document).ready(function () { signIn.createSignInAsRow($("#sign-in-form-gfb-popup")); $(".target-field").each(function () { var uris = $(this).val().split("/"); if (uris.pop() === "login") { $(this).val(uris.toString().replace(",","/")); } }); });

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. The paper's references may be in our DB but unresolved to ``paper_id`` (resolution happens at ingest when the cited DOI matches a row we already have). Run the cross-source citation reconcile pass to retry.

Source provenance

europepmc
last seen: 2026-05-19T01:45:01.086888+00:00
unpaywall
last seen: 2026-06-06T02:00:05.402940+00:00
License: CC-BY-4.0