A Bioconductor workflow for the Bayesian analysis of spatial proteomics

preprint OA: closed
Full text JSON View at publisher

Abstract

Knowledge of the subcellular location of a protein gives valuable insight into its function. The field of spatial proteomics has become increasingly popular due to improved multiplexing capabilities in high-throughput mass spectrometry, which have made it possible to systematically localise thousands of proteins per experiment. In parallel with these experimental advances, improved methods for analysing spatial proteomics data have also been developed. In this workflow, we demonstrate using `pRoloc` for the Bayesian analysis of spatial proteomics data. We detail the software infrastructure and then provide step-by-step guidance of the analysis, including setting up a pipeline, assessing convergence, and interpreting downstream results. In several places we provide additional details on Bayesian analysis to provide users with a holistic view of Bayesian analysis for spatial proteomics data.
Full text 170,397 characters · extracted from preprint-html · click to expand
A Bioconductor workflow for the Bayesian analysis... | F1000Research "use strict";function _typeof(t){return(_typeof="function"==typeof Symbol&&"symbol"==typeof Symbol.iterator?function(t){return typeof t}:function(t){return t&&"function"==typeof Symbol&&t.constructor===Symbol&&t!==Symbol.prototype?"symbol":typeof t})(t)}!function(){var t=function(){var t,e,o=[],n=window,r=n;for(;r;){try{if(r.frames.__tcfapiLocator){t=r;break}}catch(t){}if(r===n.top)break;r=r.parent}t||(!function t(){var e=n.document,o=!!n.frames.__tcfapiLocator;if(!o)if(e.body){var r=e.createElement("iframe");r.style.cssText="display:none",r.name="__tcfapiLocator",e.body.appendChild(r)}else setTimeout(t,5);return!o}(),n.__tcfapi=function(){for(var t=arguments.length,n=new Array(t),r=0;r 3&&2===parseInt(n[1],10)&&"boolean"==typeof n[3]&&(e=n[3],"function"==typeof n[2]&&n[2]("set",!0)):"ping"===n[0]?"function"==typeof n[2]&&n[2]({gdprApplies:e,cmpLoaded:!1,cmpStatus:"stub"}):o.push(n)},n.addEventListener("message",(function(t){var e="string"==typeof t.data,o={};if(e)try{o=JSON.parse(t.data)}catch(t){}else o=t.data;var n="object"===_typeof(o)&&null!==o?o.__tcfapiCall:null;n&&window.__tcfapi(n.command,n.version,(function(o,r){var a={__tcfapiReturn:{returnValue:o,success:r,callId:n.callId}};t&&t.source&&t.source.postMessage&&t.source.postMessage(e?JSON.stringify(a):a,"*")}),n.parameter)}),!1))};"undefined"!=typeof module?module.exports=t:t()}(); dataLayer = dataLayer || []; // Standard GTM initialization - Google Consent Mode handles consent automatically (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start': new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0], j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src= 'https://www.googletagmanager.com/gtm.js?id='+i+dl+ '>m_auth=hzk0Vc3qFsQYhCrIoHz68A>m_preview=env-1>m_cookies_win=x';f.parentNode.insertBefore(j,f); })(window,document,'script','dataLayer','GTM-MWFK8L5J'); ;window.NREUM||(NREUM={});NREUM.init={distributed_tracing:{enabled:true},privacy:{cookies_enabled:true},ajax:{deny_list:["bam.nr-data.net"]}}; ;NREUM.loader_config={accountID:"438030",trustKey:"438030",agentID:"772317073",licenseKey:"97f8f67f26",applicationID:"772317073"} ;NREUM.info={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net",licenseKey:"97f8f67f26",applicationID:"772317073",sa:1} ;/*! For license information please see nr-loader-spa-1.236.0.min.js.LICENSE.txt */ (()=>{"use strict";var e,t,r={5763:(e,t,r)=>{r.d(t,{P_:()=>l,Mt:()=>g,C5:()=>s,DL:()=>v,OP:()=>T,lF:()=>D,Yu:()=>y,Dg:()=>h,CX:()=>c,GE:()=>b,sU:()=>_});var n=r(8632),i=r(9567);const o={beacon:n.ce.beacon,errorBeacon:n.ce.errorBeacon,licenseKey:void 0,applicationID:void 0,sa:void 0,queueTime:void 0,applicationTime:void 0,ttGuid:void 0,user:void 0,account:void 0,product:void 0,extra:void 0,jsAttributes:{},userAttributes:void 0,atts:void 0,transactionName:void 0,tNamePlain:void 0},a={};function s(e){if(!e)throw new Error("All info objects require an agent identifier!");if(!a[e])throw new Error("Info for ".concat(e," was never set"));return a[e]}function c(e,t){if(!e)throw new Error("All info objects require an agent identifier!");a[e]=(0,i.D)(t,o),(0,n.Qy)(e,a[e],"info")}var u=r(7056);const d=()=>{const e={blockSelector:"[data-nr-block]",maskInputOptions:{password:!0}};return{allow_bfcache:!0,privacy:{cookies_enabled:!0},ajax:{deny_list:void 0,enabled:!0,harvestTimeSeconds:10},distributed_tracing:{enabled:void 0,exclude_newrelic_header:void 0,cors_use_newrelic_header:void 0,cors_use_tracecontext_headers:void 0,allowed_origins:void 0},session:{domain:void 0,expiresMs:u.oD,inactiveMs:u.Hb},ssl:void 0,obfuscate:void 0,jserrors:{enabled:!0,harvestTimeSeconds:10},metrics:{enabled:!0},page_action:{enabled:!0,harvestTimeSeconds:30},page_view_event:{enabled:!0},page_view_timing:{enabled:!0,harvestTimeSeconds:30,long_task:!1},session_trace:{enabled:!0,harvestTimeSeconds:10},harvest:{tooManyRequestsDelay:60},session_replay:{enabled:!1,harvestTimeSeconds:60,sampleRate:.1,errorSampleRate:.1,maskTextSelector:"*",maskAllInputs:!0,get blockClass(){return"nr-block"},get ignoreClass(){return"nr-ignore"},get maskTextClass(){return"nr-mask"},get blockSelector(){return e.blockSelector},set blockSelector(t){e.blockSelector+=",".concat(t)},get maskInputOptions(){return e.maskInputOptions},set maskInputOptions(t){e.maskInputOptions={...t,password:!0}}},spa:{enabled:!0,harvestTimeSeconds:10}}},f={};function l(e){if(!e)throw new Error("All configuration objects require an agent identifier!");if(!f[e])throw new Error("Configuration for ".concat(e," was never set"));return f[e]}function h(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");f[e]=(0,i.D)(t,d()),(0,n.Qy)(e,f[e],"config")}function g(e,t){if(!e)throw new Error("All configuration objects require an agent identifier!");var r=l(e);if(r){for(var n=t.split("."),i=0;i {r.d(t,{D:()=>i});var n=r(50);function i(e,t){try{if(!e||"object"!=typeof e)return(0,n.Z)("Setting a Configurable requires an object as input");if(!t||"object"!=typeof t)return(0,n.Z)("Setting a Configurable requires a model to set its initial properties");const r=Object.create(Object.getPrototypeOf(t),Object.getOwnPropertyDescriptors(t)),o=0===Object.keys(r).length?e:r;for(let a in o)if(void 0!==e[a])try{"object"==typeof e[a]&&"object"==typeof t[a]?r[a]=i(e[a],t[a]):r[a]=e[a]}catch(e){(0,n.Z)("An error occurred while setting a property of a Configurable",e)}return r}catch(e){(0,n.Z)("An error occured while setting a Configurable",e)}}},6818:(e,t,r)=>{r.d(t,{Re:()=>i,gF:()=>o,q4:()=>n});const n="1.236.0",i="PROD",o="CDN"},385:(e,t,r)=>{r.d(t,{FN:()=>a,IF:()=>u,Nk:()=>f,Tt:()=>s,_A:()=>o,il:()=>n,pL:()=>c,v6:()=>i,w1:()=>d});const n="undefined"!=typeof window&&!!window.document,i="undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self.navigator instanceof WorkerNavigator||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis.navigator instanceof WorkerNavigator),o=n?window:"undefined"!=typeof WorkerGlobalScope&&("undefined"!=typeof self&&self instanceof WorkerGlobalScope&&self||"undefined"!=typeof globalThis&&globalThis instanceof WorkerGlobalScope&&globalThis),a=""+o?.location,s=/iPad|iPhone|iPod/.test(navigator.userAgent),c=s&&"undefined"==typeof SharedWorker,u=(()=>{const e=navigator.userAgent.match(/Firefox[/\s](\d+\.\d+)/);return Array.isArray(e)&&e.length>=2?+e[1]:0})(),d=Boolean(n&&window.document.documentMode),f=!!navigator.sendBeacon},1117:(e,t,r)=>{r.d(t,{w:()=>o});var n=r(50);const i={agentIdentifier:"",ee:void 0};class o{constructor(e){try{if("object"!=typeof e)return(0,n.Z)("shared context requires an object as input");this.sharedContext={},Object.assign(this.sharedContext,i),Object.entries(e).forEach((e=>{let[t,r]=e;Object.keys(i).includes(t)&&(this.sharedContext[t]=r)}))}catch(e){(0,n.Z)("An error occured while setting SharedContext",e)}}}},8e3:(e,t,r)=>{r.d(t,{L:()=>d,R:()=>c});var n=r(2177),i=r(1284),o=r(4322),a=r(3325);const s={};function c(e,t){const r={staged:!1,priority:a.p[t]||0};u(e),s[e].get(t)||s[e].set(t,r)}function u(e){e&&(s[e]||(s[e]=new Map))}function d(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:"",t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:"feature";if(u(e),!e||!s[e].get(t))return a(t);s[e].get(t).staged=!0;const r=[...s[e]];function a(t){const r=e?n.ee.get(e):n.ee,a=o.X.handlers;if(r.backlog&&a){var s=r.backlog[t],c=a[t];if(c){for(var u=0;s&&u {let[t,r]=e;return r.staged}))&&(r.sort(((e,t)=>e[1].priority-t[1].priority)),r.forEach((e=>{let[t]=e;a(t)})))}function f(e,t){var r=e[1];(0,i.D)(t[r],(function(t,r){var n=e[0];if(r[0]===n){var i=r[1],o=e[3],a=e[2];i.apply(o,a)}}))}},2177:(e,t,r)=>{r.d(t,{c:()=>f,ee:()=>u});var n=r(8632),i=r(2210),o=r(1284),a=r(5763),s="nr@context";let c=(0,n.fP)();var u;function d(){}function f(e){return(0,i.X)(e,s,l)}function l(){return new d}function h(){u.aborted=!0,u.backlog={}}c.ee?u=c.ee:(u=function e(t,r){var n={},c={},f={},g=!1;try{g=16===r.length&&(0,a.OP)(r).isolatedBacklog}catch(e){}var p={on:b,addEventListener:b,removeEventListener:y,emit:v,get:x,listeners:w,context:m,buffer:A,abort:h,aborted:!1,isBuffering:E,debugId:r,backlog:g?{}:t&&"object"==typeof t.backlog?t.backlog:{}};return p;function m(e){return e&&e instanceof d?e:e?(0,i.X)(e,s,l):l()}function v(e,r,n,i,o){if(!1!==o&&(o=!0),!u.aborted||i){t&&o&&t.emit(e,r,n);for(var a=m(n),s=w(e),d=s.length,f=0;fn,p:()=>i});var n=r(2177).ee.get("handle");function i(e,t,r,i,o){o?(o.buffer([e],i),o.emit(e,t,r)):(n.buffer([e],i),n.emit(e,t,r))}},4322:(e,t,r)=>{r.d(t,{X:()=>o});var n=r(5546);o.on=a;var i=o.handlers={};function o(e,t,r,o){a(o||n.E,i,e,t,r)}function a(e,t,r,i,o){o||(o="feature"),e||(e=n.E);var a=t[o]=t[o]||{};(a[r]=a[r]||[]).push([e,i])}},3239:(e,t,r)=>{r.d(t,{bP:()=>s,iz:()=>c,m$:()=>a});var n=r(385);let i=!1,o=!1;try{const e={get passive(){return i=!0,!1},get signal(){return o=!0,!1}};n._A.addEventListener("test",null,e),n._A.removeEventListener("test",null,e)}catch(e){}function a(e,t){return i||o?{capture:!!e,passive:i,signal:t}:!!e}function s(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;window.addEventListener(e,t,a(r,n))}function c(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2],n=arguments.length>3?arguments[3]:void 0;document.addEventListener(e,t,a(r,n))}},4402:(e,t,r)=>{r.d(t,{Ht:()=>u,M:()=>c,Rl:()=>a,ky:()=>s});var n=r(385);const i="xxxxxxxx-xxxx-4xxx-yxxx-xxxxxxxxxxxx";function o(e,t){return e?15&e[t]:16*Math.random()|0}function a(){const e=n._A?.crypto||n._A?.msCrypto;let t,r=0;return e&&e.getRandomValues&&(t=e.getRandomValues(new Uint8Array(31))),i.split("").map((e=>"x"===e?o(t,++r).toString(16):"y"===e?(3&o()|8).toString(16):e)).join("")}function s(e){const t=n._A?.crypto||n._A?.msCrypto;let r,i=0;t&&t.getRandomValues&&(r=t.getRandomValues(new Uint8Array(31)));const a=[];for(var s=0;s {r.d(t,{Bq:()=>n,Hb:()=>o,oD:()=>i});const n="NRBA",i=144e5,o=18e5},7894:(e,t,r)=>{function n(){return Math.round(performance.now())}r.d(t,{z:()=>n})},7243:(e,t,r)=>{r.d(t,{e:()=>o});var n=r(385),i={};function o(e){if(e in i)return i[e];if(0===(e||"").indexOf("data:"))return{protocol:"data"};let t;var r=n._A?.location,o={};if(n.il)t=document.createElement("a"),t.href=e;else try{t=new URL(e,r.href)}catch(e){return o}o.port=t.port;var a=t.href.split("://");!o.port&&a[1]&&(o.port=a[1].split("/")[0].split("@").pop().split(":")[1]),o.port&&"0"!==o.port||(o.port="https"===a[0]?"443":"80"),o.hostname=t.hostname||r.hostname,o.pathname=t.pathname,o.protocol=a[0],"/"!==o.pathname.charAt(0)&&(o.pathname="/"+o.pathname);var s=!t.protocol||":"===t.protocol||t.protocol===r.protocol,c=t.hostname===r.hostname&&t.port===r.port;return o.sameOrigin=s&&(!t.hostname||c),"/"===o.pathname&&(i[e]=o),o}},50:(e,t,r)=>{function n(e,t){"function"==typeof console.warn&&(console.warn("New Relic: ".concat(e)),t&&console.warn(t))}r.d(t,{Z:()=>n})},2587:(e,t,r)=>{r.d(t,{N:()=>c,T:()=>u});var n=r(2177),i=r(5546),o=r(8e3),a=r(3325);const s={stn:[a.D.sessionTrace],err:[a.D.jserrors,a.D.metrics],ins:[a.D.pageAction],spa:[a.D.spa],sr:[a.D.sessionReplay,a.D.sessionTrace]};function c(e,t){const r=n.ee.get(t);e&&"object"==typeof e&&(Object.entries(e).forEach((e=>{let[t,n]=e;void 0===u[t]&&(s[t]?s[t].forEach((e=>{n?(0,i.p)("feat-"+t,[],void 0,e,r):(0,i.p)("block-"+t,[],void 0,e,r),(0,i.p)("rumresp-"+t,[Boolean(n)],void 0,e,r)})):n&&(0,i.p)("feat-"+t,[],void 0,void 0,r),u[t]=Boolean(n))})),Object.keys(s).forEach((e=>{void 0===u[e]&&(s[e]?.forEach((t=>(0,i.p)("rumresp-"+e,[!1],void 0,t,r))),u[e]=!1)})),(0,o.L)(t,a.D.pageViewEvent))}const u={}},2210:(e,t,r)=>{r.d(t,{X:()=>i});var n=Object.prototype.hasOwnProperty;function i(e,t,r){if(n.call(e,t))return e[t];var i=r();if(Object.defineProperty&&Object.keys)try{return Object.defineProperty(e,t,{value:i,writable:!0,enumerable:!1}),i}catch(e){}return e[t]=i,i}},1284:(e,t,r)=>{r.d(t,{D:()=>n});const n=(e,t)=>Object.entries(e||{}).map((e=>{let[r,n]=e;return t(r,n)}))},4351:(e,t,r)=>{r.d(t,{P:()=>o});var n=r(2177);const i=()=>{const e=new WeakSet;return(t,r)=>{if("object"==typeof r&&null!==r){if(e.has(r))return;e.add(r)}return r}};function o(e){try{return JSON.stringify(e,i())}catch(e){try{n.ee.emit("internal-error",[e])}catch(e){}}}},3960:(e,t,r)=>{r.d(t,{K:()=>a,b:()=>o});var n=r(3239);function i(){return"undefined"==typeof document||"complete"===document.readyState}function o(e,t){if(i())return e();(0,n.bP)("load",e,t)}function a(e){if(i())return e();(0,n.iz)("DOMContentLoaded",e)}},8632:(e,t,r)=>{r.d(t,{EZ:()=>u,Qy:()=>c,ce:()=>o,fP:()=>a,gG:()=>d,mF:()=>s});var n=r(7894),i=r(385);const o={beacon:"bam.nr-data.net",errorBeacon:"bam.nr-data.net"};function a(){return i._A.NREUM||(i._A.NREUM={}),void 0===i._A.newrelic&&(i._A.newrelic=i._A.NREUM),i._A.NREUM}function s(){let e=a();return e.o||(e.o={ST:i._A.setTimeout,SI:i._A.setImmediate,CT:i._A.clearTimeout,XHR:i._A.XMLHttpRequest,REQ:i._A.Request,EV:i._A.Event,PR:i._A.Promise,MO:i._A.MutationObserver,FETCH:i._A.fetch}),e}function c(e,t,r){let i=a();const o=i.initializedAgents||{},s=o[e]||{};return Object.keys(s).length||(s.initializedAt={ms:(0,n.z)(),date:new Date}),i.initializedAgents={...o,[e]:{...s,[r]:t}},i}function u(e,t){a()[e]=t}function d(){return function(){let e=a();const t=e.info||{};e.info={beacon:o.beacon,errorBeacon:o.errorBeacon,...t}}(),function(){let e=a();const t=e.init||{};e.init={...t}}(),s(),function(){let e=a();const t=e.loader_config||{};e.loader_config={...t}}(),a()}},7956:(e,t,r)=>{r.d(t,{N:()=>i});var n=r(3239);function i(e){let t=arguments.length>1&&void 0!==arguments[1]&&arguments[1],r=arguments.length>2?arguments[2]:void 0,i=arguments.length>3?arguments[3]:void 0;return void(0,n.iz)("visibilitychange",(function(){if(t)return void("hidden"==document.visibilityState&&e());e(document.visibilityState)}),r,i)}},1214:(e,t,r)=>{r.d(t,{em:()=>v,u5:()=>N,QU:()=>S,_L:()=>I,Gm:()=>L,Lg:()=>M,gy:()=>U,BV:()=>Q,Kf:()=>ee});var n=r(2177);const i="nr@original";var o=Object.prototype.hasOwnProperty,a=!1;function s(e,t){return e||(e=n.ee),r.inPlace=function(e,t,n,i,o){n||(n="");var a,s,c,u="-"===n.charAt(0);for(c=0;c 2?n-2:0),o=2;o {r(A[T],e,w),r(E[T],e,w)})),r(l._A,"fetch",y),t.on(y+"end",(function(e,r){var n=this;if(r){var i=r.headers.get("content-length");null!==i&&(n.rxSize=i),t.emit(y+"done",[null,r],n)}else t.emit(y+"done",[e],n)})),t}const O={},j=["pushState","replaceState"];function S(e){const t=function(e){return(e||n.ee).get("history")}(e);return!l.il||O[t.debugId]++||(O[t.debugId]=1,s(t).inPlace(window.history,j,"-")),t}var P=r(3239);const C={},R=["appendChild","insertBefore","replaceChild"];function I(e){const t=function(e){return(e||n.ee).get("jsonp")}(e);if(!l.il||C[t.debugId])return t;C[t.debugId]=!0;var r=s(t),i=/[?&](?:callback|cb)=([^&#]+)/,o=/(.*)\.([^.]+)/,a=/^(\w+)(\.|$)(.*)$/;function c(e,t){var r=e.match(a),n=r[1],i=r[3];return i?c(i,t[n]):t[n]}return r.inPlace(Node.prototype,R,"dom-"),t.on("dom-start",(function(e){!function(e){if(!e||"string"!=typeof e.nodeName||"script"!==e.nodeName.toLowerCase())return;if("function"!=typeof e.addEventListener)return;var n=(a=e.src,s=a.match(i),s?s[1]:null);var a,s;if(!n)return;var u=function(e){var t=e.match(o);if(t&&t.length>=3)return{key:t[2],parent:c(t[1],window)};return{key:e,parent:window}}(n);if("function"!=typeof u.parent[u.key])return;var d={};function f(){t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}function l(){t.emit("jsonp-error",[],d),t.emit("jsonp-end",[],d),e.removeEventListener("load",f,(0,P.m$)(!1)),e.removeEventListener("error",l,(0,P.m$)(!1))}r.inPlace(u.parent,[u.key],"cb-",d),e.addEventListener("load",f,(0,P.m$)(!1)),e.addEventListener("error",l,(0,P.m$)(!1)),t.emit("new-jsonp",[e.src],d)}(e[0])})),t}var k=r(5763);const H={};function L(e){const t=function(e){return(e||n.ee).get("mutation")}(e);if(!l.il||H[t.debugId])return t;H[t.debugId]=!0;var r=s(t),i=k.Yu.MO;return i&&(window.MutationObserver=function(e){return this instanceof i?new i(r(e,"fn-")):i.apply(this,arguments)},MutationObserver.prototype=i.prototype),t}const z={};function M(e){const t=function(e){return(e||n.ee).get("promise")}(e);if(z[t.debugId])return t;z[t.debugId]=!0;var r=n.c,o=s(t),a=k.Yu.PR;return a&&function(){function e(r){var n=t.context(),i=o(r,"executor-",n,null,!1);const s=Reflect.construct(a,[i],e);return t.context(s).getCtx=function(){return n},s}l._A.Promise=e,Object.defineProperty(e,"name",{value:"Promise"}),e.toString=function(){return a.toString()},Object.setPrototypeOf(e,a),["all","race"].forEach((function(r){const n=a[r];e[r]=function(e){let i=!1;[...e||[]].forEach((e=>{this.resolve(e).then(a("all"===r),a(!1))}));const o=n.apply(this,arguments);return o;function a(e){return function(){t.emit("propagate",[null,!i],o,!1,!1),i=i||!e}}}})),["resolve","reject"].forEach((function(r){const n=a[r];e[r]=function(e){const r=n.apply(this,arguments);return e!==r&&t.emit("propagate",[e,!0],r,!1,!1),r}})),e.prototype=a.prototype;const n=a.prototype.then;a.prototype.then=function(){var e=this,i=r(e);i.promise=e;for(var a=arguments.length,s=new Array(a),c=0;c e())),t};function m(e,t){i.inPlace(t,["onreadystatechange"],"fn-",E)}function b(){var e=this,t=r.context(e);e.readyState>3&&!t.resolved&&(t.resolved=!0,r.emit("xhr-resolved",[],e)),i.inPlace(e,f,"fn-",E)}if(function(e,t){for(var r in e)t[r]=e[r]}(o,p),p.prototype=o.prototype,i.inPlace(p.prototype,J,"-xhr-",E),r.on("send-xhr-start",(function(e,t){m(e,t),function(e){h.push(e),a&&(y?y.then(A):u?u(A):(w=-w,x.data=w))}(t)})),r.on("open-xhr-start",m),a){var y=c&&c.resolve();if(!u&&!c){var w=1,x=document.createTextNode(w);new a(A).observe(x,{characterData:!0})}}else t.on("fn-end",(function(e){e[0]&&e[0].type===d||A()}));function A(){for(var e=0;e {r.d(t,{t:()=>n});const n=r(3325).D.ajax},6660:(e,t,r)=>{r.d(t,{A:()=>i,t:()=>n});const n=r(3325).D.jserrors,i="nr@seenError"},3081:(e,t,r)=>{r.d(t,{gF:()=>o,mY:()=>i,t9:()=>n,vz:()=>s,xS:()=>a});const n=r(3325).D.metrics,i="sm",o="cm",a="storeSupportabilityMetrics",s="storeEventMetrics"},4649:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageAction},7633:(e,t,r)=>{r.d(t,{Dz:()=>i,OJ:()=>a,qw:()=>o,t9:()=>n});const n=r(3325).D.pageViewEvent,i="firstbyte",o="domcontent",a="windowload"},9251:(e,t,r)=>{r.d(t,{t:()=>n});const n=r(3325).D.pageViewTiming},3614:(e,t,r)=>{r.d(t,{BST_RESOURCE:()=>i,END:()=>s,FEATURE_NAME:()=>n,FN_END:()=>u,FN_START:()=>c,PUSH_STATE:()=>d,RESOURCE:()=>o,START:()=>a});const n=r(3325).D.sessionTrace,i="bstResource",o="resource",a="-start",s="-end",c="fn"+a,u="fn"+s,d="pushState"},7836:(e,t,r)=>{r.d(t,{BODY:()=>A,CB_END:()=>E,CB_START:()=>u,END:()=>x,FEATURE_NAME:()=>i,FETCH:()=>_,FETCH_BODY:()=>v,FETCH_DONE:()=>m,FETCH_START:()=>p,FN_END:()=>c,FN_START:()=>s,INTERACTION:()=>l,INTERACTION_API:()=>d,INTERACTION_EVENTS:()=>o,JSONP_END:()=>b,JSONP_NODE:()=>g,JS_TIME:()=>T,MAX_TIMER_BUDGET:()=>a,REMAINING:()=>f,SPA_NODE:()=>h,START:()=>w,originalSetTimeout:()=>y});var n=r(5763);const i=r(3325).D.spa,o=["click","submit","keypress","keydown","keyup","change"],a=999,s="fn-start",c="fn-end",u="cb-start",d="api-ixn-",f="remaining",l="interaction",h="spaNode",g="jsonpNode",p="fetch-start",m="fetch-done",v="fetch-body-",b="jsonp-end",y=n.Yu.ST,w="-start",x="-end",A="-body",E="cb"+x,T="jsTime",_="fetch"},5938:(e,t,r)=>{r.d(t,{W:()=>o});var n=r(5763),i=r(2177);class o{constructor(e,t,r){this.agentIdentifier=e,this.aggregator=t,this.ee=i.ee.get(e,(0,n.OP)(this.agentIdentifier).isolatedBacklog),this.featureName=r,this.blocked=!1}}},9144:(e,t,r)=>{r.d(t,{j:()=>m});var n=r(3325),i=r(5763),o=r(5546),a=r(2177),s=r(7894),c=r(8e3),u=r(3960),d=r(385),f=r(50),l=r(3081),h=r(8632);function g(){const e=(0,h.gG)();["setErrorHandler","finished","addToTrace","inlineHit","addRelease","addPageAction","setCurrentRouteName","setPageViewName","setCustomAttribute","interaction","noticeError","setUserId"].forEach((t=>{e[t]=function(){for(var r=arguments.length,n=new Array(r),i=0;i 1?r-1:0),i=1;i {e.exposed&&e.api[t]&&o.push(e.api[t](...n))})),o.length>1?o:o[0]}(t,...n)}}))}var p=r(2587);function m(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:{},m=arguments.length>2?arguments[2]:void 0,v=arguments.length>3?arguments[3]:void 0,{init:b,info:y,loader_config:w,runtime:x={loaderType:m},exposed:A=!0}=t;const E=(0,h.gG)();y||(b=E.init,y=E.info,w=E.loader_config),(0,i.Dg)(e,b||{}),(0,i.GE)(e,w||{}),(0,i.sU)(e,x),y.jsAttributes??={},d.v6&&(y.jsAttributes.isWorker=!0),(0,i.CX)(e,y),g();const T=function(e,t){t||(0,c.R)(e,"api");const h={};var g=a.ee.get(e),p=g.get("tracer"),m="api-",v=m+"ixn-";function b(t,r,n,o){const a=(0,i.C5)(e);return null===r?delete a.jsAttributes[t]:(0,i.CX)(e,{...a,jsAttributes:{...a.jsAttributes,[t]:r}}),x(m,n,!0,o||null===r?"session":void 0)(t,r)}function y(){}["setErrorHandler","finished","addToTrace","inlineHit","addRelease"].forEach((e=>h[e]=x(m,e,!0,"api"))),h.addPageAction=x(m,"addPageAction",!0,n.D.pageAction),h.setCurrentRouteName=x(m,"routeName",!0,n.D.spa),h.setPageViewName=function(t,r){if("string"==typeof t)return"/"!==t.charAt(0)&&(t="/"+t),(0,i.OP)(e).customTransaction=(r||"http://custom.transaction")+t,x(m,"setPageViewName",!0)()},h.setCustomAttribute=function(e,t){let r=arguments.length>2&&void 0!==arguments[2]&&arguments[2];if("string"==typeof e){if(["string","number"].includes(typeof t)||null===t)return b(e,t,"setCustomAttribute",r);(0,f.Z)("Failed to execute setCustomAttribute.\nNon-null value must be a string or number type, but a type of was provided."))}else(0,f.Z)("Failed to execute setCustomAttribute.\nName must be a string type, but a type of was provided."))},h.setUserId=function(e){if("string"==typeof e||null===e)return b("enduser.id",e,"setUserId",!0);(0,f.Z)("Failed to execute setUserId.\nNon-null value must be a string type, but a type of was provided."))},h.interaction=function(){return(new y).get()};var w=y.prototype={createTracer:function(e,t){var r={},i=this,a="function"==typeof t;return(0,o.p)(v+"tracer",[(0,s.z)(),e,r],i,n.D.spa,g),function(){if(p.emit((a?"":"no-")+"fn-start",[(0,s.z)(),i,a],r),a)try{return t.apply(this,arguments)}catch(e){throw p.emit("fn-err",[arguments,this,"string"==typeof e?new Error(e):e],r),e}finally{p.emit("fn-end",[(0,s.z)()],r)}}}};function x(e,t,r,i){return function(){return(0,o.p)(l.xS,["API/"+t+"/called"],void 0,n.D.metrics,g),i&&(0,o.p)(e+t,[(0,s.z)(),...arguments],r?null:this,i,g),r?void 0:this}}function A(){r.e(439).then(r.bind(r,7438)).then((t=>{let{setAPI:r}=t;r(e),(0,c.L)(e,"api")})).catch((()=>(0,f.Z)("Downloading runtime APIs failed...")))}return["actionText","setName","setAttribute","save","ignore","onEnd","getContext","end","get"].forEach((e=>{w[e]=x(v,e,void 0,n.D.spa)})),h.noticeError=function(e,t){"string"==typeof e&&(e=new Error(e)),(0,o.p)(l.xS,["API/noticeError/called"],void 0,n.D.metrics,g),(0,o.p)("err",[e,(0,s.z)(),!1,t],void 0,n.D.jserrors,g)},d.il?(0,u.b)((()=>A()),!0):A(),h}(e,v);return(0,h.Qy)(e,T,"api"),(0,h.Qy)(e,A,"exposed"),(0,h.EZ)("activatedFeatures",p.T),T}},3325:(e,t,r)=>{r.d(t,{D:()=>n,p:()=>i});const n={ajax:"ajax",jserrors:"jserrors",metrics:"metrics",pageAction:"page_action",pageViewEvent:"page_view_event",pageViewTiming:"page_view_timing",sessionReplay:"session_replay",sessionTrace:"session_trace",spa:"spa"},i={[n.pageViewEvent]:1,[n.pageViewTiming]:2,[n.metrics]:3,[n.jserrors]:4,[n.ajax]:5,[n.sessionTrace]:6,[n.pageAction]:7,[n.spa]:8,[n.sessionReplay]:9}}},n={};function i(e){var t=n[e];if(void 0!==t)return t.exports;var o=n[e]={exports:{}};return r[e](o,o.exports,i),o.exports}i.m=r,i.d=(e,t)=>{for(var r in t)i.o(t,r)&&!i.o(e,r)&&Object.defineProperty(e,r,{enumerable:!0,get:t[r]})},i.f={},i.e=e=>Promise.all(Object.keys(i.f).reduce(((t,r)=>(i.f[r](e,t),t)),[])),i.u=e=>(({78:"page_action-aggregate",147:"metrics-aggregate",242:"session-manager",317:"jserrors-aggregate",348:"page_view_timing-aggregate",412:"lazy-feature-loader",439:"async-api",538:"recorder",590:"session_replay-aggregate",675:"compressor",733:"session_trace-aggregate",786:"page_view_event-aggregate",873:"spa-aggregate",898:"ajax-aggregate"}[e]||e)+"."+{78:"ac76d497",147:"3dc53903",148:"1a20d5fe",242:"2a64278a",317:"49e41428",348:"bd6de33a",412:"2f55ce66",439:"30bd804e",538:"1b18459f",590:"cf0efb30",675:"ae9f91a8",733:"83105561",786:"06482edd",860:"03a8b7a5",873:"e6b09d52",898:"998ef92b"}[e]+"-1.236.0.min.js"),i.o=(e,t)=>Object.prototype.hasOwnProperty.call(e,t),e={},t="NRBA:",i.l=(r,n,o,a)=>{if(e[r])e[r].push(n);else{var s,c;if(void 0!==o)for(var u=document.getElementsByTagName("script"),d=0;d {s.onerror=s.onload=null,clearTimeout(h);var i=e[r];if(delete e[r],s.parentNode&&s.parentNode.removeChild(s),i&&i.forEach((e=>e(n))),t)return t(n)},h=setTimeout(l.bind(null,void 0,{type:"timeout",target:s}),12e4);s.onerror=l.bind(null,s.onerror),s.onload=l.bind(null,s.onload),c&&document.head.appendChild(s)}},i.r=e=>{"undefined"!=typeof Symbol&&Symbol.toStringTag&&Object.defineProperty(e,Symbol.toStringTag,{value:"Module"}),Object.defineProperty(e,"__esModule",{value:!0})},i.j=364,i.p="https://js-agent.newrelic.com/",(()=>{var e={364:0,953:0};i.f.j=(t,r)=>{var n=i.o(e,t)?e[t]:void 0;if(0!==n)if(n)r.push(n[2]);else{var o=new Promise(((r,i)=>n=e[t]=[r,i]));r.push(n[2]=o);var a=i.p+i.u(t),s=new Error;i.l(a,(r=>{if(i.o(e,t)&&(0!==(n=e[t])&&(e[t]=void 0),n)){var o=r&&("load"===r.type?"missing":r.type),a=r&&r.target&&r.target.src;s.message="Loading chunk "+t+" failed.\n("+o+": "+a+")",s.name="ChunkLoadError",s.type=o,s.request=a,n[1](s)}}),"chunk-"+t,t)}};var t=(t,r)=>{var n,o,[a,s,c]=r,u=0;if(a.some((t=>0!==e[t]))){for(n in s)i.o(s,n)&&(i.m[n]=s[n]);if(c)c(i)}for(t&&t(r);u {i.r(o);var e=i(3325),t=i(5763);const r=Object.values(e.D);function n(e){const n={};return r.forEach((r=>{n[r]=function(e,r){return!1!==(0,t.Mt)(r,"".concat(e,".enabled"))}(r,e)})),n}var a=i(9144);var s=i(5546),c=i(385),u=i(8e3),d=i(5938),f=i(3960),l=i(50);class h extends d.W{constructor(e,t,r){let n=!(arguments.length>3&&void 0!==arguments[3])||arguments[3];super(e,t,r),this.auto=n,this.abortHandler,this.featAggregate,this.onAggregateImported,n&&(0,u.R)(e,r)}importAggregator(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:{};if(this.featAggregate||!this.auto)return;const r=c.il&&!0===(0,t.Mt)(this.agentIdentifier,"privacy.cookies_enabled");let n;this.onAggregateImported=new Promise((e=>{n=e}));const o=async()=>{let t;try{if(r){const{setupAgentSession:e}=await Promise.all([i.e(860),i.e(242)]).then(i.bind(i,3228));t=e(this.agentIdentifier)}}catch(e){(0,l.Z)("A problem occurred when starting up session manager. This page will not start or extend any session.",e)}try{if(!this.shouldImportAgg(this.featureName,t))return void(0,u.L)(this.agentIdentifier,this.featureName);const{lazyFeatureLoader:r}=await i.e(412).then(i.bind(i,8582)),{Aggregate:o}=await r(this.featureName,"aggregate");this.featAggregate=new o(this.agentIdentifier,this.aggregator,e),n(!0)}catch(e){(0,l.Z)("Downloading and initializing ".concat(this.featureName," failed..."),e),this.abortHandler?.(),n(!1)}};c.il?(0,f.b)((()=>o()),!0):o()}shouldImportAgg(r,n){return r!==e.D.sessionReplay||!1!==(0,t.Mt)(this.agentIdentifier,"session_trace.enabled")&&(!!n?.isNew||!!n?.state.sessionReplay)}}var g=i(7633),p=i(7894);class m extends h{static featureName=g.t9;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];if(super(r,n,g.t9,i),("undefined"==typeof PerformanceNavigationTiming||c.Tt)&&"undefined"!=typeof PerformanceTiming){const n=(0,t.OP)(r);n[g.Dz]=Math.max(Date.now()-n.offset,0),(0,f.K)((()=>n[g.qw]=Math.max((0,p.z)()-n[g.Dz],0))),(0,f.b)((()=>{const t=(0,p.z)();n[g.OJ]=Math.max(t-n[g.Dz],0),(0,s.p)("timing",["load",t],void 0,e.D.pageViewTiming,this.ee)}))}this.importAggregator()}}var v=i(1117),b=i(1284);class y extends v.w{constructor(e){super(e),this.aggregatedData={}}store(e,t,r,n,i){var o=this.getBucket(e,t,r,i);return o.metrics=function(e,t){t||(t={count:0});return t.count+=1,(0,b.D)(e,(function(e,r){t[e]=w(r,t[e])})),t}(n,o.metrics),o}merge(e,t,r,n,i){var o=this.getBucket(e,t,n,i);if(o.metrics){var a=o.metrics;a.count+=r.count,(0,b.D)(r,(function(e,t){if("count"!==e){var n=a[e],i=r[e];i&&!i.c?a[e]=w(i.t,n):a[e]=function(e,t){if(!t)return e;t.c||(t=x(t.t));return t.min=Math.min(e.min,t.min),t.max=Math.max(e.max,t.max),t.t+=e.t,t.sos+=e.sos,t.c+=e.c,t}(i,a[e])}}))}else o.metrics=r}storeMetric(e,t,r,n){var i=this.getBucket(e,t,r);return i.stats=w(n,i.stats),i}getBucket(e,t,r,n){this.aggregatedData[e]||(this.aggregatedData[e]={});var i=this.aggregatedData[e][t];return i||(i=this.aggregatedData[e][t]={params:r||{}},n&&(i.custom=n)),i}get(e,t){return t?this.aggregatedData[e]&&this.aggregatedData[e][t]:this.aggregatedData[e]}take(e){for(var t={},r="",n=!1,i=0;i t.max&&(t.max=e),e 2&&void 0!==arguments[2])||arguments[2];super(e,r,j.t,n),c.il&&((0,t.OP)(e).initHidden=Boolean("hidden"===document.visibilityState),(0,N.N)((()=>(0,s.p)("docHidden",[(0,p.z)()],void 0,j.t,this.ee)),!0),(0,O.bP)("pagehide",(()=>(0,s.p)("winPagehide",[(0,p.z)()],void 0,j.t,this.ee))),this.importAggregator())}}var P=i(3081);class C extends h{static featureName=P.t9;constructor(e,t){let r=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(e,t,P.t9,r),this.importAggregator()}}var R,I=i(2210),k=i(1214),H=i(2177),L={};try{R=localStorage.getItem("__nr_flags").split(","),console&&"function"==typeof console.log&&(L.console=!0,-1!==R.indexOf("dev")&&(L.dev=!0),-1!==R.indexOf("nr_dev")&&(L.nrDev=!0))}catch(e){}function z(e){try{L.console&&z(e)}catch(e){}}L.nrDev&&H.ee.on("internal-error",(function(e){z(e.stack)})),L.dev&&H.ee.on("fn-err",(function(e,t,r){z(r.stack)})),L.dev&&(z("NR AGENT IN DEVELOPMENT MODE"),z("flags: "+(0,b.D)(L,(function(e,t){return e})).join(", ")));var M=i(6660);class B extends h{static featureName=M.t;constructor(r,n){let i=!(arguments.length>2&&void 0!==arguments[2])||arguments[2];super(r,n,M.t,i),this.skipNext=0;try{this.removeOnAbort=new AbortController}catch(e){}const o=this;o.ee.on("fn-start",(function(e,t,r){o.abortHandler&&(o.skipNext+=1)})),o.ee.on("fn-err",(function(t,r,n){o.abortHandler&&!n[M.A]&&((0,I.X)(n,M.A,(function(){return!0})),this.thrown=!0,(0,s.p)("err",[n,(0,p.z)()],void 0,e.D.jserrors,o.ee))})),o.ee.on("fn-end",(function(){o.abortHandler&&!this.thrown&&o.skipNext>0&&(o.skipNext-=1)})),o.ee.on("internal-error",(function(t){(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,o.ee)})),this.origOnerror=c._A.onerror,c._A.onerror=this.onerrorHandler.bind(this),c._A.addEventListener("unhandledrejection",(t=>{const r=function(e){let t="Unhandled Promise Rejection: ";if(e instanceof Error)try{return e.message=t+e.message,e}catch(t){return e}if(void 0===e)return new Error(t);try{return new Error(t+(0,D.P)(e))}catch(e){return new Error(t)}}(t.reason);(0,s.p)("err",[r,(0,p.z)(),!1,{unhandledPromiseRejection:1}],void 0,e.D.jserrors,this.ee)}),(0,O.m$)(!1,this.removeOnAbort?.signal)),(0,k.gy)(this.ee),(0,k.BV)(this.ee),(0,k.em)(this.ee),(0,t.OP)(r).xhrWrappable&&(0,k.Kf)(this.ee),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}onerrorHandler(t,r,n,i,o){"function"==typeof this.origOnerror&&this.origOnerror(...arguments);try{this.skipNext?this.skipNext-=1:(0,s.p)("err",[o||new F(t,r,n),(0,p.z)()],void 0,e.D.jserrors,this.ee)}catch(t){try{(0,s.p)("ierr",[t,(0,p.z)(),!0],void 0,e.D.jserrors,this.ee)}catch(e){}}return!1}}function F(e,t,r){this.message=e||"Uncaught error with no additional information",this.sourceURL=t,this.line=r}let U=1;const q="nr@id";function G(e){const t=typeof e;return!e||"object"!==t&&"function"!==t?-1:e===c._A?0:(0,I.X)(e,q,(function(){return U++}))}function V(e){if("string"==typeof e&&e.length)return e.length;if("object"==typeof e){if("undefined"!=typeof ArrayBuffer&&e instanceof ArrayBuffer&&e.byteLength)return e.byteLength;if("undefined"!=typeof Blob&&e instanceof Blob&&e.size)return e.size;if(!("undefined"!=typeof FormData&&e instanceof FormData))try{return(0,D.P)(e).length}catch(e){return}}}var X=i(7243);class W{constructor(e){this.agentIdentifier=e,this.generateTracePayload=this.generateTracePayload.bind(this),this.shouldGenerateTrace=this.shouldGenerateTrace.bind(this)}generateTracePayload(e){if(!this.shouldGenerateTrace(e))return null;var r=(0,t.DL)(this.agentIdentifier);if(!r)return null;var n=(r.accountID||"").toString()||null,i=(r.agentID||"").toString()||null,o=(r.trustKey||"").toString()||null;if(!n||!i)return null;var a=(0,_.M)(),s=(0,_.Ht)(),c=Date.now(),u={spanId:a,traceId:s,timestamp:c};return(e.sameOrigin||this.isAllowedOrigin(e)&&this.useTraceContextHeadersForCors())&&(u.traceContextParentHeader=this.generateTraceContextParentHeader(a,s),u.traceContextStateHeader=this.generateTraceContextStateHeader(a,c,n,i,o)),(e.sameOrigin&&!this.excludeNewrelicHeader()||!e.sameOrigin&&this.isAllowedOrigin(e)&&this.useNewrelicHeaderForCors())&&(u.newrelicHeader=this.generateTraceHeader(a,s,c,n,i,o)),u}generateTraceContextParentHeader(e,t){return"00-"+t+"-"+e+"-01"}generateTraceContextStateHeader(e,t,r,n,i){return i+"@nr=0-1-"+r+"-"+n+"-"+e+"----"+t}generateTraceHeader(e,t,r,n,i,o){if(!("function"==typeof c._A?.btoa))return null;var a={v:[0,1],d:{ty:"Browser",ac:n,ap:i,id:e,tr:t,ti:r}};return o&&n!==o&&(a.d.tk=o),btoa((0,D.P)(a))}shouldGenerateTrace(e){return this.isDtEnabled()&&this.isAllowedOrigin(e)}isAllowedOrigin(e){var r=!1,n={};if((0,t.Mt)(this.agentIdentifier,"distributed_tracing")&&(n=(0,t.P_)(this.agentIdentifier).distributed_tracing),e.sameOrigin)r=!0;else if(n.allowed_origins instanceof Array)for(var i=0;i 2&&void 0!==arguments[2])||arguments[2];super(r,n,Z.t,i),(0,t.OP)(r).xhrWrappable&&(this.dt=new W(r),this.handler=(e,t,r,n)=>(0,s.p)(e,t,r,n,this.ee),(0,k.u5)(this.ee),(0,k.Kf)(this.ee),function(r,n,i,o){function a(e){var t=this;t.totalCbs=0,t.called=0,t.cbTime=0,t.end=E,t.ended=!1,t.xhrGuids={},t.lastSize=null,t.loadCaptureCalled=!1,t.params=this.params||{},t.metrics=this.metrics||{},e.addEventListener("load",(function(r){_(t,e)}),(0,O.m$)(!1)),c.IF||e.addEventListener("progress",(function(e){t.lastSize=e.loaded}),(0,O.m$)(!1))}function s(e){this.params={method:e[0]},T(this,e[1]),this.metrics={}}function u(e,n){var i=(0,t.DL)(r);i.xpid&&this.sameOrigin&&n.setRequestHeader("X-NewRelic-ID",i.xpid);var a=o.generateTracePayload(this.parsedOrigin);if(a){var s=!1;a.newrelicHeader&&(n.setRequestHeader("newrelic",a.newrelicHeader),s=!0),a.traceContextParentHeader&&(n.setRequestHeader("traceparent",a.traceContextParentHeader),a.traceContextStateHeader&&n.setRequestHeader("tracestate",a.traceContextStateHeader),s=!0),s&&(this.dt=a)}}function d(e,t){var r=this.metrics,i=e[0],o=this;if(r&&i){var a=V(i);a&&(r.txSize=a)}this.startTime=(0,p.z)(),this.listener=function(e){try{"abort"!==e.type||o.loadCaptureCalled||(o.params.aborted=!0),("load"!==e.type||o.called===o.totalCbs&&(o.onloadCalled||"function"!=typeof t.onload)&&"function"==typeof o.end)&&o.end(t)}catch(e){try{n.emit("internal-error",[e])}catch(e){}}};for(var s=0;s 1?e[1]=i:e.push(i)}else e[0]&&e[0].headers&&s(e[0].headers,n)&&(this.dt=n);function s(e,t){var r=!1;return t.newrelicHeader&&(e.set("newrelic",t.newrelicHeader),r=!0),t.traceContextParentHeader&&(e.set("traceparent",t.traceContextParentHeader),t.traceContextStateHeader&&e.set("tracestate",t.traceContextStateHeader),r=!0),r}}function x(e,t){this.params={},this.metrics={},this.startTime=(0,p.z)(),this.dt=t,e.length>=1&&(this.target=e[0]),e.length>=2&&(this.opts=e[1]);var r,n=this.opts||{},i=this.target;"string"==typeof i?r=i:"object"==typeof i&&i instanceof Y?r=i.url:c._A?.URL&&"object"==typeof i&&i instanceof URL&&(r=i.href),T(this,r);var o=(""+(i&&i instanceof Y&&i.method||n.method||"GET")).toUpperCase();this.params.method=o,this.txSize=V(n.body)||0}function A(t,r){var n;this.endTime=(0,p.z)(),this.params||(this.params={}),this.params.status=r?r.status:0,"string"==typeof this.rxSize&&this.rxSize.length>0&&(n=+this.rxSize);var o={txSize:this.txSize,rxSize:n,duration:(0,p.z)()-this.startTime};i("xhr",[this.params,o,this.startTime,this.endTime,"fetch"],this,e.D.ajax)}function E(t){var r=this.params,n=this.metrics;if(!this.ended){this.ended=!0;for(var o=0;o 2&&void 0!==arguments[2])||arguments[2];super(e,t,we.t,r),this.importAggregator()}}new class{constructor(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:(0,_.ky)(16);c._A?(this.agentIdentifier=t,this.sharedAggregator=new y({agentIdentifier:this.agentIdentifier}),this.features={},this.desiredFeatures=new Set(e.features||[]),this.desiredFeatures.add(m),Object.assign(this,(0,a.j)(this.agentIdentifier,e,e.loaderType||"agent")),this.start()):(0,l.Z)("Failed to initial the agent. Could not determine the runtime environment.")}get config(){return{info:(0,t.C5)(this.agentIdentifier),init:(0,t.P_)(this.agentIdentifier),loader_config:(0,t.DL)(this.agentIdentifier),runtime:(0,t.OP)(this.agentIdentifier)}}start(){const t="features";try{const r=n(this.agentIdentifier),i=[...this.desiredFeatures];i.sort(((t,r)=>e.p[t.featureName]-e.p[r.featureName])),i.forEach((t=>{if(r[t.featureName]||t.featureName===e.D.pageViewEvent){const n=function(t){switch(t){case e.D.ajax:return[e.D.jserrors];case e.D.sessionTrace:return[e.D.ajax,e.D.pageViewEvent];case e.D.sessionReplay:return[e.D.sessionTrace];case e.D.pageViewTiming:return[e.D.pageViewEvent];default:return[]}}(t.featureName);n.every((e=>r[e]))||(0,l.Z)("".concat(t.featureName," is enabled but one or more dependent features has been disabled (").concat((0,D.P)(n),"). This may cause unintended consequences or missing data...")),this.features[t.featureName]=new t(this.agentIdentifier,this.sharedAggregator)}})),(0,T.Qy)(this.agentIdentifier,this.features,t)}catch(e){(0,l.Z)("Failed to initialize all enabled instrument classes (agent aborted) -",e);for(const e in this.features)this.features[e].abortHandler?.();const r=(0,T.fP)();return delete r.initializedAgents[this.agentIdentifier]?.api,delete r.initializedAgents[this.agentIdentifier]?.[t],delete this.sharedAggregator,r.ee?.abort(),delete r.ee?.get(this.agentIdentifier),!1}}}({features:[J,m,S,class extends h{static featureName=oe;constructor(t,r){if(super(t,r,oe,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;const n=this.ee;let i;(0,k.QU)(n),this.eventsEE=(0,k.em)(n),this.eventsEE.on(se,(function(e,t){this.bstStart=(0,p.z)()})),this.eventsEE.on(ae,(function(t,r){(0,s.p)("bst",[t[0],r,this.bstStart,(0,p.z)()],void 0,e.D.sessionTrace,n)})),n.on(ce+ne,(function(e){this.time=(0,p.z)(),this.startPath=location.pathname+location.hash})),n.on(ce+ie,(function(t){(0,s.p)("bstHist",[location.pathname+location.hash,this.startPath,this.time],void 0,e.D.sessionTrace,n)}));try{i=new PerformanceObserver((t=>{const r=t.getEntries();(0,s.p)(te,[r],void 0,e.D.sessionTrace,n)})),i.observe({type:re,buffered:!0})}catch(e){}this.importAggregator({resourceObserver:i})}},C,xe,B,class extends h{static featureName=de;constructor(e,r){if(super(e,r,de,!(arguments.length>2&&void 0!==arguments[2])||arguments[2]),!c.il)return;if(!(0,t.OP)(e).xhrWrappable)return;try{this.removeOnAbort=new AbortController}catch(e){}let n,i=0;const o=this.ee.get("tracer"),a=(0,k._L)(this.ee),s=(0,k.Lg)(this.ee),u=(0,k.BV)(this.ee),d=(0,k.Kf)(this.ee),f=this.ee.get("events"),l=(0,k.u5)(this.ee),h=(0,k.QU)(this.ee),g=(0,k.Gm)(this.ee);function m(e,t){h.emit("newURL",[""+window.location,t])}function v(){i++,n=window.location.hash,this[ve]=(0,p.z)()}function b(){i--,window.location.hash!==n&&m(0,!0);var e=(0,p.z)();this[pe]=~~this[pe]+e-this[ve],this[ye]=e}function y(e,t){e.on(t,(function(){this[t]=(0,p.z)()}))}this.ee.on(ve,v),s.on(be,v),a.on(be,v),this.ee.on(ye,b),s.on(ge,b),a.on(ge,b),this.ee.buffer([ve,ye,"xhr-resolved"],this.featureName),f.buffer([ve],this.featureName),u.buffer(["setTimeout"+le,"clearTimeout"+fe,ve],this.featureName),d.buffer([ve,"new-xhr","send-xhr"+fe],this.featureName),l.buffer([me+fe,me+"-done",me+he+fe,me+he+le],this.featureName),h.buffer(["newURL"],this.featureName),g.buffer([ve],this.featureName),s.buffer(["propagate",be,ge,"executor-err","resolve"+fe],this.featureName),o.buffer([ve,"no-"+ve],this.featureName),a.buffer(["new-jsonp","cb-start","jsonp-error","jsonp-end"],this.featureName),y(l,me+fe),y(l,me+"-done"),y(a,"new-jsonp"),y(a,"jsonp-end"),y(a,"cb-start"),h.on("pushState-end",m),h.on("replaceState-end",m),window.addEventListener("hashchange",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("load",m,(0,O.m$)(!0,this.removeOnAbort?.signal)),window.addEventListener("popstate",(function(){m(0,i>1)}),(0,O.m$)(!0,this.removeOnAbort?.signal)),this.abortHandler=this.#e,this.importAggregator()}#e(){this.removeOnAbort?.abort(),this.abortHandler=void 0}}],loaderType:"spa"})})(),window.NRBA=o})(); window.jQuery || document.write(' ') CKEDITOR_BASEPATH='https://f1000research.com/js/vendor/ckeditor/' window.reactTheme = 'research'; window.MathJax = { CommonHTML: { linebreaks: { automatic: true } }, 'HTML-CSS': { linebreaks: { automatic: true } }, SVG: { linebreaks: { automatic: true } }, AuthorInit: function() { MathJax.Hub.Register.MessageHook('End Process', function () { let timeout = false; // holder for timeout id const delay = 250; // delay after event is "complete" to run callback const reflowMath = function() { const dispFormulas = document.querySelectorAll('.disp-formula.panel'); if (!dispFormulas) { return; } for (const dispFormula of dispFormulas) { const child = dispFormula.querySelector('.MathJax_Preview').nextSibling.firstChild; const isMultiline = MathJax.Hub.getAllJax(dispFormula)[0].root.isMultiline; if (dispFormula.offsetWidth < child.offsetWidth || isMultiline) { MathJax.Hub.Queue(['Rerender', MathJax.Hub, dispFormula]); } } }; window.addEventListener('resize', function() { clearTimeout(timeout); // clear the timeout timeout = setTimeout(reflowMath, delay); // start timing for event "completion" }); }); }, }; if (window.location.hash == '#_=_'){ window.location = window.location.href.split('#')[0] } !function(f,b,e,v,n,t,s){if(f.fbq)return;n=f.fbq=function() {n.callMethod? n.callMethod.apply(n,arguments):n.queue.push(arguments)} ;if(!f._fbq)f._fbq=n; n.push=n;n.loaded=!0;n.version='2.0';n.queue=[];t=b.createElement(e);t.async=!0; t.src=v;s=b.getElementsByTagName(e)[0];s.parentNode.insertBefore(t,s)}(window, document,'script','https://connect.facebook.net/en_US/fbevents.js'); fbq('init', '1641728616063202'); fbq('track', "PixelInitialized", {}); (function(h,o,t,j,a,r){ h.hj=h.hj||function(){(h.hj.q=h.hj.q||[]).push(arguments)}; h._hjSettings={hjid:2318163,hjsv:6}; a=o.getElementsByTagName('head')[0]; r=o.createElement('script');r.async=1; r.src=t+h._hjSettings.hjid+j+h._hjSettings.hjsv; a.appendChild(r); })(window,document,'https://static.hotjar.com/c/hotjar-','.js?sv='); search file_upload Submit your research search menu close search Browse Gateways & Collections How to Publish Submit your Research My Submissions Article Guidelines Article Guidelines (New Versions) Open Data, Software and Code Guidelines Open Data and Accessible Source Materials Guidelines (HSS) Open Data, Software and Code Guidelines (PSE) Prepublication Checks Production Process Posters and Slides Guidelines Document Guidelines Article Processing Charges Peer Review Finding Article Reviewers About How it Works For Reviewers Our Advisors Policies Glossary FAQs For Developers Newsroom Contact My Research Submissions Content and Tracking Alerts My Details Sign In file_upload Submit your research { "@context": "https://schema.org", "@type": "ScholarlyArticle", "mainEntityOfPage": { "@type": "WebPage", "@id": "https://f1000research.com/articles/8-446" }, "headline": "A Bioconductor workflow for the Bayesian analysis of spatial proteomics", "datePublished": "2019-04-11T14:50:05", "dateModified": "2019-04-11T14:50:05", "author": [ { "@type": "Person", "name": "Oliver M. Crook" }, { "@type": "Person", "name": "Lisa M. Breckels" }, { "@type": "Person", "name": "Kathryn S. Lilley" }, { "@type": "Person", "name": "Paul D.W. Kirk" }, { "@type": "Person", "name": "Laurent Gatto" } ], "publisher": { "@type": "Organization", "name": "F1000Research", "logo": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 480, "width": 60 } }, "image": { "@type": "ImageObject", "url": "https://f1000research.com/img/AMP/F1000Research_image.png", "height": 1200, "width": 150 }, "description": "Knowledge of the subcellular location of a protein gives valuable insight into its function. The field of spatial proteomics has become increasingly popular due to improved multiplexing capabilities in high-throughput mass spectrometry, which have made it possible to systematically localise thousands of proteins per experiment. In parallel with these experimental advances, improved methods for analysing spatial proteomics data have also been developed. In this workflow, we demonstrate using `pRoloc` for the Bayesian analysis of spatial proteomics data. We detail the software infrastructure and then provide step-by-step guidance of the analysis, including setting up a pipeline, assessing convergence, and interpreting downstream results. In several places we provide additional details on Bayesian analysis to provide users with a holistic view of Bayesian analysis for spatial proteomics data." } { "@context": "http://schema.org", "@type": "BreadcrumbList", "itemListElement": [ { "@type": "ListItem", "position": "1", "item": { "@id": "https://f1000research.com/", "name": "Home" } }, { "@type": "ListItem", "position": "2", "item": { "@id": "https://f1000research.com/browse/articles", "name": "Browse" } }, { "@type": "ListItem", "position": "3", "item": { "@id": "https://f1000research.com/articles/8-446/v1", "name": "A Bioconductor workflow for the Bayesian analysis of spatial proteomics" } } ] } Home Browse A Bioconductor workflow for the Bayesian analysis of spatial proteomics ALL Metrics - Views Downloads Get PDF Get XML Cite How to cite this article Crook OM, Breckels LM, Lilley KS et al. A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.12688/f1000research.18636.1 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. Close Copy Citation Details Export Export Citation Sciwheel EndNote Ref. Manager Bibtex ProCite Sente EXPORT Select a format first Track Share ▬ ✚ Method Article A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] Oliver M. Crook 1,2 , Lisa M. Breckels https://orcid.org/0000-0001-8918-7171 1 , Kathryn S. Lilley 1 , Paul D.W. Kirk 2 , Laurent Gatto https://orcid.org/0000-0002-1520-2268 3 Oliver M. Crook 1,2 , Lisa M. Breckels https://orcid.org/0000-0001-8918-7171 1 , [...] Kathryn S. Lilley 1 , Paul D.W. Kirk 2 , Laurent Gatto https://orcid.org/0000-0002-1520-2268 3 PUBLISHED 11 Apr 2019 Author details Author details 1 Cambridge Centre for Proteomics, Department of Biochemistry, University of Cambridge, Cambridge, CB2 1QR, UK 2 MRC Biostatistics Unit, Cambridge Institute for Public Health, Cambridge Institute for Public Health, Cambridge, CB2 0SR, UK 3 Catholic University of Louvain, Brussels, 1200, Belgium Oliver M. Crook Roles: Conceptualization, Data Curation, Formal Analysis, Funding Acquisition, Investigation, Methodology, Software, Validation, Visualization, Writing – Original Draft Preparation, Writing – Review & Editing Lisa M. Breckels Roles: Data Curation, Software, Visualization, Writing – Review & Editing Kathryn S. Lilley Roles: Funding Acquisition, Resources, Supervision, Writing – Review & Editing Paul D.W. Kirk Roles: Conceptualization, Investigation, Methodology, Supervision, Writing – Review & Editing Laurent Gatto Roles: Conceptualization, Data Curation, Investigation, Methodology, Project Administration, Resources, Software, Supervision, Validation, Writing – Review & Editing OPEN PEER REVIEW DETAILS REVIEWER STATUS This article is included in the Artificial Intelligence and Machine Learning gateway. This article is included in the RPackage gateway. This article is included in the Bioconductor gateway. This article is included in the Machine learning: life sciences collection. Abstract Knowledge of the subcellular location of a protein gives valuable insight into its function. The field of spatial proteomics has become increasingly popular due to improved multiplexing capabilities in high-throughput mass spectrometry, which have made it possible to systematically localise thousands of proteins per experiment. In parallel with these experimental advances, improved methods for analysing spatial proteomics data have also been developed. In this workflow, we demonstrate using `pRoloc` for the Bayesian analysis of spatial proteomics data. We detail the software infrastructure and then provide step-by-step guidance of the analysis, including setting up a pipeline, assessing convergence, and interpreting downstream results. In several places we provide additional details on Bayesian analysis to provide users with a holistic view of Bayesian analysis for spatial proteomics data. READ ALL READ LESS Keywords pRoloc, pRolocdata, proteomics, Bayesian, software, Bioconductor, spatial proteomics, machine learning Corresponding Author(s) Laurent Gatto ( [email protected] ) Close Corresponding author: Laurent Gatto Competing interests: No competing interests were disclosed. Grant information: PDWK was supported by the MRC (project reference MC_UU_00002/10). LMB was supported by a BBSRC Tools and Resources Development grant (Award BB/N023129/1) and a Wellcome Trust Technology Development Grant (Grant number 108441/Z/15/Z). OMC is a Wellcome Trust Mathematical Genomics and Medicine student supported financially by the School of Clinical Medicine, University of Cambridge. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript. Copyright: © 2019 Crook OM et al . This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. How to cite: Crook OM, Breckels LM, Lilley KS et al. A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.12688/f1000research.18636.1 ) First published: 11 Apr 2019, 8 :446 ( https://doi.org/10.12688/f1000research.18636.1 ) Latest published: 11 Apr 2019, 8 :446 ( https://doi.org/10.12688/f1000research.18636.1 ) Introduction Determining the the spatial subcellular distribution of proteins enables novel insight into protein function 1 . Many proteins function within a single location within the cell; however, it is estimated that up to half of the proteome is thought to reside in multiple locations, with some of these undergoing dynamic relocalisation 2 . These phenomena lead to variability and uncertainty in robustly assigning proteins to a unique localisation. Functional compartmentalisation of proteins allows the cell to control biomolecular pathways and biochemical processes within the cell. Therefore, proteins with multiple localisations may have multiple functional roles 3 . Machine learning algorithms that fail to quantify uncertainty are unable to draw deeper insight into understanding cell biology from mass spectrometry (MS)-based spatial proteomics experiments. Hence, quantifying uncertainty allows us to make rigorous assessments of protein subcellular localisation and multi-localisation. For proteins to carry out their functional role they must be localised to the correct subcellular compartment, ensuring the biochemical conditions for desired molecular interactions are met 4 . Many pathologies, including cancer and obesity are characterised by protein mis-localisations 5 – 14 . High-throughput spatial proteomics technologies have seen rapid improvement over the last decade and now a single experiment can provide spatial information on thousands of proteins at once 15 – 18 . As a result of these spatial proteomics technologies many biological systems have been characterised 2 , 15 , 17 , 19 – 21 . The popularity of such methods is now evident with many new studies in recent years 17 , 22 – 29 . Bayesian approaches to machine learning and statistics can provide more insight, by providing uncertainty quantification 30 . In a parametric Bayesian setting, a parametric model is proposed, along with a statement about our prior beliefs of the model parameters. Bayes’ theorem tells us how to update the prior distribution of the parameters to obtain the posterior distribution of the parameters after observing the data. It is the posterior distribution which quantifies the uncertainty in the parameters. This contrasts from a maximum-likelihood approach where we obtain only a point estimate of the parameters. Adopting a Bayesian framework for data analysis, though of much interest to experimentalists, can be challenging. Once we have specified a probabilistic model, computational approaches are typically used to obtain the posterior distribution upon observation of the data. These algorithms can have parameters that require tuning and a variety of settings, hindering their practical use by those not familiar with Bayesian methodology. Even once the algorithms have been correctly set-up, assessments of convergence and guidance on how to interpret the results are often sparse. This workflow presents a Bayesian analysis of spatial proteomics to elucidate the process for practitioners. Our workflow also provides a template for others interested in designing tools for the biological community which rely on Bayesian inference. Our model for the data is the t-augmented Gaussian mixture (TAGM) model proposed in 1 . Crook et al . 1 provide a detailed description of the model, rigorous comparisons and testing on many spatial proteomics datasets, including a case study in which a hyperLOPIT experiment is performed on mouse pluripotent stem cells 17 , 31 . Revisiting these details is not the purpose of this computational protocol; rather we present how to correctly use the software and provide step-by-step guidance for interpreting the results. In brief, the TAGM model posits that each annotated sub-cellular niche can be modelled using a Gaussian distribution. Thus the full complement of proteins within the cell is captured as a mixture of Gaussians. The highly dynamic nature of the cell means that many proteins are not well captured by any of these multivariate Gaussian distributions, and thus the model also includes an outlier component, which is mathematically described as a multivariate student’s t distribution. The heavy tails of the t distribution allow it to better capture dispersed proteins. There are two approaches to perform inference in the TAGM model. The first, which we refer to as TAGM MAP, allows us to obtain maximum a posteriori estimates of posterior localisation probabilities; that is, the modal posterior probability that a protein localises to that class. This approach uses the expectation-maximisation (EM) algorithm to perform inference 32 . Whilst this is a interpretable summary of the TAGM model, it only provides point estimates. For a richer analysis, we also present a Markov-chain Monte-Carlo (MCMC) method to perform fully Bayesian inference in our model, allowing us to obtain full posterior localisation distributions. This method is referred to as TAGM MCMC throughout the text. This workflow begins with a brief review of some of the basic features of mass spectrometry-based spatial proteomics data, including our state-of-the-art computational infrastructure and bespoke software suite. We then present each method in turn, detailing how to obtain high quality results. We provide an extended discussion of the TAGM MCMC method to highlight some of the challenges that may arise when applying this method. This includes how to assess convergence of MCMC methods, as well as methods for manipulating the output. We then take the processed output and explain how to interpret the results, as well as providing some tools for visualisation. We conclude with some remarks and directions for the future. Source code for this workflow, including code used to generate tables and figures, is available on GitHub 33 Getting started and infrastructure In this workflow, we are using version 1.23.2 of pRoloc 34 . The package pRoloc contains algorithms and methods for analysing spatial proteomics data, building on the MSnSet structure provided in MSnbase . The pRolocdata package provides many annotated datasets from a variety of species and experimental procedures. The following code chunks install and load the suite of packages require for the analysis. if ( ! require ( "BiocManager" )) install.package ( "BiocManager" ) BiocManager :: install ( c ( "pRoloc" , "pRolocdata" )) library ( "pRoloc" ) ## ## This is pRoloc version 1.23.2 ## Visit https://lgatto.github.io/pRoloc/ to get started. library ( "pRolocdata" ) ## ## This is pRolocdata version 1.21.1. ## Use ’pRolocdata()’ to list available data sets. We assume that we have a MS-based spatial proteomics dataset contained in a MSnSet structure. For information on how to import data, perform basic data processing, quality control, supervised machine learning and transfer learning we refer the reader to 35 . Here, we start by loading a spatial proteomics dataset on mouse E14TG2a embryonic stem cells 36 . The LOPIT protocol 15 , 37 was used and the normalised intensity of proteins from eight iTRAQ 8-plex labelled fraction are provided. The methods provided here are independent of labelling procedure, fractionation process or workflow. Examples of valid experimental protocols are LOPIT 37 , hyperLOPIT 17 , 31 , label-free methods such as PCP 16 , and when fractionation is perform by differential centrifugation 18 , 38 . In the code chunk below, we load the aforementioned dataset. The printout demonstrates that this experiment quantified 2031 proteins over 8 fractions. data ( "E14TG2aR" ) # load experimental data E14TG2aR ## MSnSet (storageMode: lockedEnvironment) ## assayData: 2031 features, 8 samples ## element names: exprs ## protocolData: none ## phenoData ## sampleNames: n113 n114 ... n121 (8 total) ## varLabels: Fraction.information ## varMetadata: labelDescription ## featureData ## featureNames: Q62261 Q9JHU4 ... Q9EQ93 (2031 total) ## fvarLabels: Uniprot.ID UniprotName ... markers (8 total) ## fvarMetadata: labelDescription ## experimentData: use ’experimentData(object)’ ## Annotation: ## - - - Processing information - - - ## Loaded on Thu Jul 16 15:02:29 2015. ## Normalised to sum of intensities. ## Added markers from ’mrk’ marker vector. Thu Jul 16 15:02:29 2015 ## MSnbase version: 1.17.12 In Figure 1 , we can visualise the mouse stem cell dataset use the plot2D function. We observe that some of the organelle classes overlap and this is a typical feature of biological datasets. Thus, it is vital to perform uncertainty quantification when analysing biological data. Figure 1. First two principal components of mouse stem cell data. plot2D (E14TG2aR) addLegend (E14TG2aR, where = "topleft" , cex = 0.6 ) Methods: TAGM MAP Introduction to TAGM MAP We can use maximum a posteriori (MAP) estimation to perform Bayesian parameter estimation for our model. The maximum a posteriori estimate is the mode of the posterior distribution and can be used to provide a point estimate summary of the posterior localisation probabilities. In contrast to TAGM MCMC (see later), it does not provide samples from the posterior distribution, however it allows for faster inference by using an extended version of the expectation-maximisation (EM) algorithm. The EM algorithm iterates between an expectation step and a maximisation step. This allows us to find parameters which maximise the logarithm of the posterior, in the presence of latent (unobserved) variables. The EM algorithm is guaranteed to converge to a local mode. The code chunk below executes the tagmMapTrain function for a default of 100 iterations. We use the default priors for simplicity and convenience, however they can be changed, which we explain in a later section. The output is an object of class MAPParams , that captures the details of the TAGM MAP model. set.seed ( 2 ) mappars <- tagmMapTrain (E14TG2aR) ## co-linearity detected; a small multiple of ## the identity was added to the covariance mappars ## Object of class "MAPParams" ## Method: MAP Aside: collinearity The previous code chunk outputs a message concerning data collinearity. This is because the covariance matrix of the data has become ill-conditioned and as a result the inversion of this matrix becomes unstable with floating point arithmetic. This can lead to the failure of standard matrix algorithms upon which our method depends. In this case, it is standard practice to add a small multiple of the identity to stabilise this matrix. The printed message is a statement that this operation has been performed for these data. Model visualisation The results of the modelling can be visualised with the plotEllipse function on Figure 2 . The outer ellipse contains 99% of the total probability whilst the middle and inner ellipses contain 95% and 90% of the probability respectively. The centres of the clusters are represented by black circumpunct (circled dot). We can also plot the model in other principal components. The code chunk below plots the probability ellipses along the first and second, as well as the fourth principal component. The user can change the components visualised by altering the dims argument. Figure 2. PCA plot with probability ellipses along PC 1 and 2 (left) and PC 1 and 4 (right). par ( mfrow = c ( 1 , 2 )) plotEllipse (E14TG2aR, mappars) plotEllipse (E14TG2aR, mappars, dims = c ( 1 , 4 )) The expectation-maximisation algorithm The EM algorithm is iterative; that is, the algorithm iterates between an expectation step and a maximisation step until the value of the log-posterior does not change 32 . This fact can be used to assess the convergence of the EM algorithm. The value of the log-posterior at each iteration can be accessed with the logPosteriors function on the MAPParams object. The code chuck below plots the log posterior at each iteration and we see on Figure 3 the algorithm rapidly plateaus and so we have achieved convergence. If convergence has not been reached during this time, we suggest to increase the number of iterations by changing the parameter numIter in the tagmMapTrain method. In practice, it is not unexpected to observe small fluctuations due to numerical errors and this should not concern users. Figure 3. Log-posterior at each iteration of the EM algorithm demonstrating convergence. plot ( logPosteriors (mappars), type = "b" , col = "blue" , cex = 0.3 , ylab = "log-posterior" , xlab = "iteration" ) The code chuck below uses the mappars object generated above, along with the E14RG2aR dataset, to classify the proteins of unknown localisation using tagmPredict function. The results of running tagmPredict are appended to the fData columns of the MSnSet . E14TG2aR <- tagmPredict (E14TG2aR, mappars) # Predict protein localisation The new feature variables that are generated are: tagm.map.allocation: the TAGM MAP predictions for the most probable protein sub-cellular allocation. table ( fData (E14TG2aR) $ tagm.map.allocation) ## ## 40S Ribosome 60S Ribosome Cytosol ## 34 85 328 ## Endoplasmic reticulum Lysosome Mitochondrion ## 284 147 341 ## Nucleus - Chromatin Nucleus - Nucleolus Plasma membrane ## 143 322 326 ## Proteasome ## 21 tagm.map.probability: the posterior probability for the protein sub-cellular allocations. summary ( fData (E14TG2aR) $ tagm.map.probability) ## Min. 1st Qu. Median Mean 3rd Qu. Max. ## 0.00000 0.06963 0.93943 0.63829 0.99934 1.00000 tagm.map.outlier: the posterior probability for that protein to belong to the outlier component rather than any annotated component. summary ( fData (E14TG2aR) $ tagm.map.outlier) ## Min. 1st Qu. Median Mean 3rd Qu. Max. ## 0.0000000 0.0002363 0.0305487 0.3452624 0.9249810 1.0000000 We can visualise the results by scaling the pointer according the posterior localisation probabilities. To do this we extract the MAP localisation probabilities from the feature columns of the the MSnSet and pass these to the plot2D function ( Figure 4 ). Figure 4. TAGM MAP allocations, where the pointer is scaled according to the localisation probability and coloured according to the most probable subcellular niche. ptsze <- fData (E14TG2aR) $ tagm.map.probability # Scale pointer size plot2D (E14TG2aR, fcol = "tagm.map.allocation" , cex = ptsze) addLegend (E14TG2aR, where = "topleft" , cex = 0.6 , fcol = "tagm.map.allocation" ) The TAGM MAP method is easy to use and it is simple to check convergence, however it is limited in that it can only provide point estimates of the posterior localisation distributions. To obtain the full posterior distributions and therefore a rich analysis of the data, we use Markov-Chain Monte-Carlo methods. In our particular case, we use a collapsed Gibbs sampler 39 . Methods: TAGM MCMC a brief overview The TAGM MCMC method allows a fully Bayesian analysis of spatial proteomics datasets. It employs a collapsed Gibbs sampler to sample from the posterior distribution of localisation probablities, providing a rich analysis of the data. This section demonstrates the advantage of taking a Bayesian approach and the biological information that can be extracted from this analysis. For those unfamiliar with Bayesian methodology, some of the key ideas for a more complete understanding are as follows. Firstly, MCMC based inference contrasts with MAP based inference in that it samples from the posterior distribution of localisation probabilities. Hence, we do not just have a single estimate for each quantity but a distribution of estimates. MCMC methods are a large class of algorithms used to sample from a probability distribution, in our case the posterior distribution of the parameters 40 . Once we have sampled from the posterior distribution, we can estimate the mean of the posterior distribution by simply taking the mean of the samples. In a similar fashion, we can obtain estimates of other summaries of the posterior distribution. A schematic of MCMC sampling is provided in Figure 5 to aid understanding. Proteins, coloured blue, are visualised along two variables of the data. Probability ellipses representing contours of a probability distribution matching the distribution of the proteins are overlaid. We now wish to obtain samples from this distribution. The MCMC algorithm is initialised with a starting location, then at each iteration a new value is proposed. These proposed values are either accepted or rejected (according to a carefully computed acceptance probability) and over many iterations the algorithm converges and produces samples from the desired distribution. Samples from the mean of this distribution are coloured in red in the schematic figure. A large portion of the earlier samples may not reflect the true distribution, because the MCMC sampler has yet to converge. These early samples are usually discarded and this is referred to as burn-in. The next state of the algorithm depends on its current state and this leads to auto-correlation in the samples. To suppress this auto-correlation, we only retain every r th sample. This is known as thinning. The details of burn-in and thinning are further explained in later sections. Figure 5. A schematic figure of MCMC sampling. Proteins are coloured in blue and probability ellipses are overlaid representing contours of a probability distribution matching the distribution of the proteins. MCMC samples from the mean of this distribution are then coloured in red. The TAGM MCMC method is computationally intensive and requires at least modest processing power. Leaving the MCMC algorithm to run overnight on a modern desktop is usually sufficient, however this, of course, depends on the particular dataset being analysed. For guidance: it should not be expected that the analysis will finish in just a couple of hours on a medium specification laptop, for example. To demonstrate the class structure and expected outputs of the TAGM MCMC method, we run a brief analysis on a subset (400 randomly chosen proteins) of the tan2009r1 dataset from the pRolocdata , purely for illustration. This is to provide a bare bones analysis of these data without being held back by computational requirements. We perform a complete demonstration and provide precise details of the analysis of the stem cell dataset considered above in the next section. set.seed ( 1 ) data (tan2009r1) tan2009r1 <- tan2009r1 [ sample ( nrow (tan2009r1), 400 ), ] The first step is to run a few MCMC chains (below we use only 2 chains) for a few iterations (we specify 3 iterations in the below code, but typically we would suggest in the order of tens of thousands; see for example the algorithms default settings by typing ?tagmMcmcTrain) using the tagmMcmcTrain function. This function will generate a object of class MCMCParams . p <- tagmMcmcTrain ( object = tan2009r1, numIter = 3 , burnin = 1 , thin = 1 , numChains = 2 ) p ## Object of class "MCMCParams" ## Method: TAGM.MCMC ## Number of chains: 2 Information for each MCMC chain is contained within the chains slot. If needed, this information can be accessed manually. The function tagmMcmcProcess processes the MCMCParams object and populates the summary slot. p <- tagmMcmcProcess (p) p ## Object of class "MCMCParams" ## Method: TAGM.MCMC ## Number of chains: 2 ## Summary available The summary slot has now been populated to include basic summaries of the MCMC chains, such as organelle allocations and localisation probabilities. Protein information can be appended to the feature columns of the MSnSet by using the tagmPredict function, which extracts the required information from the summary slot of the MCMCParams object. res <- tagmPredict ( object = tan2009r1, params = p) We can now access new variables: tagm.mcmc.allocation: the TAGM MCMC prediction for the most likely protein sub-cellular annotation. table ( fData (res) $ tagm.mcmc.allocation) ## ## Cytoskeleton ER Golgi Lysosome mitochondrion ## 11 98 22 9 40 ## Nucleus Peroxisome PM Proteasome Ribosome 40S ## 25 3 104 29 31 ## Ribosome 60S ## 28 tagm.mcmc.probability: the mean posterior probability for the protein sub-cellular allocations. summary ( fData ( res) $ tagm.mcmc.probability) ## Min. 1st Qu. Median Mean 3rd Qu. Max. ## 0.3035 0.8974 0.9889 0.9088 1.0000 1.0000 We can also access other useful summaries of the MCMC methods: tagm.mcmc.outlier the posterior probability for the protein to belong to the outlier component. tagm.mcmc.probability.lowerquantile and tagm.mcmc.probability.upperquantile are the lower and upper boundaries to the equi-tailed 95% credible interval of tagm.mcmc.probability . tagm.mcmc.mean.shannon a Monte-Carlo averaged Shannon entropy, which is a measure of uncertainty in the allocations. Methods: TAGM MCMC the details This section explains how to manually manipulate the MCMC output of the TAGM model. In the code chunk below, we load a pre-computed TAGM MCMC model. The data file e14tagm.rda is available online 1 and is not directly loaded into this package due to its size. The file itself if around 500mb, which is too large to directly load into a package. load ( "e14Tagm.rda" ) The following code, which is not evaluated dynamically, was used to produce the tagmE14 MCMCParams object. We run the MCMC algorithm for 20,000 iterations with 10,000 iterations discarded for burn-in. We then thin the chain by 20. We ran 6 chains in parallel and so we obtain 500 samples for each of the 6 chains, totalling 3,000 samples. The resulting file is assumed to be in our working directory. e14Tagm <- tagmMcmcTrain (E14TG2aR, numIter = 20000 , burnin = 10000 , thin = 20 , numChains = 6 ) Manually inspecting the object, we see that it is a MCMCParams object with 6 chains. e14Tagm ## Object of class "MCMCParams" ## Method: TAGM.MCMC ## Number of chains: 6 Data exploration and convergence diagnostics Assessing whether or not an MCMC algorithm has converged is challenging. Assessing and diagnosing convergence is an active area of research and throughout the 1990s many approaches were proposed 41 – 44 . We provide a more detailed exploration of this issue, but readers should bare in mind that the methods provided below are diagnostics and cannot guarantee convergence. We direct readers to several important works in the literature discussing the assessment of convergence. Users that do not assess convergence and base their downstream analysis on unconverged chains are likely to obtain poor quality results. We first assess convergence using a parallel chains approach. We find producing multiple chains is benificial not only for computational advantages but also for analysis of convergence of our chains. ## Get number of chains nChains <- length (e14Tagm) nChains ## [1] 6 The following code chunks set up a manual convergence diagnostic check. We make use of objects and methods in the package coda to perform this analysis 45 . Our function below automatically coerces our objects into coda for ease of analysis. We first calculate the total number of outliers at each iteration of each chain and, if the algorithm has converged, this number should be the same (or very similar) across all 6 chains. ## Convergence diagnostic to see if we need to discard any ## iterations or entire chains: compute the number of outliers for ## each iteration for each chain out <- mcmc_get_outliers (e14Tagm) We can observe this from the trace plots and histograms for each MCMC chain ( Figure 6 ). Unconverged chains should be discarded from downstream analysis. ## Using coda S3 objects to produce trace plots and histograms for (i in seq_len (nChains)) plot (out[[i]], main = paste ( "Chain" , i), auto.layout = FALSE , col = i) Figure 6. Trace (left) and density (right) of the 6 MCMC chains. Chains 3, 5 and 6 are centred around an average of 153, with rapid back and forth oscillations. Chain 2 should be immediately discarded, since it has a large jump in the chain with clearly skewed histogram. The other two chains oscillate differently with contrasting quantiles to the 3 chains (3, 5 and 6) that agree with one another, suggesting these chains have yet to converge. We can use the coda package to produce summaries of our chains. Here is the coda summary for the third chain. ## Chains average around 153 outliers summary (out[[ 3 ]]) ## ## Iterations = 1:500 ## Thinning interval = 1 ## Number of chains = 1 ## Sample size per chain = 500 ## ## 1. Empirical mean and standard deviation for each variable, ## plus standard error of the mean: ## ## Mean SD Naive SE Time-series SE ## 153.4520 14.0771 0.6295 0.6820 ## ## 2. Quantiles for each variable: ## ## 2.5% 25% 50% 75% 97.5% ## 127 144 153 162 183 Applying the Gelman diagnostic So far, our analysis appears promising. Three of our chains are centred around an average of 153 outliers and there is no observed monotonicity in our output. However, for a more rigorous and unbiased analysis of convergence we can calculate the Gelman diagnostic using the coda package 42 , 44 . This statistic is often referred to as R ^ or the potential scale reduction factor. The idea of the Gelman diagnostics is to compare the inter and intra chain variances. The ratio of these quantities should be close to one. A more detailed and in depth discussion can be found in the references. The coda package also reports the 95% upper confidence interval of the R ^ statistic. In this case, our samples are approximately normally distributed (see histograms on the right in Figure 6 ). The coda package allows for transformations to improve normality of the data, and in some cases we set the transform argument to apply log transformation. Gelman and Rubin 42 suggest that chains with R ^ value of less than 1.2 are likely to have converged. gelman.diag (out, transform = FALSE ) ## Potential scale reduction factors: ## ## Point est. Upper C.I. ## [1,] 1.14 1.32 gelman.diag (out[ c ( 1 , 3 , 4 , 5 , 6 )], transform = FALSE ) ## Potential scale reduction factors: ## ## Point est. Upper C.I. ## [1,] 1.13 1.31 gelman.diag (out[ c ( 3 , 5 , 6 )], transform = FALSE ) ## Potential scale reduction factors: ## ## Point est. Upper C.I. ## [1,] 1 1.01 In all cases, we see that the Gelman diagnostic for convergence is < 1.2. However, the upper confidence interval is 1.32 when all chains are used; 1.31 when chain 2 is removed and when chains 1, 2 and 4 are removed the upper confidence interval is 1.01 indicating that the MCMC algorithm for chains 3,5 and 6 might have converged. We can also look at the Gelman diagnostics statistics for groups or pairs of chains. The first line below computes the Gelman diagnostic across the first three chains, whereas the second calculates the diagnostic between chain 3 and chain 5. gelman.diag (out[ 1 : 3 ], transform = FALSE ) # the upper C.I is 1.62 ## Potential scale reduction factors: ## ## Point est. Upper C.I. ## [1,] 1.22 1.62 gelman.diag (out[ c ( 3 , 5 )], transform = TRUE ) # the upper C.I is 1.01 ## Potential scale reduction factors: ## ## Point est. Upper C.I. ## [1,] 1.01 1.01 To assess another summary statistic, we can look at the mean component allocation at each iteration of the MCMC algorithm and as before we produce trace plots of this quantity ( Figure 7 ). Figure 7. Trace (left) and density (right) of the mean component allocation of the 6 MCMC chains. meanAlloc <- mcmc_get_meanComponent (e14Tagm) for (i in seq_len (nChains)) plot (meanAlloc[[i]], main = paste ( "Chain" , i), auto.layout = FALSE , col = i) As before we can produce summaries of the data. summary (meanAlloc[[ 1 ]]) ## ## Iterations = 1:500 ## Thinning interval = 1 ## Number of chains = 1 ## Sample size per chain = 500 ## ## 1. Empirical mean and standard deviation for each variable, ## plus standard error of the mean: ## ## Mean SD Naive SE Time-series SE ## 5.686713 0.059112 0.002644 0.002644 ## ## 2. Quantiles for each variable: ## ## 2.5% 25% 50% 75% 97.5% ## 5.552 5.646 5.692 5.728 5.795 We can already observe that there are some slight difference between these chains which raises suspicion that some of the chains may not have converged. For example each chain appears to be centred around 5.7, but chains 2 and 4 have clear jumps in the their trace plots. For a more quantitative analysis, we again apply the Gelman diagnostics to these summaries. gelman.diag (meanAlloc) ## Potential scale reduction factors: ## ## Point est. Upper C.I. ## [1,] 1 1.01 The above values are close to 1 and so we there are no significant difference between the chains. As observed previously, chains 2 and 4 look quite different from the other chains and so we recalculate the diagnostic excluding these chains. The computed Gelman diagnostic below suggest that chains 3, 5 and 6 have converged and that we should discard chains 1, 2 and 4 from further analysis. gelman.diag (meanAlloc[ c ( 3 , 5 , 6 )]) ## Potential scale reduction factors: ## ## Point est. Upper C.I. ## [1,] 1 1 For a further check, we can look at the mean outlier probability at each iteration of the MCMC algorithm and again computing the Gelman diagnostics between chains 4, 5 and 6. An R ^ statistics of 1 is indicative of convergence, since it is less than the recommend value of 1.2. meanoutProb <- mcmc_get_meanoutliersProb (e14Tagm) gelman.diag (meanoutProb[ c ( 3 , 5 , 6 )]) ## Potential scale reduction factors: ## ## Point est. Upper C.I. ## [1,] 1 1.01 Applying the Geweke diagnostic Along with the Gelman diagnostic, which uses parallel chains, we can also apply a single chain analysis using the Geweke diagnostic 41 . The Geweke diagnostic tests to see whether the mean calculated from the first 10% of iterations is significantly different from the mean calculated from the last 50% of iterations. If they are significantly different, at say a level 0.01, then this is evidence that particular chains have not converged. The following code chunk calculates the Geweke diagnostic for each chain on the summarising quantities we have previously computed. geweke_test (out) ## chain 1 chain 2 chain 3 chain 4 chain 5 chain 6 ## z.value 0.5749775 8.816632e+00 0.470203 -0.3204500 -0.6270787 -0.7328168 ## p.value 0.5653065 1.179541e-18 0.638210 0.7486272 0.5306076 0.4636702 geweke_test (meanAlloc) ## chain 1 chain 2 chain 3 chain 4 chain 5 chain 6 ## z.value 1.1952967 -3.3737051063 -1.2232102 2.48951993 0.3605882 -0.1358850 ## p.value 0.2319711 0.0007416377 0.2212503 0.01279157 0.7184073 0.8919122 geweke_test (meanoutProb) ## chain 1 chain 2 chain 3 chain 4 chain 5 chain 6 ## z.value 0.1785882 1.205500e+01 0.6189637 -0.5164987 -0.2141086 -0.02379004 ## p.value 0.8582611 1.825379e-33 0.5359403 0.6055062 0.8304624 0.98102008 The first test suggests chain 2 has not converged, since the p-value is less than 10 −10 suggesting that the mean in the first 10% of iterations is significantly different from those in the final 50%. Moreover, the second test and third tests also suggest that chain 2 has not converged. Furthermore, for the second test chain 4 has a marginally small p-value, providing further evidence that this chain is of low quality. These convergence diagnostics are not limited to the quantities we have computed here and further diagnostics can be performed on any summary of the data. An important question to consider is whether removing an early portion of the chain might lead to an improvement of the convergence diagnostics. This might be particularly relevant if a chain converges some iterations after our orginally specified burn-in . For example, let us take the second Geweke test above, which suggested chains 2 and 4 had not converged and see if discarding the initial 10% of the chain improves the statistic. The function below removes 50 samples, known as burn-in , from the beginning of each chain and the output shows that we now have 450 samples in each chain. In practice, as 2 chains are sufficient for good posterior estimates and convergence we could simply discard chains 2 and 4 and proceed with downstream analysis with the remaining chains. burn_e14Tagm <- mcmc_burn_chains (e14Tagm, 50 ) chains (burn_e14Tagm) ## Object of class "MCMCChains" ## Number of chains: 6 chains (burn_e14Tagm)[[ 4 ]] ## Object of class "MCMCChain" ## Number of components: 10 ## Number of proteins: 1663 ## Number of iterations: 450 The following function recomputes the number of outliers in each chain at each iteration of each Markov-chain. out2 <- mcmc_get_outliers (burn_e14Tagm) The code chuck below computes the Geweke diagnostic for this new truncated chain and demonstrates that chain 4 has an improved Geweke diagnostic, whilst chain 2 does not. Thus, in practice, it maybe useful to remove iterations from the beginning of the chain. However, as chain 4 did not pass the Gelman diagnostics we still discard it from downstream analysis. geweke_test (out2) ## chain 1 chain 2 chain 3 chain 4 chain 5 chain 6 ## z.value -0.1455345 6.379618e+00 -1.6392215 0.3836940 0.1241201 0.6654703 ## p.value 0.8842889 1.775298e-10 0.1011671 0.7012053 0.9012202 0.5057497 Processing converged chains Having made an assessment of convergence, we decide to discard chains 1,2 and 4 from any further analysis. The code chunk below removes these chains and creates a new object to store the converged chains. removeChain <- c ( 1 , 2 , 4 ) # The chains to be removed e14Tagm_converged <- e14Tagm[ - removeChain] # Create new object The MCMCParams object can be large and therefore if we have a large number of samples we may want to subsample our chain, known as thinning , to reduce the number of samples. Thinning also has another purpose. We may desire independent samples from our posterior distribution but the MCMC algorithm produces autocorrelated samples. Thinning can be applied to reduce the auto-correlation between samples. The code chuck below, which is not evaluated, demonstrates retaining every 5 th iteration. Recall that we thinned by 20 when we first ran the MCMC algorithm. e14Tagm_converged_thinned <- mcmc_thin_chains (e14Tagm_converged, freq = 5 ) We initially ran 6 chains and, after having made an assessment of convergence, we decided to discard 3 of the chains. We desire to make inference using samples from all 3 chains, since this leads to better posterior estimates. In their current class structure all the chains are stored separately, so the following function pools all sample for all chains together to make a single longer chain with all samplers. Pooling a mixture of converged and unconverged chains is likely to lead to poor quality results so should be done with care. e14Tagm_converged_pooled <- mcmc_pool_chains (e14Tagm_converged) e14Tagm_converged_pooled ## Object of class "MCMCParams" ## Method: TAGM.MCMC ## Number of chains: 1 e14Tagm_converged_pooled[[ 1 ]] ## Object of class "MCMCChain" ## Number of components: 10 ## Number of proteins: 1663 ## Number of iterations: 1500 To populate the summary slot of the converged and pooled chain, we can use the tagmMcmcProcess function. As we can see from the object below a summary is now available. The information now available in the summary slot was detailed in the previous section. We note that if there is more than 1 chain in the MCMCParams object then the chains are automatically pooled to compute the summaries. e14Tagm_converged_pooled <- tagmMcmcProcess (e14Tagm_converged_pooled) e14Tagm_converged_pooled ## Object of class "MCMCParams" ## Method: TAGM.MCMC ## Number of chains: 1 ## Summary available To create new feature columns in the MSnSet and append the summary information, we apply the tagmPredict function. The probJoint argument indicates whether or not to add probabilistic information for all organelles for all proteins, rather than just the information for the most probable organelle. The outlier probabilities are also returned by default, but users can change this using the probOutlier argument. E14TG2aR <- tagmPredict ( object = E14TG2aR, params = e14Tagm_converged_pooled, probJoint = TRUE ) head ( fData ( E14TG2aR)) ## Uniprot.ID UniprotName ## Q62261 Q62261 SPTB2_MOUSE ## Q9JHU4 Q9JHU4 DYHC1_MOUSE ## Q9QXS1 Q9QXS1 PLEC_MOUSE ## Protein.Description Peptides PSMs ## Q62261 Spectrin beta chain, brain 1 (multiple isoforms) 42 42 ## Q9JHU4 Cytoplasmic dynein 1 heavy chain 1 33 33 ## Q9QXS1 Isoform PLEC-1I of Plectin 33 33 ## GOannotation markers.orig markers tagm.map.allocation ## Q62261 PLM-SKE unknown unknown Endoplasmic reticulum ## Q9JHU4 SKE unknown unknown Nucleus - Chromatin ## Q9QXS1 unknown unknown unknown Plasma membrane ## tagm.map.probability tagm.map.outlier tagm.mcmc.allocation ## Q62261 8.165817e-09 0.9999999857 Endoplasmic reticulum ## Q9JHU4 9.996798e-01 0.0003202255 Nucleus - Chromatin ## Q9QXS1 1.250898e-06 0.9999987491 Proteasome ## tagm.mcmc.probability tagm.mcmc.probability.lowerquantile ## Q62261 0.5765793 0.0020296117 ## Q9JHU4 0.9738206 0.7594516090 ## Q9QXS1 0.4957129 0.0002886457 ## tagm.mcmc.probability.upperquantile tagm.mcmc.mean.shannon ## Q62261 0.9992504 0.201623229 ## Q9JHU4 0.9998822 0.081450206 ## Q9QXS1 0.9947100 0.447665536 ## tagm.mcmc.outlier tagm.mcmc.joint.40S Ribosome ## Q62261 2.547793e-01 4.401228e-10 ## Q9JHU4 3.335134e-05 1.936225e-18 ## Q9QXS1 6.423799e-01 2.213861e-07 ## tagm.mcmc.joint.60S Ribosome tagm.mcmc.joint.Cytosol ## Q62261 2.778620e-07 2.650861e-12 ## Q9JHU4 1.645727e-21 1.887645e-17 ## Q9QXS1 1.495170e-01 9.062280e-09 ## tagm.mcmc.joint.Endoplasmic reticulum tagm.mcmc.joint.Lysosome ## Q62261 5.765793e-01 1.108757e-11 ## Q9JHU4 1.548053e-17 5.577415e-24 ## Q9QXS1 1.768681e-04 1.150706e-04 ## tagm.mcmc.joint.Mitochondrion tagm.mcmc.joint.Nucleus - Chromatin ## Q62261 5.020528e-08 4.231731e-01 ## Q9JHU4 2.835919e-22 9.738206e-01 ## Q9QXS1 5.832273e-19 7.920397e-03 ## tagm.mcmc.joint.Nucleus - Nucleolus tagm.mcmc.joint.Plasma membrane ## Q62261 1.279255e-05 1.914808e-11 ## Q9JHU4 2.617943e-02 3.514851e-29 ## Q9QXS1 1.130580e-05 3.465462e-01 ## tagm.mcmc.joint.Proteasome ## Q62261 2.345204e-04 ## Q9JHU4 7.841425e-11 ## Q9QXS1 4.957129e-01 ## [ reached getOption("max.print") -- omitted 2 rows ] ## [ reached ’max’ / getOption("max.print") -- omitted 1 rows ] Aside: Priors Bayesian analysis requires users to specify prior information about the parameters. This may appear to be a challenging task; however, good default options are often possible. Should expert information be available for any of these priors then the users should provide this, otherwise we have found that the default choices work well in practice. The priors also provide regularisation and shrinkage to avoid overfitting. Given enough data the likelihood overwhelms the prior and the influence of the prior is weak. We place a normal inverse-Wishart prior on the parameters of the mutivariate normal mixture components. The normal inverse-Wishart prior has 4 hyperparameters that must be specified. These are: the prior mean mu0 expressing the prior location of each organelle; a prior shrinkage lambda0 , which is a scalar expressing uncertainty in the prior mean; the prior degrees of freedom nu0 ; and a scale prior S0 on the covariance. Together, nu0 and S0 specify the prior variability on organelle covariances. The same prior distribution is assumed for the parameters of all mutivariate normal mixture components. The default options for these are based on the choice recommended by 46 . The prior mean mu0 is set to be the mean of the data. lambda0 is set to be 0.01 meaning some uncertainty in the covariance is propagated to the mean, increasing lambda0 increases shrinkage towards the prior. nu0 is set to the number of feature variables plus 2, which is the smallest integer value that ensures a finite covariance matrix. The prior scale matrix S 0 is set to S 0 = diag ( 1 n ∑ ( X − X ¯ ) 2 ) K 1 / D , ( 1 ) and represents a diffuse prior on the covariance. Another good choice which is often used is a constant multiple of the identity matrix. The prior for the Dirichlet distribution concentration parameters beta0 is set to 1 for each organelle. Another reasonable choice would be the non-informative Jeffery’s prior for the Dirichlet hyperparameter, which sets beta0 to 0.5 for each organelle. The prior weight for the outlier detection class is a ℬ ( u , v ) distribution. The default for u = 2 and the default for v = 10. This represents the reasonable belief that u u + v = 1 6 proteins a priori might be an outlier and we believe is unlikely that more than 50% of proteins are outliers. Decreasing the value of v , represents more uncertainty about the number of protein that are outliers. Analysis, visualisation and interpretation of results Now that we have a single pooled chain of samples from a converged MCMC algorithm, we can begin to analyse the results. Preliminary analysis includes visualising the allocated organelle and localisation probability of each protein to its most probable organelle, as shown on Figure 8 . par ( mfrow = c ( 1 , 2 )) plot2D (E14TG2aR, fcol = "tagm.mcmc.allocation" , cex = fData (E14TG2aR) $ tagm.mcmc.probability, main = "TAGM MCMC allocations" ) addLegend (E14TG2aR, fcol = "markers" , where = "topleft" , ncol = 2 , cex = 0.6 ) plot2D (E14TG2aR, fcol = "tagm.mcmc.allocation" , cex = fData (E14TG2aR) $ tagm.mcmc.mean.shannon, main = "Visualising global uncertainty" ) addLegend (E14TG2aR, fcol = "markers" , where = "topleft" , ncol = 2 , cex = 0.6 ) Figure 8. TAGM MCMC allocations. On the left, point size have been scaled based on allocation probabilities. On the right, the point size have been scaled based on the global uncertainty using the mean Shannon entropy. We can visualise other summaries of the data including a Monte-Carlo averaged Shannon entropy, as shown in Figure 8 on the right. This is a measure of uncertainty and proteins with greater Shannon entropy have more uncertainty in their localisation. We observe global patterns of uncertainty, particularly in areas where organelle boundaries overlap. There are also regions of low uncertainty indicating little doubt about the localisation of these proteins. We are also interested in the relationship between localisation probability to the most probable class and the Shannon entropy ( Figure 9 ). Even though the two quantities are evidently correlated there is still considerable spread. Thus it is important to base inference not only on localisation probability but also a measure of uncertainty, for example the Shannon entropy. Proteins with low Shannon entropy have low uncertainty in their localisation, whilst those with higher Shannon entropy have uncertain localisation. Since multi-localised protein have uncertain localisation to a single subcellular niche, exploring the Shannon can aid in identifying multi-localised proteins. Figure 9. Shannon entropy and localisation probability. cls <- getStockcol ()[ as.factor ( fData (E14TG2aR) $ tagm.mcmc.allocation)] plot ( fData (E14TG2aR) $ tagm.mcmc.probability, fData ( E14TG2aR) $ tagm.mcmc.mean.shannon, col = cls, pch = 19 , xlab = "Localisation probability" , ylab = "Shannon entropy" ) addLegend (E14TG2aR, fcol = "markers" , where = "topright" , ncol = 2 , cex = 0.6 ) Aside from global visualisation of the data, we can also interrogate each individual protein. As illustrated on Figure 10 , we can obtain the full posterior distribution of localisation probabilities for each protein from the e14Tagm_converged_pooled object. We can use the plot generic on the MCMCParams object to obtain a violin plot of the localisation distribution. Simply providing the name of the protein in the second argument produces the plot for that protein. The solute carrier transporter protein E9QMX3, also referred to as Slc15a1, is most probably localised to plasma membrane in line with its role as a transmembrane transporter but also shows some uncertainty, potentially also localising to other comparments. The first violin plot visualises this uncertainty. The protein Q3V1Z5 is a supposed constitute of the 40S ribosome and has poor UniProt annotation with evidence only at the transcript level. From the plot below is is clear that Q3V1Z5 is a ribosomal associated protein, but it previous localisation has only been computational inferred and here we provide experimental evidence of a ribosomal annotation. Thus, quantifying uncertainty recovers important additional annotations. plot (e14Tagm_converged_pooled, "E9QMX3" ) plot (e14Tagm_converged_pooled, "Q3V1Z5" ) Figure 10. Full posterior distribution of localisation probabilities for individual proteins. Discussion The Bayesian analysis of biological data is of clear interest to many because of its ability to provide richer information about the experimental results. A fully Bayesian analysis differs from other machine learning approaches, since it can quantify the uncertainty in our inferences. Furthermore, we use a generative model to explicitly describe the data, which makes inferences more interpretable compared to the less interpretable outputs of black-box classifiers such as, for example, support vector machines (SVM). Bayesian analysis is often characterised by its provision of a (posterior) probability distribution over the biological parameters of interest, as opposed to single point estimate of these parameters. In the case that is presented in this workflow, a Bayesian analysis “computes” a posterior probability distribution over the protein localisation probabilities. These probability distributions can then be rigorously interrogated for greater biological insight; in addition, it may allow us to ask additional questions about the data, such as whether a protein might be multi-localised. Despite the wealth of information a Bayesian analysis can provide, the uptake amongst cell biologists is still low. This is because a Bayesian analysis presents a new set of challenges and little practical guidance exists regarding how to address these challenges. Bayesian analyses often rely on computatinally intensive approaches such as Markov-chain Monte-Carlo (MCMC) and a practical understanding of these algorithms and the interpretation of their output is a key barrier to their use. A Bayesian analysis usually consists of three broad steps: (1) Data pre-processing and algorithmic implementation, (2) assessing algorithmic convergence and (3) summarising and visualising the results. This workflow provides a set of tools to simplify these steps and provides step-by-step guidance in the context of the analysis of spatial proteomics data. We have provided a workflow for the Bayesian analysis of spatial proteomics using the pRoloc and MSnbase software. We have demonstrated, in a step-by-step fashion, the challenges and advantages associated with taking a Bayesian approach to data analysis. We hope this workflow will help spatial proteomics practitioners to apply our methods and will motivate others to create detailed documentation for the Bayesian analysis of biological data. Session information Below, we provide a summary of all packages and versions used to generate this document. sessionInfo () ## R version 3.5.2 Patched (2019-01-24 r76018) ## Platform: x86_64-pc-linux-gnu (64-bit) ## Running under: Manjaro Linux ## ## Matrix products: default ## BLAS: /usr/lib/libblas.so.3.8.0 ## LAPACK: /usr/lib/liblapack.so.3.8.0 ## ## locale: ## [1] LC_CTYPE=en_US.UTF-8 LC_NUMERIC=C ## [3] LC_TIME=en_US.UTF-8 LC_COLLATE=en_US.UTF-8 ## [5] LC_MONETARY=en_US.UTF-8 LC_MESSAGES=en_US.UTF-8 ## [7] LC_PAPER=en_US.UTF-8 LC_NAME=C ## [9] LC_ADDRESS=C LC_TELEPHONE=C ## [11] LC_MEASUREMENT=en_US.UTF-8 LC_IDENTIFICATION=C ## ## attached base packages: ## [1] stats4 parallel stats graphics grDevices utils datasets ## [8] methods base ## ## other attached packages: ## [1] patchwork_0.0.1 pRolocdata_1.21.1 pRoloc_1.23.2 ## [4] coda_0.19-2 mixtools_1.1.0 BiocParallel_1.16.6 ## [7] MLInterfaces_1.62.0 cluster_2.0.7-1 annotate_1.60.1 ## [10] XML_3.98-1.19 AnnotationDbi_1.44.0 IRanges_2.16.0 ## [13] MSnbase_2.9.3 ProtGenerics_1.14.0 S4Vectors_0.20.1 ## [16] mzR_2.17.2 Rcpp_1.0.1 Biobase_2.42.0 ## [19] BiocGenerics_0.28.0 ## ## loaded via a namespace (and not attached): ## [1] tidyselect_0.2.5 RSQLite_2.1.1 ## [3] htmlwidgets_1.3 grid_3.5.2 ## [5] trimcluster_0.1-2.1 lpSolve_5.6.13 ## [7] rda_1.0.2-2.1 devtools_2.0.1 ## [9] munsell_0.5.0 codetools_0.2-16 ## [11] preprocessCore_1.44.0 withr_2.1.2 ## [13] colorspace_1.4-1 knitr_1.22 ## [15] rstudioapi_0.10 robustbase_0.93-4 ## [17] mzID_1.20.1 labeling_0.3 ## [19] git2r_0.25.2 hwriter_1.3.2 ## [21] bit64_0.9-7 ggvis_0.4.4 ## [23] rprojroot_1.3-2 generics_0.0.2 ## [25] ipred_0.9-8 xfun_0.5 ## [27] randomForest_4.6-14 diptest_0.75-7 ## [29] R6_2.4.0 doParallel_1.0.14 ## [31] flexmix_2.3-15 bitops_1.0-6 ## [33] assertthat_0.2.0 promises_1.0.1 ## [35] scales_1.0.0 nnet_7.3-12 ## [37] gtable_0.2.0 affy_1.60.0 ## [39] processx_3.3.0 timeDate_3043.102 ## [41] rlang_0.3.1 genefilter_1.64.0 ## [43] splines_3.5.2 lazyeval_0.2.2 ## [45] ModelMetrics_1.2.2 impute_1.56.0 ## [47] hexbin_1.27.2 BiocManager_1.30.4 ## [49] yaml_2.2.0 reshape2_1.4.3 ## [51] threejs_0.3.1 crosstalk_1.0.0 ## [53] backports_1.1.3 httpuv_1.5.0 ## [55] caret_6.0-81 tools_3.5.2 ## [57] lava_1.6.5 usethis_1.4.0 ## [59] bookdown_0.9 ggplot2_3.1.0 ## [61] affyio_1.52.0 RColorBrewer_1.1-2 ## [63] proxy_0.4-23 sessioninfo_1.1.1 ## [65] plyr_1.8.4 base64enc_0.1-3 ## [67] progress_1.2.0 zlibbioc_1.28.0 ## [69] purrr_0.3.2 RCurl_1.95-4.12 ## [71] ps_1.3.0 prettyunits_1.0.2 ## [73] rpart_4.1-13 viridis_0.5.1 ## [75] sampling_2.8 sfsmisc_1.1-3 ## [77] LaplacesDemon_16.1.1 fs_1.2.7 ## [79] magrittr_1.5 data.table_1.12.0 ## [81] pcaMethods_1.74.0 mvtnorm_1.0-10 ## [83] whisker_0.3-2 pkgload_1.0.2 ## [85] hms_0.4.2 mime_0.6 ## [87] evaluate_0.13 xtable_1.8-3 ## [89] mclust_5.4.3 gridExtra_2.3 ## [91] testthat_2.0.1 compiler_3.5.2 ## [93] biomaRt_2.38.0 tibble_2.1.1 ## [95] ncdf4_1.16.1 crayon_1.3.4 ## [97] htmltools_0.3.6 segmented_0.5-3.0 ## [99] later_0.8.0 BiocWorkflowTools_1.8.0 ## [ reached getOption("max.print") -- omitted 49 entries ] The source of this document, including the code necessary to reproduce the analyses and figures is available in a public manuscript repository on GitHub 47 . Data availability The data used in this workflow was first published in Breckels et al . (2016) 36 and is available in the pRolocdata package. Software availability Computational workflow for this study available from: https://github.com/ococrook/TAGMworkflow 47 Archived source code at time of publication: https://doi.org/10.5281/zenodo.2593712 33 License: CC BY 4.0 Grant information PDWK was supported by the MRC (project reference MC_UU_00002/10). LMB was supported by a BBSRC Tools and Resources Development grant (Award BB/N023129/1) and a Wellcome Trust Technology Development Grant (Grant number 108441/Z/15/Z). OMC is a Wellcome Trust Mathematical Genomics and Medicine student supported financially by the School of Clinical Medicine, University of Cambridge. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript. Footnotes 1 https://drive.google.com/open?id=1zozntDhE6YZ-q8wjtQ-lxZ66EEszOGYi Faculty Opinions recommended References 1. Crook OM, Mulvey CM, Kirk PDW, et al. : A Bayesian mixture modelling approach for spatial proteomics. PLoS Comput Biol. 2018; 14 (11): e1006516. PubMed Abstract | Publisher Full Text | Free Full Text 2. Thul PJ, Åkesson L, Wiking M, et al. : A subcellular map of the human proteome. Science. 2017; 356 (6340): pii: eaal3321. PubMed Abstract | Publisher Full Text 3. Jeffery CJ: Moonlighting proteins--an update. Mol Biosyst. 2009; 5 (4): 345–350. PubMed Abstract | Publisher Full Text 4. Gibson TJ: Cell regulation: determined to signal discrete cooperation. Trends Biochem Sci. 2009; 34 (10): 471–482. PubMed Abstract | Publisher Full Text 5. Olkkonen VM, Ikonen E: When intracellular logistics fails--genetic defects in membrane trafficking. J Cell Sci. 2006; 119 (Pt 24): 5031–5045. PubMed Abstract | Publisher Full Text 6. Laurila K, Vihinen M: Prediction of disease-related mutations affecting protein localization. BMC genomics. 2009; 10 (1): 122. PubMed Abstract | Publisher Full Text | Free Full Text 7. Luheshi LM, Crowther DC, Dobson CM: Protein misfolding and disease: from the test tube to the organism. Curr Opin Chem Biol. 2008; 12 (1): 25–31. PubMed Abstract | Publisher Full Text 8. De Matteis MA, Luini A: Mendelian disorders of membrane trafficking. N Engl J Med. 2011; 365 (10): 927–938. PubMed Abstract | Publisher Full Text 9. Cody NA, Iampietro C, Lécuyer E: The many functions of mRNA localization during normal development and disease: from pillar to post. Wiley Interdiscip Rev Dev Biol. 2013; 2 (6): 781–796. PubMed Abstract | Publisher Full Text 10. Kau TR, Way JC, Silver PA: Nuclear transport and cancer: from mechanism to intervention. Nat Rev Cancer. 2004; 4 (2): 106–117. PubMed Abstract | Publisher Full Text 11. Rodriguez JA, Au WW, Henderson BR: Cytoplasmic mislocalization of BRCA1 caused by cancer-associated mutations in the BRCT domain. Exp Cell Res. 2004; 293 (1): 14–21. PubMed Abstract | Publisher Full Text 12. Latorre IJ, Roh MH, Frese KK, et al. : Viral oncoprotein-induced mislocalization of select PDZ proteins disrupts tight junctions and causes polarity defects in epithelial cells. J Cell Sci. 2005; 118 (Pt 18): 4283–4293. PubMed Abstract | Publisher Full Text | Free Full Text 13. Shin SJ, Smith JA, Rezniczek GA, et al. : Unexpected gain of function for the scaffolding protein plectin due to mislocalization in pancreatic cancer. Proc Natl Acad Sci U S A. 2013; 110 (48): 19414–19419. PubMed Abstract | Publisher Full Text | Free Full Text 14. Siljee JE, Wang Y, Bernard AA, et al. : Subcellular localization of MC4R with ADCY3 at neuronal primary cilia underlies a common pathway for genetic predisposition to obesity. Nat Genet. 2018; 50 (2): 180–185. PubMed Abstract | Publisher Full Text | Free Full Text 15. Dunkley TP, Hester S, Shadforth IP, et al. : Mapping the Arabidopsis organelle proteome. Proc Natl Acad Sci U S A. 2006; 103 (17): 6518–6523. PubMed Abstract | Publisher Full Text | Free Full Text 16. Foster LJ, de Hoog CL, Zhang Y, et al. : A mammalian organelle map by protein correlation profiling. Cell. 2006; 125 (1): 187–199. PubMed Abstract | Publisher Full Text 17. Christoforou A, Mulvey CM, Breckels LM, et al. : A draft map of the mouse pluripotent stem cell spatial proteome. Nat Commun. 2016; 7 : 8992. PubMed Abstract | Publisher Full Text | Free Full Text 18. Geladaki A, Kočevar Britovšek N, Breckels LM, et al. : Combining LOPIT with differential ultracentrifugation for high-resolution spatial proteomics. Nat Commun. 2019; 10 (1): 331. PubMed Abstract | Publisher Full Text | Free Full Text 19. Tan DJ, Dvinge H, Christoforou A, et al. : Mapping organelle proteins and protein complexes in Drosophila melanogaster . J Proteome Res. 2009; 8 (6): 2667–2678. PubMed Abstract | Publisher Full Text 20. Hall SL, Hester S, Griffin JL, et al. : The organelle proteome of the DT40 lymphocyte cell line. Mol Cell Proteomics. 2009; 8 (6): 1295–1305. PubMed Abstract | Publisher Full Text | Free Full Text 21. Breckels LM, Gatto L, Christoforou A, et al. : The effect of organelle discovery upon sub-cellular protein localisation. J Proteomics. 2013; 88 : 129–140. PubMed Abstract | Publisher Full Text 22. Beltran PM, Mathias RA, Cristea IM: A Portrait of the Human Organelle Proteome In Space and Time during Cytomegalovirus Infection. Cell Syst. 2016; 3 (4): 361–373.e6. PubMed Abstract | Publisher Full Text | Free Full Text 23. Jadot M, Boonen M, Thirion J, et al. : Accounting for Protein Subcellular Localization: A Compartmental Map of the Rat Liver Proteome. Mol Cell Proteomics. 2017; 16 (2): 194–212. PubMed Abstract | Publisher Full Text | Free Full Text 24. Itzhak DN, Davies C, Tyanova S, et al. : A Mass Spectrometry-Based Approach for Mapping Protein Subcellular Localization Reveals the Spatial Proteome of Mouse Primary Neurons. Cell Rep. 2017; 20 (11): 2706–2718. PubMed Abstract | Publisher Full Text | Free Full Text 25. Mendes M, Peláez-García A, López-Lucendo M, et al. : Mapping the Spatial Proteome of Metastatic Cells in Colorectal Cancer. Proteomics. 2017; 17 (19): 1700094. PubMed Abstract | Publisher Full Text 26. Hirst J, Itzhak DN, Antrobus R, et al. : Role of the AP-5 adaptor protein complex in late endosome-to-Golgi retrieval. PLoS Biol. 2018; 16 (1): e2004411. PubMed Abstract | Publisher Full Text | Free Full Text 27. Davies AK, Itzhak DN, Edgar JR, et al. : AP-4 vesicles contribute to spatial control of autophagy via RUSC-dependent peripheral delivery of ATG9A. Nat Commun. 2018; 9 (1): 3958. PubMed Abstract | Publisher Full Text | Free Full Text 28. Orre LM, Vesterlund M, Pan Y, et al. : SubCellBarCode: Proteome-wide Mapping of Protein Localization and Relocalization. Mol Cell. 2019; 73 (1): 166–182.e7. PubMed Abstract | Publisher Full Text 29. Nightingale DJ, Geladaki A, Breckels LM, et al. : The subcellular organisation of Saccharomyces cerevisiae . Curr Opin Chem Biol. 2019; 48 : 86–95. PubMed Abstract | Publisher Full Text | Free Full Text 30. Gelman A, Carlin JB, Stern HS, et al. : Bayesian Data Analysis. Chapman & Hall, London, 1995. Reference Source 31. Mulvey CM, Breckels LM, Geladaki A, et al. : Using hyperLOPIT to perform high-resolution mapping of the spatial proteome. Nat Protoc. 2017; 12 (6): 1110–1135. PubMed Abstract | Publisher Full Text 32. Dempster AP, Laird NM, Rubin DB: Maximum likelihood from incomplete data via the em algorithm. J Roy Stat Soc B Met. 1977; 39 (1): 1–38. Reference Source 33. Crook OM, Gatto L, Kirk PDW, et al. : ococrook/tagmworkflow: F1000 submission. 2019. http://www.doi.org/10.5281/zenodo.2593712 34. Gatto L, Breckels LM, Wieczorek S, et al. : Mass-spectrometry-based spatial proteomics data analysis using pRoloc and pRolocdata. Bioinformatics. 2014; 30 (9): 1322–4. PubMed Abstract | Publisher Full Text | Free Full Text 35. Breckels LM, Mulvey CM, Lilley KS, et al. : A Bioconductor workflow for processing and analysing spatial proteomics data [version 2; peer review: 2 approved]. F1000Res. 2016; 5 : 2926. PubMed Abstract | Publisher Full Text | Free Full Text 36. Breckels LM, Holden SB, Wojnar D, et al. : Learning from Heterogeneous Data Sources: An Application in Spatial Proteomics. PLoS Comput Biol. 2016; 12 (5): e1004920. PubMed Abstract | Publisher Full Text | Free Full Text 37. Dunkley TP, Watson R, Griffin JL, et al. : Localization of organelle proteins by isotope tagging (LOPIT). Mol Cell Proteomics. 2004; 3 (11): 1128–1134. PubMed Abstract | Publisher Full Text 38. Itzhak DN, Tyanova S, Cox J, et al. : Global, quantitative and dynamic mapping of protein subcellular localization. eLife. 2016; 5 : pii: e16950. PubMed Abstract | Publisher Full Text | Free Full Text 39. Smith AFM, Roberts GO: Bayesian computation via the gibbs sampler and related markov chain monte carlo methods. J Roy Stat Soc B Met. 1993; 55 (1): 3–23. Publisher Full Text 40. Gilks WR, Richardson S, Spiegelhalter D: Markov chain Monte Carlo in practice. Chapman and Hall/CRC, 1995. Reference Source 41. Geweke J: Evaluating the accuracy of sampling-based approaches to the calculation of posterior moments. BAYESIAN STATISTICS. 1992. Reference Source 42. Gelman A, Rubin DB: Inference from iterative simulation using multiple sequences. Stat Sci. 1992; 7 (4): 457–472. Publisher Full Text 43. Roberts GO, Smith AFM: Simple conditions for the convergence of the gibbs sampler and metropolis-hastings algorithms. Stoch Process Their Appl. 1994; 49 (2): 207–216. Publisher Full Text 44. Brooks SP, Gelman A: General methods for monitoring convergence of iterative simulations. J Comput Graph Stat. 1998; 7 (4): 434–455. Publisher Full Text 45. Plummer M, Best N, Cowles K, et al. : Coda: Convergence diagnosis and output analysis for mcmc. R News. 2006; 6 (1): 7–11. Reference Source 46. Fraley C, Raftery AE: Bayesian regularization for normal mixture estimation and model-based clustering. Technical report, Washington Univ Seattle Dept of Statistics, 2005. Reference Source 47. Crook OM, Gatto L: A bioconductor workflow for the bayesian analysis of spatial proteomics. 2019. Reference Source Comments on this article Comments (0) Version 1 VERSION 1 PUBLISHED 11 Apr 2019 ADD YOUR COMMENT Comment Author details Author details 1 Cambridge Centre for Proteomics, Department of Biochemistry, University of Cambridge, Cambridge, CB2 1QR, UK 2 MRC Biostatistics Unit, Cambridge Institute for Public Health, Cambridge Institute for Public Health, Cambridge, CB2 0SR, UK 3 Catholic University of Louvain, Brussels, 1200, Belgium Oliver M. Crook Roles: Conceptualization, Data Curation, Formal Analysis, Funding Acquisition, Investigation, Methodology, Software, Validation, Visualization, Writing – Original Draft Preparation, Writing – Review & Editing Lisa M. Breckels Roles: Data Curation, Software, Visualization, Writing – Review & Editing Kathryn S. Lilley Roles: Funding Acquisition, Resources, Supervision, Writing – Review & Editing Paul D.W. Kirk Roles: Conceptualization, Investigation, Methodology, Supervision, Writing – Review & Editing Laurent Gatto Roles: Conceptualization, Data Curation, Investigation, Methodology, Project Administration, Resources, Software, Supervision, Validation, Writing – Review & Editing Competing interests No competing interests were disclosed. Grant information PDWK was supported by the MRC (project reference MC_UU_00002/10). LMB was supported by a BBSRC Tools and Resources Development grant (Award BB/N023129/1) and a Wellcome Trust Technology Development Grant (Grant number 108441/Z/15/Z). OMC is a Wellcome Trust Mathematical Genomics and Medicine student supported financially by the School of Clinical Medicine, University of Cambridge. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript. Article Versions (1) version 1 Published: 11 Apr 2019, 8:446 https://doi.org/10.12688/f1000research.18636.1 Copyright © 2019 Crook OM et al . This is an open access article distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. Download Export To Sciwheel Bibtex EndNote ProCite Ref. Manager (RIS) Sente metrics Views Downloads F1000Research - - PubMed Central info_outline Data from PMC are received and updated monthly. - - Citations open_in_new 0 open_in_new 0 open_in_new SEE MORE DETAILS CITE how to cite this article Crook OM, Breckels LM, Lilley KS et al. A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.12688/f1000research.18636.1 ) NOTE: If applicable, it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS track receive updates on this article Track an article to receive email alerts on any updates to this article. TRACK THIS ARTICLE Share Open Peer Review Current Reviewer Status: ? Key to Reviewer Statuses VIEW HIDE Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Version 1 VERSION 1 PUBLISHED 11 Apr 2019 Views 0 Cite How to cite this report: Lamond A and Brenes Murillo A. Reviewer Report For: A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.5256/f1000research.20403.r47080 ) The direct URL for this report is: https://f1000research.com/articles/8-446/v1#referee-response-47080 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 09 May 2019 Angus Lamond , Centre for Gene Regulation and Expression, School of Life Sciences, University of Dundee, Dundee, UK Alejandro Brenes Murillo , University of Dundee, Dundee, UK Approved VIEWS 0 https://doi.org/10.5256/f1000research.20403.r47080 The subcellular distribution of the proteome is a very important determinant of cell function and therefore the ability to analyse subcellular protein localisation is vital for studying cell biology and regulation. Consequently, technical developments that facilitate the analysis of how ... Continue reading READ ALL The subcellular distribution of the proteome is a very important determinant of cell function and therefore the ability to analyse subcellular protein localisation is vital for studying cell biology and regulation. Consequently, technical developments that facilitate the analysis of how the proteome is distributed between cell compartments are of great value to both the proteomics and molecular cell biology research communities. Bioconductor is also a fantastic source of tools that can be used to analyse ‘omics’ datasets, hence further additions to this toolbox are always welcome and of value. Here the authors provide a detailed workflow in Bioconductor that enables the convenient Bayesian analysis of spatial proteomics data. The explanations of the models provided by the authors are user friendly and allow also non-bioinformaticians to understand clearly the mechanics of the different methods. The suggestions provided relating to default parameters for the models are also very welcome. Overall, I view this as a timely and very useful contribution for the community that will be well received and of genuine value. I note below a few specific points that the authors should address. Specific Points: I was initially unable to load either pRoloc or pRolocdata on R and therefore could not follow any of the examples provided. I had to upgrade to a new version of R to resolve this issue. The authors should specify the R version compatibility required. Line 1: Remove one ‘the’ 3rd page 5 paragraph, second sentence is disconnected. Example dataset uses iTRAQ, which means the data will have false positives all across The authors should explain/justify their assertion that their Method is independent of the isotope labelling method because TMT and iTRAQ are expected to provide different challenges? It would be helpful for the Gelman diagnostic to suggest minimum a number of chains for MCMC - Suggestion for an ideal number of chains for a medium range desktop computer would be useful too “Trace for Chain 2 & 4 have clear jumps” This is for figure 7 - This sounds imprecise; hard to spot and replicate. Also, I do not see the “clear jumps” the authors refer to – can this be clarified? Suggesting default priors for the Bayesian analysis is very useful. However, it would be good to show the visual consequences of changing S0 from beta0 of 1 to beta0 of 0.5 The suggestion by the authors that it is likely that 16.7% of proteins are outliers needs to be justified. The use of Shannon entropy to focus on proteins with multiple localisations is an interesting approach that could be discussed further. Is the rationale for developing the new method (or application) clearly explained? Yes Is the description of the method technically sound? Yes Are sufficient details provided to allow replication of the method development and its use by others? Yes If any results are presented, are all the source data underlying the results available to ensure full reproducibility? Yes Are the conclusions about the method and its performance adequately supported by the findings presented in the article? Yes Competing Interests: No competing interests were disclosed. We confirm that we have read this submission and believe that we have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Lamond A and Brenes Murillo A. Reviewer Report For: A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.5256/f1000research.20403.r47080 ) The direct URL for this report is: https://f1000research.com/articles/8-446/v1#referee-response-47080 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Views 0 Cite How to cite this report: Holmes S. Reviewer Report For: A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.5256/f1000research.20403.r47076 ) The direct URL for this report is: https://f1000research.com/articles/8-446/v1#referee-response-47076 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 01 May 2019 Susan Holmes , Department of Statistics, Stanford University, Stanford, CA, USA Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.20403.r47076 This paper provides an important advance in the study of spatial proteomics. Through the use of Monte Carlo sampling the authors are able to provide evaluations of uncertainty by using the T-augmented Gaussian mixture model proposed by Crook et al. ... Continue reading READ ALL This paper provides an important advance in the study of spatial proteomics. Through the use of Monte Carlo sampling the authors are able to provide evaluations of uncertainty by using the T-augmented Gaussian mixture model proposed by Crook et al. The authors have been careful to provide all the code and the relevant data are included in the package 'pRolocdata'. Although the focus is on the spatial localisation of proteins, the authors only study the probabilities of correct localizations and do not show a spatial plot of the probabilities so one does not see clearly how the probabilities vary spatially. First Issue: PCA plots a) Ratios: There are a few corrections needed to the figures that show PCA plots. It is probably not worth showing the percentages of variance up to two decimal digits, however it is important to respect the axes relative scale and units. For instance, since the ratio of variances in Figure 1 is 40:25 the plot should be rectangular, this is more egregious in Figures 2, especially the right side panel. I suggest making the figures on top of each other so they are not narrow (For a reference see Chapter 7 of the book 1 that explains why usually PCA plots should be rectangular and not square). b) Construction of ellipses on the PCA plots in Figure 2 is very confusing. These cannot be the posterior distributions and their construction presupposes there is a relevant multivariate normal distribution which I do not think is justified here. Second Issue: Bayesian computations a) Priors In the "Aside" subsection on priors, I suggest exploring the possibility of using prior data to construct a prior as in practical situations this is often how pragmatic Bayesians think about building priors. I think it is worth supporting the statement that the posterior is overwhelmed by the data with a small example. This could be done following the type of process recommended by Betancourt who gives very good examples of the effect of priors through the use of posterior predictive distributions (see here https://betanalpha.github.io/assets/case_studies/principled_bayesian_workflow.html#24_model_adequacy ). b) Convergence diagnostics The use of trace plots and Gelman and Rubin's Rhat can be quite confusing and can only show non-convergence, the current version puts too much emphasis on a few short runs. The authors have shown some chains that have not converged but do not mention that the contemporary literature is much more careful of how small the Rhat should be, see the preprint by Vats,D and Knudson, C 2018 2 and the discussions by Dan Simpson and Andrew Gelman for instance (https://statmodeling.stat.columbia.edu/2019/03/19/maybe-its-time-to-let-the-old-ways-die-or-we-broke-r-hat-so-now-we-have-to-fix-it/), the preprint is on the arXiv 3 . Third Issue: Visualization of results: Figure 8 does not show the uncertainty in an intuitive way. It is very hard to differentiate the size of the dots on the left hand plot and the dots give the impression of a circular posterior distribution which is incorrect. The reference Ren et al, 2017 4 contains several examples of posterior confidence contours that communicate uncertainty in a more precise way which enables the comparisons of uncertainty across different locations in the PCA. It would also be illuminating if the authors could make a plot with uncertainty contours as a function of the underlying spatial coordinates. Is the rationale for developing the new method (or application) clearly explained? Yes Is the description of the method technically sound? Partly Are sufficient details provided to allow replication of the method development and its use by others? Yes If any results are presented, are all the source data underlying the results available to ensure full reproducibility? Yes Are the conclusions about the method and its performance adequately supported by the findings presented in the article? Partly References 1. Holmes S, Wolfgang H: Modern Statistics for Modern Biology. Cambridge University Press . 2018. 2. Vats D, Knudson C: Revisiting the Gelman-Rubin Diagnostic. arXiv . 2018. 3. Vehtari A, Gelman A, Simpson D, Carpenter B, et al.: Rank-normalization, folding, and localization: An improved Rˆ for assessing convergence of MCMC. arXiv . 2019. 4. Ren B, Bacallado S, Favaro S, Holmes S, et al.: Bayesian Nonparametric Ordination for the Analysis of Microbial Communities. J Am Stat Assoc . 2017; 112 (520): 1430-1442 PubMed Abstract | Publisher Full Text Competing Interests: No competing interests were disclosed. Reviewer Expertise: Multivariate statistics, Bayesian methods, MCMC generation of posterior samples, data visualization, microbiome studies involving NGS and metabolomics. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Holmes S. Reviewer Report For: A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.5256/f1000research.20403.r47076 ) The direct URL for this report is: https://f1000research.com/articles/8-446/v1#referee-response-47076 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Views 0 Cite How to cite this report: Jean Beltran P. Reviewer Report For: A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.5256/f1000research.20403.r47079 ) The direct URL for this report is: https://f1000research.com/articles/8-446/v1#referee-response-47079 NOTE: it is important to ensure the information in square brackets after the title is included in this citation. Close Copy Citation Details Reviewer Report 30 Apr 2019 Pierre M. Jean Beltran , Proteomics Platform, Broad Institute, Cambridge, USA Approved with Reservations VIEWS 0 https://doi.org/10.5256/f1000research.20403.r47079 The spatial proteomics field has seen increased popularity over the past few years through development of experimental, statistical, and computational methodologies. In this Method Article, Crook OM and colleagues present a bioinformatics workflow for the analysis of spatial proteomics data ... Continue reading READ ALL The spatial proteomics field has seen increased popularity over the past few years through development of experimental, statistical, and computational methodologies. In this Method Article, Crook OM and colleagues present a bioinformatics workflow for the analysis of spatial proteomics data using a set of Bayesian analysis tools. This work is a useful guide for biologists that wish to properly apply and diagnose a Markov-chain Monte-Carlo (MCMC) inference procedure using the pRoloc package. The article is timely and relevant given the increasing number of biologists acquiring programming skills yet lacking extensive expertise in this area of statistics and modeling. In particular, the authors did a good job explaining basic terminology and rationale for the diagnostics necessary during MCMC inference. The methodology is clearly explained, all the code is provided, and the packages and datasets are available through Bioconductor, making them easily accessible and reproducible. However, I believe this article lacks in background information and details for those readers that are not experts in the spatial proteomics field. In addition, the value of applying the more resource-intensive MCMC method compared to the expectation-maximisation (EM) algorithm is not clear from the data presented. I recommend this article for indexing given that the following issues are resolved. Major comments Currently, there is no direct comparison of the results provided by the EM and MCMC algorithms. Given that MCMC is computationally intensive and requires additional diagnostics, the reader needs to be convinced that there is a tangible benefit from using the MCMC approach. A discussion of scenarios in which the EM algorithm would be preferable over MCMC would be useful for the reader. For example, can MCMC fail to converge? If so, what kind of interpretation can be obtained by looking at the point-estimates from the EM algorithm instead? Additional background information and references would be useful, as this article seems to assume that the reader is familiar with spatial proteomics data sets. For example, the author can explain the organelle MS-proteomics data sets, the use of organelle markers to train the model, and the prediction of unannotated proteins from the model fitted on the organelle markers. In addition, I would suggest pointing the reader to the review by Gatto L et al. 2014 1 in MCP, or another similar review as a primer. Page 20 – The authors claim that lower Shannon entropy is an indication of low uncertainty in localization and therefore can aid in identifying multi-localized proteins. This claim should be backed up by some examples or evidence from the literature that agree with the data presented. Minor comments Figure 1 – Additional details should be provided in the figure legend. Is this plot only showing organelle markers? A short description of the outlier component and how the analyst can interpret this component would be particularly helpful. Page 15 – It is not clear why the Gelman diagnostic performed on mean allocation of all chains helps discriminate chains 1, 2, and 4. The upper C.I. is already near to 1 (1.01) for this example. I observed several small writing mistakes; the article should be proof-read before publication. Some examples are indicated below. Page 15 – sentence ending with “… Gelman diagnostics between chains 4, 5, 6”. However, the code shows that the diagnostic was performed on chains 3, 5, 6. Page 21 – “From the plot below is is…” should read “it is”. Is the rationale for developing the new method (or application) clearly explained? Yes Is the description of the method technically sound? Yes Are sufficient details provided to allow replication of the method development and its use by others? Yes If any results are presented, are all the source data underlying the results available to ensure full reproducibility? Yes Are the conclusions about the method and its performance adequately supported by the findings presented in the article? Partly References 1. Gatto L, Breckels LM, Burger T, Nightingale DJ, et al.: A foundation for reliable spatial proteomics data analysis. Mol Cell Proteomics . 2014; 13 (8): 1937-52 PubMed Abstract | Publisher Full Text Competing Interests: No competing interests were disclosed. Reviewer Expertise: My area of expertise is in interaction, global, and organelle proteomics analysis. My work includes application of experimental, statistical, and computational methods for the study of disease using a proteomics approach. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. Close READ LESS CITE CITE HOW TO CITE THIS REPORT Jean Beltran P. Reviewer Report For: A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.5256/f1000research.20403.r47079 ) The direct URL for this report is: https://f1000research.com/articles/8-446/v1#referee-response-47079 NOTE: it is important to ensure the information in square brackets after the title is included in all citations of this article. COPY CITATION DETAILS Report a concern Respond or Comment COMMENT ON THIS REPORT Comments on this article Comments (0) Version 1 VERSION 1 PUBLISHED 11 Apr 2019 ADD YOUR COMMENT Comment keyboard_arrow_left keyboard_arrow_right Open Peer Review Reviewer Status info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions Reviewer Reports Invited Reviewers 1 2 3 Version 1 11 Apr 19 read read read Pierre M. Jean Beltran , Broad Institute, Cambridge, USA Susan Holmes , Stanford University, Stanford, USA Angus Lamond , University of Dundee, Dundee, UK Alejandro Brenes Murillo , University of Dundee, Dundee, UK Comments on this article All Comments (0) Add a comment Sign up for content alerts Sign Up You are now signed up to receive this alert Browse by related subjects keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2019 Lamond A et al. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 09 May 2019 | for Version 1 Angus Lamond , Centre for Gene Regulation and Expression, School of Life Sciences, University of Dundee, Dundee, UK Alejandro Brenes Murillo , University of Dundee, Dundee, UK 0 Views copyright © 2019 Lamond A et al. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions The subcellular distribution of the proteome is a very important determinant of cell function and therefore the ability to analyse subcellular protein localisation is vital for studying cell biology and regulation. Consequently, technical developments that facilitate the analysis of how the proteome is distributed between cell compartments are of great value to both the proteomics and molecular cell biology research communities. Bioconductor is also a fantastic source of tools that can be used to analyse ‘omics’ datasets, hence further additions to this toolbox are always welcome and of value. Here the authors provide a detailed workflow in Bioconductor that enables the convenient Bayesian analysis of spatial proteomics data. The explanations of the models provided by the authors are user friendly and allow also non-bioinformaticians to understand clearly the mechanics of the different methods. The suggestions provided relating to default parameters for the models are also very welcome. Overall, I view this as a timely and very useful contribution for the community that will be well received and of genuine value. I note below a few specific points that the authors should address. Specific Points: I was initially unable to load either pRoloc or pRolocdata on R and therefore could not follow any of the examples provided. I had to upgrade to a new version of R to resolve this issue. The authors should specify the R version compatibility required. Line 1: Remove one ‘the’ 3rd page 5 paragraph, second sentence is disconnected. Example dataset uses iTRAQ, which means the data will have false positives all across The authors should explain/justify their assertion that their Method is independent of the isotope labelling method because TMT and iTRAQ are expected to provide different challenges? It would be helpful for the Gelman diagnostic to suggest minimum a number of chains for MCMC - Suggestion for an ideal number of chains for a medium range desktop computer would be useful too “Trace for Chain 2 & 4 have clear jumps” This is for figure 7 - This sounds imprecise; hard to spot and replicate. Also, I do not see the “clear jumps” the authors refer to – can this be clarified? Suggesting default priors for the Bayesian analysis is very useful. However, it would be good to show the visual consequences of changing S0 from beta0 of 1 to beta0 of 0.5 The suggestion by the authors that it is likely that 16.7% of proteins are outliers needs to be justified. The use of Shannon entropy to focus on proteins with multiple localisations is an interesting approach that could be discussed further. Is the rationale for developing the new method (or application) clearly explained? Yes Is the description of the method technically sound? Yes Are sufficient details provided to allow replication of the method development and its use by others? Yes If any results are presented, are all the source data underlying the results available to ensure full reproducibility? Yes Are the conclusions about the method and its performance adequately supported by the findings presented in the article? Yes Competing Interests No competing interests were disclosed. We confirm that we have read this submission and believe that we have an appropriate level of expertise to confirm that it is of an acceptable scientific standard. reply Respond to this report Responses (0) Lamond A and Brenes Murillo A. Peer Review Report For: A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.5256/f1000research.20403.r47080) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/8-446/v1#referee-response-47080 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2019 Holmes S. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 01 May 2019 | for Version 1 Susan Holmes , Department of Statistics, Stanford University, Stanford, CA, USA 0 Views copyright © 2019 Holmes S. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions This paper provides an important advance in the study of spatial proteomics. Through the use of Monte Carlo sampling the authors are able to provide evaluations of uncertainty by using the T-augmented Gaussian mixture model proposed by Crook et al. The authors have been careful to provide all the code and the relevant data are included in the package 'pRolocdata'. Although the focus is on the spatial localisation of proteins, the authors only study the probabilities of correct localizations and do not show a spatial plot of the probabilities so one does not see clearly how the probabilities vary spatially. First Issue: PCA plots a) Ratios: There are a few corrections needed to the figures that show PCA plots. It is probably not worth showing the percentages of variance up to two decimal digits, however it is important to respect the axes relative scale and units. For instance, since the ratio of variances in Figure 1 is 40:25 the plot should be rectangular, this is more egregious in Figures 2, especially the right side panel. I suggest making the figures on top of each other so they are not narrow (For a reference see Chapter 7 of the book 1 that explains why usually PCA plots should be rectangular and not square). b) Construction of ellipses on the PCA plots in Figure 2 is very confusing. These cannot be the posterior distributions and their construction presupposes there is a relevant multivariate normal distribution which I do not think is justified here. Second Issue: Bayesian computations a) Priors In the "Aside" subsection on priors, I suggest exploring the possibility of using prior data to construct a prior as in practical situations this is often how pragmatic Bayesians think about building priors. I think it is worth supporting the statement that the posterior is overwhelmed by the data with a small example. This could be done following the type of process recommended by Betancourt who gives very good examples of the effect of priors through the use of posterior predictive distributions (see here https://betanalpha.github.io/assets/case_studies/principled_bayesian_workflow.html#24_model_adequacy ). b) Convergence diagnostics The use of trace plots and Gelman and Rubin's Rhat can be quite confusing and can only show non-convergence, the current version puts too much emphasis on a few short runs. The authors have shown some chains that have not converged but do not mention that the contemporary literature is much more careful of how small the Rhat should be, see the preprint by Vats,D and Knudson, C 2018 2 and the discussions by Dan Simpson and Andrew Gelman for instance (https://statmodeling.stat.columbia.edu/2019/03/19/maybe-its-time-to-let-the-old-ways-die-or-we-broke-r-hat-so-now-we-have-to-fix-it/), the preprint is on the arXiv 3 . Third Issue: Visualization of results: Figure 8 does not show the uncertainty in an intuitive way. It is very hard to differentiate the size of the dots on the left hand plot and the dots give the impression of a circular posterior distribution which is incorrect. The reference Ren et al, 2017 4 contains several examples of posterior confidence contours that communicate uncertainty in a more precise way which enables the comparisons of uncertainty across different locations in the PCA. It would also be illuminating if the authors could make a plot with uncertainty contours as a function of the underlying spatial coordinates. Is the rationale for developing the new method (or application) clearly explained? Yes Is the description of the method technically sound? Partly Are sufficient details provided to allow replication of the method development and its use by others? Yes If any results are presented, are all the source data underlying the results available to ensure full reproducibility? Yes Are the conclusions about the method and its performance adequately supported by the findings presented in the article? Partly References 1. Holmes S, Wolfgang H: Modern Statistics for Modern Biology. Cambridge University Press . 2018. 2. Vats D, Knudson C: Revisiting the Gelman-Rubin Diagnostic. arXiv . 2018. 3. Vehtari A, Gelman A, Simpson D, Carpenter B, et al.: Rank-normalization, folding, and localization: An improved Rˆ for assessing convergence of MCMC. arXiv . 2019. 4. Ren B, Bacallado S, Favaro S, Holmes S, et al.: Bayesian Nonparametric Ordination for the Analysis of Microbial Communities. J Am Stat Assoc . 2017; 112 (520): 1430-1442 PubMed Abstract | Publisher Full Text Competing Interests No competing interests were disclosed. Reviewer Expertise Multivariate statistics, Bayesian methods, MCMC generation of posterior samples, data visualization, microbiome studies involving NGS and metabolomics. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (0) Holmes S. Peer Review Report For: A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.5256/f1000research.20403.r47076) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/8-446/v1#referee-response-47076 keyboard_arrow_left Back to all reports Reviewer Report 0 Views copyright © 2019 Jean Beltran P. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. 30 Apr 2019 | for Version 1 Pierre M. Jean Beltran , Proteomics Platform, Broad Institute, Cambridge, USA 0 Views copyright © 2019 Jean Beltran P. This is an open access peer review report distributed under the terms of the Creative Commons Attribution License , which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited. format_quote Cite this report speaker_notes Responses (0) Approved With Reservations info_outline Alongside their report, reviewers assign a status to the article: Approved The paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved Fundamental flaws in the paper seriously undermine the findings and conclusions The spatial proteomics field has seen increased popularity over the past few years through development of experimental, statistical, and computational methodologies. In this Method Article, Crook OM and colleagues present a bioinformatics workflow for the analysis of spatial proteomics data using a set of Bayesian analysis tools. This work is a useful guide for biologists that wish to properly apply and diagnose a Markov-chain Monte-Carlo (MCMC) inference procedure using the pRoloc package. The article is timely and relevant given the increasing number of biologists acquiring programming skills yet lacking extensive expertise in this area of statistics and modeling. In particular, the authors did a good job explaining basic terminology and rationale for the diagnostics necessary during MCMC inference. The methodology is clearly explained, all the code is provided, and the packages and datasets are available through Bioconductor, making them easily accessible and reproducible. However, I believe this article lacks in background information and details for those readers that are not experts in the spatial proteomics field. In addition, the value of applying the more resource-intensive MCMC method compared to the expectation-maximisation (EM) algorithm is not clear from the data presented. I recommend this article for indexing given that the following issues are resolved. Major comments Currently, there is no direct comparison of the results provided by the EM and MCMC algorithms. Given that MCMC is computationally intensive and requires additional diagnostics, the reader needs to be convinced that there is a tangible benefit from using the MCMC approach. A discussion of scenarios in which the EM algorithm would be preferable over MCMC would be useful for the reader. For example, can MCMC fail to converge? If so, what kind of interpretation can be obtained by looking at the point-estimates from the EM algorithm instead? Additional background information and references would be useful, as this article seems to assume that the reader is familiar with spatial proteomics data sets. For example, the author can explain the organelle MS-proteomics data sets, the use of organelle markers to train the model, and the prediction of unannotated proteins from the model fitted on the organelle markers. In addition, I would suggest pointing the reader to the review by Gatto L et al. 2014 1 in MCP, or another similar review as a primer. Page 20 – The authors claim that lower Shannon entropy is an indication of low uncertainty in localization and therefore can aid in identifying multi-localized proteins. This claim should be backed up by some examples or evidence from the literature that agree with the data presented. Minor comments Figure 1 – Additional details should be provided in the figure legend. Is this plot only showing organelle markers? A short description of the outlier component and how the analyst can interpret this component would be particularly helpful. Page 15 – It is not clear why the Gelman diagnostic performed on mean allocation of all chains helps discriminate chains 1, 2, and 4. The upper C.I. is already near to 1 (1.01) for this example. I observed several small writing mistakes; the article should be proof-read before publication. Some examples are indicated below. Page 15 – sentence ending with “… Gelman diagnostics between chains 4, 5, 6”. However, the code shows that the diagnostic was performed on chains 3, 5, 6. Page 21 – “From the plot below is is…” should read “it is”. Is the rationale for developing the new method (or application) clearly explained? Yes Is the description of the method technically sound? Yes Are sufficient details provided to allow replication of the method development and its use by others? Yes If any results are presented, are all the source data underlying the results available to ensure full reproducibility? Yes Are the conclusions about the method and its performance adequately supported by the findings presented in the article? Partly References 1. Gatto L, Breckels LM, Burger T, Nightingale DJ, et al.: A foundation for reliable spatial proteomics data analysis. Mol Cell Proteomics . 2014; 13 (8): 1937-52 PubMed Abstract | Publisher Full Text Competing Interests No competing interests were disclosed. Reviewer Expertise My area of expertise is in interaction, global, and organelle proteomics analysis. My work includes application of experimental, statistical, and computational methods for the study of disease using a proteomics approach. I confirm that I have read this submission and believe that I have an appropriate level of expertise to confirm that it is of an acceptable scientific standard, however I have significant reservations, as outlined above. reply Respond to this report Responses (0) Jean Beltran P. Peer Review Report For: A Bioconductor workflow for the Bayesian analysis of spatial proteomics [version 1; peer review: 1 approved, 2 approved with reservations] . F1000Research 2019, 8 :446 ( https://doi.org/10.5256/f1000research.20403.r47079) NOTE: it is important to ensure the information in square brackets after the title is included in this citation. The direct URL for this report is: https://f1000research.com/articles/8-446/v1#referee-response-47079 Alongside their report, reviewers assign a status to the article: Approved - the paper is scientifically sound in its current form and only minor, if any, improvements are suggested Approved with reservations - A number of small changes, sometimes more significant revisions are required to address specific details and improve the papers academic merit. Not approved - fundamental flaws in the paper seriously undermine the findings and conclusions Adjust parameters to alter display View on desktop for interactive features Includes Interactive Elements View on desktop for interactive features Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Stay Updated Sign up for content alerts and receive a weekly or monthly email with all newly published articles Register with F1000Research Already registered? Sign in Not now, thanks close PLEASE NOTE If you are an AUTHOR of this article, please check that you signed in with the account associated with this article otherwise we cannot automatically identify your role as an author and your comment will be labelled as a “User Comment”. If you are a REVIEWER of this article, please check that you have signed in with the account associated with this article and then go to your account to submit your report, please do not post your review here. If you do not have access to your original account, please contact us . All commenters must hold a formal affiliation as per our Policies . The information that you give us will be displayed next to your comment. User comments must be in English, comprehensible and relevant to the article under discussion. We reserve the right to remove any comments that we consider to be inappropriate, offensive or otherwise in breach of the User Comment Terms and Conditions . Commenters must not use a comment for personal attacks. When criticisms of the article are based on unpublished data, the data should be made available. I accept the User Comment Terms and Conditions Please confirm that you accept the User Comment Terms and Conditions. Affiliation ✕ refresh Please enter your institution. Note: To add your institution or organisation, start typing the name and then select the correct name from the list. Where applicable, the name will appear in both the original language and in English. Do not paste in the name. If the name does not appear in the drop-down list, we will display the information you have entered. ✕ refresh Country/Region * USA UK Canada China France Germany Afghanistan Aland Islands Albania Algeria American Samoa Andorra Angola Anguilla Antarctica Antigua and Barbuda Argentina Armenia Aruba Australia Austria Azerbaijan Bahamas Bahrain Bangladesh Barbados Belarus Belgium Belize Benin Bermuda Bhutan Bolivia Bosnia and Herzegovina Botswana Bouvet Island Brazil British Indian Ocean Territory British Virgin Islands Brunei Bulgaria Burkina Faso Burundi Cambodia Cameroon Canada Cape Verde Cayman Islands Central African Republic Chad Chile China Christmas Island Cocos (Keeling) Islands Colombia Comoros Congo Cook Islands Costa Rica Cote d'Ivoire Croatia Cuba Cyprus Czech Republic Democratic Republic of the Congo Denmark Djibouti Dominica Dominican Republic Ecuador Egypt El Salvador Equatorial Guinea Eritrea Estonia Ethiopia Falkland Islands Faroe Islands Federated States of Micronesia Fiji Finland France French Guiana French Polynesia French Southern Territories Gabon Georgia Germany Ghana Gibraltar Greece Greenland Grenada Guadeloupe Guam Guatemala Guernsey Guinea Guinea-Bissau Guyana Haiti Heard Island and Mcdonald Islands Holy See (Vatican City State) Honduras Hong Kong Hungary Iceland India Indonesia Iran Iraq Ireland Israel Italy Jamaica Japan Jersey Jordan Kazakhstan Kenya Kiribati Kosovo (Serbia and Montenegro) Kuwait Kyrgyzstan Lao People's Democratic Republic Latvia Lebanon Lesotho Liberia Libya Liechtenstein Lithuania Luxembourg Macao Madagascar Malawi Malaysia Maldives Mali Malta Marshall Islands Martinique Mauritania Mauritius Mayotte Mexico Minor Outlying Islands of the United States Moldova Monaco Mongolia Montenegro Montserrat Morocco Mozambique Myanmar Namibia Nauru Nepal Netherlands Antilles New Caledonia New Zealand Nicaragua Niger Nigeria Niue Norfolk Island North Korea North Macedonia Northern Mariana Islands Norway Oman Pakistan Palau Palestinian Territory Panama Papua New Guinea Paraguay Peru Philippines Pitcairn Poland Portugal Puerto Rico Qatar Reunion Romania Russian Federation Rwanda Saint Helena Saint Kitts and Nevis Saint Lucia Saint Pierre and Miquelon Saint Vincent and the Grenadines Samoa San Marino Sao Tome and Principe Saudi Arabia Senegal Serbia Seychelles Sierra Leone Singapore Slovakia Slovenia Solomon Islands Somalia South Africa South Georgia and the South Sandwich Is South Korea South Sudan Spain Sri Lanka Sudan Suriname Svalbard and Jan Mayen Swaziland Sweden Switzerland Syria Taiwan Tajikistan Tanzania Thailand The Gambia The Netherlands Timor-Leste Togo Tokelau Tonga Trinidad and Tobago Tunisia Turkey Turkmenistan Turks and Caicos Islands Tuvalu UK USA Uganda Ukraine United Arab Emirates United States Virgin Islands Uruguay Uzbekistan Vanuatu Venezuela Vietnam Wallis and Futuna West Bank and Gaza Strip Western Sahara Yemen Zambia Zimbabwe Please select your country/region. You must enter a comment. Competing Interests Please disclose any competing interests that might be construed to influence your judgment of the article's or peer review report's validity or importance. Competing Interests Policy Provide sufficient details of any financial or non-financial competing interests to enable users to assess whether your comments might lead a reasonable person to question your impartiality. Consider the following examples, but note that this is not an exhaustive list: Examples of 'Non-Financial Competing Interests' Within the past 4 years, you have held joint grants, published or collaborated with any of the authors of the selected paper. You have a close personal relationship (e.g. parent, spouse, sibling, or domestic partner) with any of the authors. You are a close professional associate of any of the authors (e.g. scientific mentor, recent student). You work at the same institute as any of the authors. You hope/expect to benefit (e.g. favour or employment) as a result of your submission. You are an Editor for the journal in which the article is published. Examples of 'Financial Competing Interests' You expect to receive, or in the past 4 years have received, any of the following from any commercial organisation that may gain financially from your submission: a salary, fees, funding, reimbursements. You expect to receive, or in the past 4 years have received, shared grant support or other funding with any of the authors. You hold, or are currently applying for, any patents or significant stocks/shares relating to the subject matter of the paper you are commenting on. Please state your competing interests The comment has been saved. An error has occurred. Please try again. Cancel Post var lTitle = "A Bioconductor workflow for the Bayesian...".replace("'", ''); var linkedInUrl = "http://www.linkedin.com/shareArticle?url=https://f1000research.com/articles/8-446/v1" + "&title=" + encodeURIComponent(lTitle) + "&summary=" + encodeURIComponent('Read the article by '); var deliciousUrl = "https://del.icio.us/post?url=https://f1000research.com/articles/8-446/v1&title=" + encodeURIComponent(lTitle); var redditUrl = "http://reddit.com/submit?url=https://f1000research.com/articles/8-446/v1" + "&title=" + encodeURIComponent(lTitle); linkedInUrl += encodeURIComponent('Crook OM et al.'); var offsetTop = /chrome/i.test( navigator.userAgent ) ? 4 : -10; var addthis_config = { ui_offset_top: offsetTop, services_compact : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_expanded : "facebook,twitter,www.linkedin.com,www.mendeley.com,reddit.com", services_custom : [ { name: "LinkedIn", url: linkedInUrl, icon:"/img/icon/at_linkedin.svg" }, { name: "Mendeley", url: "http://www.mendeley.com/import/?url=https://f1000research.com/articles/8-446/v1/mendeley", icon:"/img/icon/at_mendeley.svg" }, { name: "Reddit", url: redditUrl, icon:"/img/icon/at_reddit.svg" }, ] }; var addthis_share = { url: "https://f1000research.com/articles/8-446", templates : { twitter : "A Bioconductor workflow for the Bayesian analysis of spatial.... Crook OM et al., published by " + "@F1000Research" + ", https://f1000research.com/articles/8-446/v1" } }; if (typeof(addthis) != "undefined"){ addthis.addEventListener('addthis.ready', checkCount); addthis.addEventListener('addthis.menu.share', checkCount); } $(".f1r-shares-twitter").attr("href", "https://twitter.com/intent/tweet?text=" + addthis_share.templates.twitter); $(".f1r-shares-facebook").attr("href", "https://www.facebook.com/sharer/sharer.php?u=" + addthis_share.url); $(".f1r-shares-linkedin").attr("href", addthis_config.services_custom[0].url); $(".f1r-shares-reddit").attr("href", addthis_config.services_custom[2].url); $(".f1r-shares-mendelay").attr("href", addthis_config.services_custom[1].url); function checkCount(){ setTimeout(function(){ $(".addthis_button_expanded").each(function(){ var count = $(this).text(); if (count !== "" && count != "0") $(this).removeClass("is-hidden"); else $(this).addClass("is-hidden"); }); }, 1000); } close How to cite this report {{reportCitation}} Cancel Copy Citation Details $(function(){R.ui.buttonDropdowns('.dropdown-for-downloads');}); $(function(){R.ui.toolbarDropdowns('.toolbar-dropdown-for-downloads');}); $.get("/articles/acj/18636/20403") new F1000.Clipboard(); new F1000.ThesaurusTermsDisplay("articles", "article", "20403"); $(document).ready(function() { $( "#frame1" ).on('load', function() { var mydiv = $(this).contents().find("div"); var h = mydiv.height(); console.log(h) }); var tooltipLivingFigure = jQuery(".interactive-living-figure-label .icon-more-info"), titleLivingFigure = tooltipLivingFigure.attr("title"); tooltipLivingFigure.simpletip({ fixed: true, position: ["-115", "30"], baseClass: 'small-tooltip', content:titleLivingFigure + " " }); tooltipLivingFigure.removeAttr("title"); $("body").on("click", ".cite-living-figure", function(e) { e.preventDefault(); var ref = $(this).attr("data-ref"); $(this).closest(".living-figure-list-container").find("#" + ref).fadeIn(200); }); $("body").on("click", ".close-cite-living-figure", function(e) { e.preventDefault(); $(this).closest(".popup-window-wrapper").fadeOut(200); }); $(document).on("mouseup", function(e) { var metricsContainer = $(".article-metrics-popover-wrapper"); if (!metricsContainer.is(e.target) && metricsContainer.has(e.target).length === 0) { $(".article-metrics-close-button").click(); } }); var articleId = $('#articleId').val(); if($("#main-article-count-box").attachArticleMetrics) { $("#main-article-count-box").attachArticleMetrics(articleId, { articleMetricsView: true }); } }); var figshareWidget = $(".new_figshare_widget"); if (figshareWidget.length > 0) { window.figshare.load("f1000", function(Widget) { // Select a tag/tags defined in your page. In this tag we will place the widget. _.map(figshareWidget, function(el){ var widget = new Widget({ articleId: $(el).attr("figshare_articleId") //height:300 // this is the height of the viewer part. [Default: 550] }); widget.initialize(); // initialize the widget widget.mount(el); // mount it in a tag that's on your page // this will save the widget on the global scope for later use from // your JS scripts. This line is optional. //window.widget = widget; }); }); } close Error Close Add Reset F1000.MICROSERVICES.AFFILIATION = ''; $(document).ready(function () { $('.js-affiliations-form').each((index, form) => { new AffiliationForm({ formId: form.id, institutionErrorSelector: '.comment-enter-institution', departmentErrorSelector: '.comment-enter-department', placeSelector: '.js-add-comment-place', stateSelector: '.js-add-comment-state', zipCodeSelector: '.js-add-comment-zipcode', countrySelector: '.js-add-comment-country', countryErrorSelector: '.comment-enter-country', }); }); }); $(document).ready(function () { var reportIds = { "47076": 48, "47077": 0, "47078": 0, "47079": 35, "47080": 29, "47081": 0, "47082": 0, }; $(".referee-response-container,.js-referee-report").each(function(index, el) { var reportId = $(el).attr("data-reportid"), reportCount = reportIds[reportId] || 0; $(el).find(".comments-count-container,.js-referee-report-views").html(reportCount); }); var uuidInput = $("#article_uuid"), oldUUId = uuidInput.val(), newUUId = "6f53d74a-f68a-4ac5-87a2-b1f9aa1c2ff6"; uuidInput.val(newUUId); $("a[href*='article_uuid=']").each(function(index, el) { var newHref = $(el).attr("href").replace(oldUUId, newUUId); $(el).attr("href", newHref); }); }); An innovative open access publishing platform offering rapid publication and open peer review, whilst supporting data deposition and sharing. Browse Gateways Collections How it Works Contact For Developers Cookie Notice Privacy Notice RSS Submit Your Research Follow us © 2012-2026 F1000 Research Ltd. ISSN 2046-1402 | Legal | Partner of Research4Life • CrossRef • ORCID • FAIRSharing R.templateTests.simpleTemplate = R.template(' $text $text $text $text $text '); R.templateTests.runTests(); var F1000platform = new F1000.Platform({ name: "f1000research", displayName: "F1000Research", hostName: "f1000research.com", id: "1", editorialEmail: "[email protected]", infoEmail: "[email protected]", usePmcStats: true }); $(function(){R.ui.dropdowns('.dropdown-for-authors, .dropdown-for-about, .dropdown-for-myresearch');}); // $(function(){R.ui.dropdowns('.dropdown-for-referees');}); $(document).ready(function () { if ($(".cookie-warning").is(":visible")) { $(".sticky").css("margin-bottom", "35px"); $(".devices").addClass("devices-and-cookie-warning"); } $(".cookie-warning .close-button").click(function (e) { $(".devices").removeClass("devices-and-cookie-warning"); $(".sticky").css("margin-bottom", "0"); }); $("#tweeter-feed .tweet-message").each(function (i, message) { var self = $(message); self.html(linkify(self.html())); }); $(".partner").on("mouseenter mouseleave", function() { $(this).find(".gray-scale, .colour").toggleClass("is-hidden"); }); }); Sign In Remember me Forgotten your password? Sign In Cancel Email or password not correct. Please try again Please wait... $(function(){ // Note: All the setup needs to run against a name attribute and *not* the id due the clonish // nature of facebox... $("a[id=googleSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("GOOGLE"); $("form[id=oAuthForm]").submit(); }); $("a[id=facebookSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("FACEBOOK"); $("form[id=oAuthForm]").submit(); }); $("a[id=orcidSignInButton]").click(function(event){ event.preventDefault(); $("input[id=oAuthSystem]").val("ORCID"); $("form[id=oAuthForm]").submit(); }); }); If you've forgotten your password, please enter your email address below and we'll send you instructions on how to reset your password. The email address should be the one you originally registered with F1000. Email address not valid, please try again You registered with F1000 via Google, so we cannot reset your password. To sign in, please click here . If you still need help with your Google account password, please click here . You registered with F1000 via Facebook, so we cannot reset your password. To sign in, please click here . If you still need help with your Facebook account password, please click here . Code not correct, please try again Reset password Cancel Email us for further assistance. Server error, please try again. If your email address is registered with us, we will email you instructions to reset your password. If you think you should have received this email but it has not arrived, please check your spam filters and/or contact for further assistance. Please wait... Register $(document).ready(function () { signIn.createSignInAsRow($("#sign-in-form-gfb-popup")); $(".target-field").each(function () { var uris = $(this).val().split("/"); if (uris.pop() === "login") { $(this).val(uris.toString().replace(",","/")); } }); });

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. The paper's references may be in our DB but unresolved to ``paper_id`` (resolution happens at ingest when the cited DOI matches a row we already have). Run the cross-source citation reconcile pass to retry.

Source provenance

europepmc
last seen: 2026-05-19T01:45:01.086888+00:00