Full text
70,902 characters
· extracted from
preprint-html
· click to expand
Leveraging Soil Mapping and Machine Learning to Improve Spatial Adjustments in Plant Breeding Trials | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Leveraging Soil Mapping and Machine Learning to Improve Spatial Adjustments in Plant Breeding Trials Matthew E. Carroll , Luis G. Riera , Bradley A. Miller , Philip M. Dixon , Baskar Ganapathysubramanian , Soumik Sarkar , Asheesh K. Singh doi: https://doi.org/10.1101/2024.01.03.574114 Matthew E. Carroll 1 Department of Agronomy, Iowa State University , Ames, Iowa, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Luis G. Riera 2 Department of Mechanical Engineering, Iowa State University , Ames, Iowa, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Bradley A. Miller 1 Department of Agronomy, Iowa State University , Ames, Iowa, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Philip M. Dixon 3 Department of Statistics, Iowa State University , Ames, Iowa, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Baskar Ganapathysubramanian 2 Department of Mechanical Engineering, Iowa State University , Ames, Iowa, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Soumik Sarkar 2 Department of Mechanical Engineering, Iowa State University , Ames, Iowa, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: singhak{at}iastate.edu soumiks{at}iastate.edu Asheesh K. Singh 1 Department of Agronomy, Iowa State University , Ames, Iowa, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: singhak{at}iastate.edu soumiks{at}iastate.edu Abstract Full Text Info/History Metrics Preview PDF Abstract Spatial adjustments are used to improve the estimate of plot seed yield across crops and geographies. Moving mean and P-Spline are examples of spatial adjustment methods used in plant breeding trials to deal with field heterogeneity. Within trial spatial variability primarily comes from soil feature gradients, such as nutrients, but study of the importance of various soil factors including nutrients is lacking. We analyzed plant breeding progeny row and preliminary yield trial data of a public soybean breeding program across three years consisting of 43,545 plots. We compared several spatial adjustment methods: unadjusted (as a control), moving means adjustment, P-spline adjustment, and a machine learning based method called XGBoost. XGBoost modeled soil features at (a) local field scale for each generation and per year, and (b) all inclusive field scale spanning all generations and years. We report the usefulness of spatial adjustments at both progeny row and preliminary yield trial stages of field testing, and additionally provide ways to utilize interpretability insights of soil features in spatial adjustments. These results empower breeders to further refine selection criteria to make more accurate selections, and furthermore include soil variables to select for macro– and micro-nutrients stress tolerance. 1 Introduction Plant breeders make selection decisions within their programs to advance lines with the highest genetic value for the target population of environments [ 1 ]. Breeders must address non-uniform field conditions (i.e., field heterogeneity) in the selection decision making process, as an incorrect decision causes economic strain on the program and limits success [ 2 ]. Spatial adjustment methods have been proposed to alleviate the challenges of non-uniform field conditions. These methods allow breeders to make more appropriate comparisons of entries to each other, as well as to checks in yield plot testing. Spatial adjustments set up an effective and efficient selection process in the plot testing stages. Spatial adjustments are particularly applicable in unreplicated trials such as early stage yield testing of hybrids in cross-pollinating crop species, and for purelines in progeny row (PR) and preliminary yield trials (PYT) stages in self-pollinating crops. Reducing the size of the error associated with each genotype (i.e., pureline) gives breeders the ability to discriminate between yield levels of competing genotypes for selection more accurately; and therefore several methods have been proposed. Augmented designs (AD) with replicated checks [ 3 ], give some form of local control within a field, giving the breeder a reasonable estimate of the trends that unreplicated genotypes may be experiencing. AD offers breeders the possibility of estimating the standard errors based on the replicated check performance between similar counterparts. Check plot methods can be used by placing checks throughout the field so that a breeder can adjust for field trends based on known checks and their relative performance [ 4 ]. An advantage of the check plot method is that there is no limit to the number of lines used in the trial; on the other hand, it requires a large number of experimental plots to be allocated to the check plots [ 4 ]. The grid method was introduced for a mass selection experiment in maize [ 5 ], where small grids of 40 plants across the field were set up and the top-yielding 10% of plants in each grid were selected to minimize the spatial trends within the field. Another popular method is the moving means (MM) method [ 6 ], where the use of a MM coefficient to adjust yields resulted in a significant decrease in the error when estimating genotypic effects. Most spatial adjustment methods correct for field trends and do not directly quantify the soil features for each plot. A common theme in all spatial adjustment experiments for yield trials is the axiom that these methods reduce the effects of soil heterogeneity, thereby facilitating an unbiased comparison of plot values. While there is a general acceptance of this notion in breeding trials, limited or no information is available that directly quantifies the effects of different soil characteristics and use these values to adjust plot yields. Soil is an integral feature of all breeding testing, with scant consideration of accounting for within field variability. Soil features include calcium, potassium, magnesium, phosphorous, cation exchange capacity, organic matter, pH, sand, silt and clay. Instead of directly quantifying soil variables and their effect on yield, the spatial adjustment methods either model the yield trends or reduce the area of the selection into smaller homogeneous blocks. These methods reduce the standard error of the difference between estimated genotypic values that are being evaluated, and increases the precision in which yield differences can be compared within breeding trials [ 7 ]. The utilization of soil features can provide an association between genotype, soil, and yield opening up avenues for improvement and utilization of spatial adjustments. Despite the use of soil characterization in production fields [ 8 – 10 ], they have not been utilized in breeding trials due to the resolution needed for soil maps, as well as the complexity of parsing out environmental and genotypic variance within a model. Furthermore, in breeding trials, the scale of experiments (# of entries, locations, years etc.) necessitates machine learning (ML) approaches to generate relationships and learn trends along with improved interpretability. Machine learning is a powerful tool in plant science research, with wide ranging trait investigations such as biotic identification and quantification[ 11 – 21 ], organ detection [ 22 – 26 ], microscopic objects [ 27 , 28 ], abiotic stresses [ 29 – 32 ], and crop yield [ 33 – 35 ]. These ML methods provide an opportunity to integrate multiple variables for spatial adjustment while parsing out the trends and role of soil features on plot yield in tests. Among the extensive diversity of methods in ML, “Extreme Gradient Boosting” (XGBoost) [ 36 ] is well suited for the spatial adjustment problem since it is a hierarchical ensemble modeling tree structure that derives its outputs utilizing classification rules. XGBoost is scalable and gives high model accuracy, allows for utilization of multiple features in prediction, and has been shown to be very useful in regression and classification problems [ 36 , 37 ]. XGBoost, like other decision tree methods, is an algorithm that can effectively use multiple variables included in a dataset for prediction. Decision tree structures do not assume a linear relationship between variables, or the predicted values, which can help to predict nonlinear multivariate problems. This structure allows for variables to have multiple weights depending on the values of other variables. This is particularity applicable to soil plant growth interactions, as it is known that plant growth and development is dependent on multiple interacting soil features, weather, and genetics. The motivation for the study was to create a paradigm shift from current spatial adjustments of yield plot data, primarily made on the plot yield values of neighbors, to a method that includes plot yield and soil features. To achieve our goals, digital soil maps and ML were leveraged to model the effect of soil features on plot yield in a soybean breeding program. XGBoost made it possible to adjust for the growing conditions experienced by each genotype. Furthermore, we present the use of Shapley values to aid in the model interpretability and explore its usefulness to breeders by providing previously underutilized knowledge about the soil conditions from trials, and its integration in breeding decisions. Finally, we propose the usefulness of this method to consciously select for nutrient deficiency tolerance in field trials that experience abiotic stresses. 2 Materials and Methods 2.1 Field experiments and data collection Soybean [ Glycine Max (L.) Merr. ] breeding trial plot data were collected in three growing seasons (2019, 2020, 2021). Plot yield data were from progeny rows (coded as A-test; 2019, 2020, 2021) and next stage testing, i.e., preliminary yield trials (coded as B-test; 2019, 2020, 2021). Data was collected from the central Iowa research locations, centered around the Ag Engineering and Agronomy research farm (Boone, IA, USA). A brief description of each field can be found in table 1 , no border rows were used between plots. The centers of plots in the same row were 1.52 meters apart. Plot lengths are recorded as the harvested length of each plot, all fields had a .91 meter alleyway between plots in each column giving a plot to plot center distance of 3.04 and 6.09 meters for plots in the same column for the A and B tests, respectively. To minimize interplot competition, each field consisted of multiple experiments that grouped lines with similar crop maturity, phenology and genetic background. Plots were harvested with an ALMACO (Nevada, IA USA) small plot combine, and yields were adjusted to 13% moisture, and converted to kilograms/hectare (kg/ha) for each harvested plot. View this table: View inline View popup Download powerpoint Table 1: Listing of A-test (i.e., progeny row tests), and B-tests (i.e., preliminary yield trials) that were used in this study. Tests in each year, location, geographical location, and plots details are listed. 2.2 Soil data collection Soil data was collected on a 25 meter grid pattern for each field at a coring depth of 15cm, and then sent to Midwest labs (Omaha, NE USA) for analysis. Soil maps were created from the 25 meter grid samples to a resolution of 3×3 meter grids for soil nutrients and particle size fractions [Calcium (Ca), Cation Exchange Capacity (CEC), Phosphorous (P1), soil pH (PH), Magnesium (MG), Potassium (K), Organic Matter (OM), Clay, Silt, and Sand], using the Cubist machine learning algorithm [ 38 ]. Soil values were extracted for each plot using Esri’s ArcGIS Python library (Redland CA, USA), arcpy, and Zonal Statistics toolbox. The extracted soil values for each plot were then used to calculate mean soil values on a per plot basis using a 3×3 grid, which is represented by Eq. 1 . Where is the mean soil feature value of the r th row and c th column in each field. All further analysis using soil features used the values as input. 2.3 Base Line Spatial Adjustments 2.3.1 Moving Means Plot seed yields were adjusted using Moving Means with a grid size of 5×5 plots, and were calculated using the R package “mvnGrad” [ 39 ]. This method excluded the plot of interest when calculating the mean for each plot. The Moving Means for each experiment within a test were calculated separately. The Moving Means method makes an assumption that neighboring plots do not exhibit interplot competition, and that the nonrandom error observed in a field is due to field environmental trends [ 40 ]. Breeders can reduce the interplot competition by subdividing fields into smaller experiments that are grouped by similar maturities and genetic backgrounds.The process for calculating the Moving Means can be found in the mvnGrad documentation [ 39 ]. Results from the Moving Mean model from here on will be referred to as ADJ 5 MM. 2.3.2 P-Splines P-splines are an alternative method to Moving Means, commonly used in spatial adjustments for field trials. The R package “statgenSTA” [ 41 ] was used to correct for field variations using P-splines as described in Rodriguez et al. [ 42 ]. For this study, genotypes were treated as a random factor. The design was specified as a row-column design, which treats the row and column as random effects, and used the “SpATS” engine. Results from the P-Spline model from here on will be referred to as Spline. 2.4 Soil-Based Adjustments 2.4.1 Statistical Learning XGBoost training process requires tuning parameter coefficients, usually denoted by θ . Among all the parameters, the number of estimators, maximum depth, learning rate, γ , subsample size, and minimum child weights are commonly tuned using a greedy search algorithm. Most XGBoost models are configured using a relatively shallow depth or a small number of trees because of their sequential characteristic, where each new tree corrects the errors made by previous trees, quickly reaching a point of diminishing returns. This training structure works well for soil-based spatial adjustments, because it is able to train models quickly, and model complex data with multiple interactions. Our model used 240 estimators and a depth of 9 for training. Model parameter tuning was done using Sklearn grid search API [ 43 ] with 10-fold cross validation. Two methods were compared for spatial adjustments using XGBoost. Model 1, from here on referred to as XGB Global, combined data from all nine fields and attempted to make a generalized model. Model 2, from here on referred to as XGB Local, was trained on each field individually to create a more specific model for each field location. Both models used the same tuning parameters as listed in table 2 . View this table: View inline View popup Download powerpoint Table 2: Parameters used for the XGBoost model. 2.4.2 Model overview Plant breeders routinely express yield as P = G + E , where P is the observed phenotypic value of a line, e.g., seed yield. G is the genotypic value of the line, which can be estimated when lines are replicated and grown in multiple locations. Lastly, E is the environmental effect, which breeders, try to minimize, as it can confound the estimated genetic value of a line. Historically, E has been a macro-scale environment effect such as a single field, and micro-scale plot level environmental effects have been reduced using various spatial adjustment methods. Furthermore, genotypic value can be better estimated by using phenotypic values of related lines. If the mean of the population is included, the new equation is represented as P = G + E + µ , where µ is the phenoytpic mean of the population. In this context, we define population to consist of purelines from the same breeding family, that is they have the same the pedigree. Our proposed method uses soil features to estimate the value of E for each plot, and to increase the correlation between the adjusted phenotypic values and the estimated genotypic values. The first step was to collect the yield data for each plot in a given field which was in a row and column configuration within each field. Each experiment has multiple lines that are from the same or related populations, and the mean of each population within each experiment was calculated and used as the estimate for the population mean within the experiment. Checks were averaged across the entire field and this average was used as the estimate for the check population mean. Eq. 2 represents the basis for the soil-based adjustment methods. P r,c,j,k is the the seed yield of the plot in the r th row of the c th column from the j th population in the k th experiment. G r,c,j,k is the the genetic value of the plot in the r th row of the c th column from the j th population in the k th experiment. S r,c is the soil feature values of the plot in the r th row of the c th column. µ j,k is the mean of all plots in the j th population of the k th experiment. The next step was to subtract the population mean within each experiment from the observed yield of each plot, giving the phenotypic deviation, δ r,c , from the population mean of each plot, this is shown in Eq. 3 . It is assumed that part of δ r,c can be attributed to the soil features and part can be attributed to the genetic value of the line. The third step was then to train a model, using XGBoost regression, to predict the deviation, δ r,c , from the population mean based on the soil features, S r,c , of each plot. This trained model was then used to estimate the effect of the soil on each plot’s observed yield, represented by in Eq. 4 The last step estimates the genetic value of each plot by taking the observed phenotypic value, and subtracting the estimated effect of the soil, , from the observed phenotypic value, P r,c,j,k , which is represented as in Eq. 5 . was used for further analysis and comparison to other spatial adjustment techniques. 2.5 Model Interpretability ‘SHAP” (SHapley Additive exPlanations) [ 44 ] was used to assist interpreting the results from XGBoost models. SHAP uses traditional Shapley values from game theory, and it has shown to be very successful in explaining the output of any machine learning model [ 45 , 46 ]. Shapley values estimate the feature’s importance in terms of their contribution to the outcome. 2.6 Model Performance for Selection Three evaluation metrics were used to assess the performance of the different methods. The first metric was the relative efficiency using the average standard error of the difference ( SED ) between the check lines in each field. The efficiency of different spatial adjustment methods was calculated using Eq. 6 where SED Y ield is the SED using no spatial adjustments, and SED Adj is the SED using one of the tested spatial adjustment methods. In A-test, entries are not replicated due to seed limitation and in B-test stage breeders prefer to test in more than one location in unreplicated manner; therefore, checks were used to estimate the standard error of difference. Previous works evaluating the efficiency of spatial and experimental designs in plant breeding trials have used this as a method for evaluation of different selection methods [ 7 , 47 , 48 ]. As the standard error of the difference decreases, more of the observed phenotypic yield variance can be attributed to the genotypic value. The number of genetically distinct checks and the average number of times each were replicated in the field can be found in table 1 . The second metric used to evaluate the models was a similarity coefficient called the Czekanowski coefficient ( CZ ), also known as the Sorensen Dice coefficient. This coefficient gives the percent of lines selected by both selection methods at a given selection intensity. The Czekanowski coefficient was calculated according to [ 7 ] where a is the number of lines selected by both selection methods, b is the number of lines selected only by model 1, and c is the number of lines selected only by model 2. CZ is a ratio given by Eq. 7 . If CZ has a value of 1 it would mean that both methods selected the same lines, if CZ has a value of 0 it would mean that there was no overlap in the lines that were selected. For our purposes model 1 used no spatial adjustment, and model 2 was one of the listed spatial adjustment methods. The last metric is the Moran’s I statistic, a measure of the spatial autocorrelation present within a user defined grid [ 49 ]. Our implementation uses a grid of 5 by 5 plots to define the plot neighbors given the row column field design of each test. Neighbor distances used relative coordinates to determine plot neighbors, plot widths and lengths were considered to have a unit of 1. Plots that were within a distance of of a plot were considered neighbors. All other plots were not considered to be neighbors. Moran’s I interpretation is similar to a regular correlation coefficient with values ranging from approximately –1 to 1. A value of 1 is a perfect spatial autocorrelation, and a value of –1 is a negative spatial autocorrelation. The null hypothesis expected values for the Moran’s I statistic are close to zero, but are slightly negative. P-values are also reported with each statistic to assist in the interpretation of the significance of the correlation. Moran’s I statistics were calculated in R version 4.0.3 with the package “spdep” [ 50 ] and the Moran.test function. 3 Results 3.1 Genetic Variation Across three years, 238 experiments were examined across nine individual fields. Experiment sizes ranged from 50 to 370 plots with a mean of 183.0 plots per experiment. The A and B-test plot number means were 209.7 and 124.9 plots respectively. In these 238 experiments, 282 unique breeding populations were assessed. Breeding population sizes ranged from 1 to 263 genotypes with a mean size of 70.6. The average population size in the A and B test, was 133.9 and 42.3, respectively. Population mean seed yields ranged from 74.0 to 6032.3 kilograms per hectare (kg/ha) with a mean yield of 4250.2 (kg/ha). The average population seed yields for the A and B tests were 4391.4 and 3624.8 (kg/ha) respectively. While B-tests have one additional generation of selection in the previous year, the lower seed yield is likely due to shorter plot size in A-test (2.3 m) compared to B-test (5.18 m) giving a larger plot edge effect [ 51 ]. Table 3 gives a further summary of the data across all nine fields. View this table: View inline View popup Download powerpoint Table 3: Listing of the fields evaluated for this study. The number of experiments, average experiment plot number, mean population size, and mean population seed yields are listed. Values in parentheses are the standard deviations. 3.2 Soil Variability Soil core samples were taken for each of the nine fields with ten soil features measured (see M and M). Organic matter ranged from 1.1% to 7.4% with a mean of 3.9%. PH ranged from 3.9 to 8.5 with a mean of 6.1. Clay had a range from 0.1% to 33.9% with an mean of 27.4%. Sand had a range from 12.2% to 42.7% with a mean of 30.4%. Silt had a range from 18.9% to 50.5% with a mean of 41.6%. Calcium parts per million (ppm) ranged from 507.1 to 6360.5 with a mean of 2774.9. CEC (meq/100 g soil) ranged from 9.8 to 36.5 with a mean of 21.1. Potassium (ppm) had a range from 85.4 to 286.1 with a mean of 169.1. Magnesium (ppm) ranged from 107.3 to 768.9 with a mean of 399.7. Phosphorous (ppm) had a range from 5.5 to 73.5 with a mean of 21.7. Table 4 shows the range of each soil parameter for each individual field to better represent the variation that was present across locations. It should be noted that due to a small sized 2020 A AEA field test, there was a lack of variability in the samples taken in the clay content, and this value was constant for that test. View this table: View inline View popup Download powerpoint Table 4: Soil descriptive statistics for each field. The minimum, maximum, and mean values are listed. 3.3 XGBoost and Model Interpretability XGB Local model’s predictions, ground truth and predicted yield histograms, and correlation are shown in Fig. 1 . These predictions are on the centered variables, as described in Eq. 3 as δ r,c . The Concordance Correlation Coefficient between the ground truth and predicted values ranged from 0.71 to 0.86. Centered values, ranged from –81.5 to 48.8. When looking at the two stages of the breeding program, i.e., progeny row A-test and preliminary yield trial B-test, δ r,c values in A-tests ranged from from –81.5 to 48.8 and in B-tests ranged from –47.7 to 32.5. This demonstrates that the more advanced stage of testing has less phenotypic variability. Predicted values, which were obtained from XGB Local model and reflect estimated effects of soil features on seed yield, ranged from –62.7 to 36.3. These values represent S ′ as defined in Eq. 4 . The A-test predicted values ranged from –62.7 to 36.3 and the B-test predicted values ranges from –40.2 to 21.6. Download figure Open in new tab Figure 1: This figure shows the XGB Local Ground true ( δ ) vs prediction ( δ ′ ) scatter plots with their respective Concordance Correlation Coefficient. These charts show how well the model learned the ground true data distribution. The red line shows a 1:1 relationship between predicted and observed values Fig. 2 shows results as an example for the 2021 A Marsden East test. In feature selection, correlation provides a measure of independence between input variables; the higher the correlation, the greater the linear dependency. The heatmap plot in Fig. 2a shows the correlation between the features, and the dendrogram on top provides a clustering mechanism of these features according to their correlation values. The inverse “u” shape-like links group elements into clusters; the lower the “u” shape is in the figure, the higher the correlation of those elements is in relation to the others. The MM OM and MM CEC correlation is 0.9, the highest between elements, and are clustered in the dendrogram first. Then the MM OM MM CEC cluster is merged with MM CA to form another cluster. For this article, highly correlated features were kept to explore their outcomes using Shapley values. Download figure Open in new tab Figure 2: XGBoost model explainability plots using data from 2021 Marsden East Test A. The heatmap (a) shows the correlation between the features used to train the XGBoost model and their respective clustering hierarchy by the dendrogram. The feature importance, figure (b), illustrates a measure of the prediction power of each feature for the model with this dataset. Each dot in the Shapley value Summary plot (c) represents a single plot. Its colors are normalized per feature using its maximum and minimum respectively. The summary plot helps illustrate the relationship between the individual feature value range and their impact over the predicted yield. The Shapley values waterfall plots (d) provide an insight into how features values contribute to the final expected value in two plots (Identities “IA2102”) at different locations in the field where the feature values are substantially different. Fig. 2b provides a summary of the features of importance for the model. The higher the value on the bars in this plot, the greater the information gained by the model when performing the tree branch split during training. In the 2021 Marsden A test East, MM CEC provided the most significant information gain, and the other top four driving prediction features were MM P1, MM MG, and MM PH. To aid in interpreting results we investigated the use of Shapley values. Fig. 2c describes how each plot reacted to the input features. Each dot is a plot, and its color represents the plot feature value relative to the feature’s range scale, where red is high and blue is low. The accumulation of the dots across the x-axis characterizes the global behavior of the features, similar to a histogram. For the 2021 Marsden A test East, we see that high values of MM CEC, MM P1, and MM MG tend to have positive Shapley values. However, it is not conclusive for the other soil features, where we see high feature values on both tails. For our model, we can not draw any causal inference conclusions because we suspect the existence of many confounding factors. The waterfall plots in Fig. 2d describe the reaction of the the check genotype (IA2102) compared to the soil composition in distinct locations of the field used on the 2021 Marsden A test East. A check variety was used so that the comparison is only looking at the soil features and not including additional genetic differences between two lines. The model’s mean output prediction for the entire field is 0.018. If we start from the mean predicted value of 0.018 and add the respective Shapley values, we obtain the model prediction of 4.275 for the right image and –3.602 for the left. This represents the relative quality of the growing conditions for each individual plot, and the expected contribution of each soil feature, as determined by the model. 3.4 Spatial Adjustments Mean relative efficiencies were calculated for each model, with values of 112.8%, 132.8%, 161.6%, and 178.6% for ADJ 5 MM, XGB Global, Spline, and XGB Local. No spatial adjustments was the base model to which all models were compared. The models are ranked from lowest to highest relative efficiency; where the Moving Mean method has the lowest relative efficiency and XGB Local has the highest average relative efficiency across all nine fields that were tested. All of the models presented have a higher relative efficiency than the unadjusted yields. This indicates that XGBoost using plot level soil fertility variables can be used to increase the precision of the genotypic estimates of the checks. These results infer that the estimated genotypic values for unreplicated lines will be more precise as well. Increasing the precision of the estimates of the genetic values of a line will lead to higher selection accuracy and increases in genetic gain. Similarity coefficients were calculated at three selection intensities (0.1, 0.3, and 0.4), and examined the differences in the selected lines based on the method used for spatial adjustments. All models were compared to the unadjusted yield. Figure 3 shows the variation in the CZ value across different selection intensities and the methods of spatial adjustment. The x-axis is the adjustment method, and the y-axis is the CZ coefficient. The median CZ across all three selection intensities was 0.82, 0.74, 0.88, and 0.70 for the Moving Mean, Spline, XGB Global and XGB Local, respectively. To put this in perspective, 18% of the lines that were selected by the Moving Mean method were dissimilar from no-adjustment while 30% of the lines that were selected by the XGB Local model were dissimilar from no-adjustment. These results show that utilizing spatial adjustments results in differences in the lines that are selected for advancement within a breeding program. The lowest median CZ value observed across all selection intensities was the XGB Local model, meaning that the least amount of overlap between selected lines with no spatial adjustment would be achieved by using this method. However, it should be clarified that spatial adjustments are only needed when there is spatial autocorrelation in yield present within the field. In the absence of yield spatial autocorrelation, the use of unadjusted plot mean in statistical analysis will give the same outcome as spatial adjusted means. Download figure Open in new tab Figure 3: Czekanowski box plots for the different methods. This graph shows the similarity in which lines would be selected between the different models using selection intensities of 0.1, 0.3, and 0.4, in comparison to no spatial adjustment. The XGB Global model exhibited the highest CZ coefficient while XGB Local the lowest. The higher the CZ is, the greater the selection overlapping between the models and the selections made based on no spatial adjustment. Within breeding trials where lines have similar genetic backgrounds and maturities, it can be expected that there will be similar performance among lines, and because the lines within a trial are randomized it is highly unlikely that there should be evidence of spatial autocorrelation if the environmental effect has been accounted for. The reduction in the spatial trends observed in trials is a good indication that spatial adjustments were effective at removing non-random field effects. Figure 4 shows an example of an experiment before and after the yield has been adjusted using the XGB Local method. The Moran’s I is reduced from 0.58 to 0.09 Download figure Open in new tab Figure 4: This figure shows the observed yield for both the Unadjusted and Adjusted (XGB Local) yield in an experiment. The Moran’s I for the left image is 0.58, and 0.09 for the right. Figure 5 shows the range of Moran’s I values across all 238 experiments that were tested. The first two methods, ADJ 5 MM and Spline, are used by breeders for selection decision making. The next two methods are variations of the XGBoost method that used soil data instead of neighbor-based adjustments. The last column labeled as Yield is the Moran’s I values calculated based on no spatial adjustment, and serves as the baseline for all of the comparisons that are made. All methods reduce the spatial autocorrelation within trials, with median Moran’s I values of –0.04, –0.03, 0.04, 0.01, 0.13 for Moving Mean, Spline, XGB Global, XGB Local, and Yield spatial adjustment methods respectively. Figure 5 shows the differing levels of spatial correlation across fields, and provides strong evidence that both traditional neighbor-based adjustments, and soil-based adjustments reduced the overall Moran’s I statistic in the majority of trials. It should be noted that the XGB Local method had a median Moran’s I statistic value (0.01) that was closest to 0 compared to all other methods. Download figure Open in new tab Figure 5: This figure shows the variation of the Moran’s I statistic that was calculated for each experiment. The closer the value is to zero the better evidence there is to suggest that there is a minimal field trend. 4 Discussion Directly modeling the effect of soil on a plot level basis has not been done for plant breeding field trials. The reason for this is that the labor and cost required to collect this type of data is cost prohibitive. As newer technologies arise from the combination of remote sensing and ML this type of data can be obtained at a much lower cost than previously possible [ 38 , 52 – 54 ]. Traditional spatial adjustments in field trials have focused on reducing the effect of environmental trends that are evident in most large scale breeding trials. Removing or accounting for the environmental effect of a plot will lead to more accurate selection, and an increase in genetic gain for each breeding program [ 55 , 56 ]. Soil-based adjustments give breeders additional tools when optimizing for selection of multiple traits [ 57 ], such as yield, and biotic stress induced nutrient deficiencies. Soil fertility recommendations for soybean production in the Midwest have primarily been focused on P, K and pH. Optimal recommended P level is to target ≥ 16 ppm, optimal K levels are recommended to be ≥ 161 ppm [ 58 ], and optimal pH range is between 6.5 and 7.5 [ 8 ]. Based on these recommendations, 3 of the 9 fields had an average P level that was below optimum, 7 fields had a mean soil pH outside of the optimal zone, and 4 fields had a mean K level below the optimum. It should be noted that these recommendations are not based on maximizing yield, but based on maximizing profitability. Other factors such as CEC have been reported to have an effect on soil nutrient availability and it’s interactions with plant growth and development [ 59 ]. These soil nutrients have well established univariate effects on soybean growth and development [ 9 , 10 ], but when they are combined within multiple interacting complex systems it becomes difficult to quantify the effect of any individual factor. Previous studies have looked at the effects of multiple soil factors and their effects on predicting yield using Random Forest in production fields and found that some of the topranking predictor variables were P, K, OM, and pH [ 60 ]. Our findings show that these factors had an effect on the quality of the growing conditions of each plot. While feature importance measurements are useful for determining how often a variable is used to split a decision tree, it does not provide any additional insight into the effect that a variable has on a modeled outcome. The three metrics we used to evaluate the different selection methods examine the utility of different spatial adjustment methods. Neighbor-based adjustments make an assumption that neighboring genotypes should be phenologically similar to reduce the effects of interplot competition [ 40 ]. Neighbor-based adjustments are based on linear models, which follow the assumptions that the samples come from the same distribution. If this assumption is no longer valid it can result in poor performance of spatial adjustment methods. This is why experiments usually consist of similar maturity groups and genetic backgrounds, which helps to reduce the interplot competition as genotypes invariably have similar phenological/architectural traits. The XGB Local model had the best metrics for reducing the spatial autocorrelation for seed yield within tests, greatest increase in the mean relative efficiency, and also had the lowest median CZ coefficient across all tests and selection thresholds. For caution, it is important to clarify that dissimilar selection decisions stemming from unadjusted or any of the spatial adjustment method require further replicated field trials in the following year to confirm the accuracy of selection of different methods. Since we used an active breeding pipeline for this research, which used Moving Means in A-test in the previous year, we do not have a completely independent set to compare the efficiency of selection across methods. The advantage of using soil features with XGBoost is that the model is based on the feature space of each field where as neighbor-based adjustments utilize only the location space of each dataset. Utilizing the feature space becomes more important when breeders want to grow dissimilar genotypes in the same test. This happens because the nearest neighbor methods have a proclivity for better estimates for similar phenotypes. The other advantage of XGBoost is interpretability of the adjustments giving increased confidence in breeders’ decision making. The overall results suggest that using a locally trained model using the proposed methodology can increase the efficiency of selection within the breeding program. Interestingly, the local model consistently performed better than global model for applications of spatial adjustments within a breeding program. We hypothesize that this is due to the variation in environments, and that the local models can pick up on environment specific factors. While the global model was not generalized across fields, the non-generalizable models were effective because they were able to determine the local field effects and interactions, whereas the global model became a statistical average of all models and removed the sensitivity that was obtained in the local model that was refined for each individual field. There are several avenues for integration of ML based spatial adjustment models in other on-going areas of research in crop science. For example, crop modeling has been shown to be an effective tool for parsing out the effects of genotype, environment, and management for crop production [ 61 ]. The fusion of better environmental data and increased modeling capabilities, along with HTP methods for genotype specific calibrations, can be used by breeders to make selections for both current and future breeding environments [ 1 , 62 ]. High resolution soil maps and spatial adjustments can be invaluable for these modeling projects, and it has been shown that spatial adjustments can be useful for increasing genomic prediction model accuracies [ 63 , 64 ]. Comprehensive weather and genetics data have been used to predict complex phenotypic traits [ 34 , 65 , 66 ], and integration of soil maps and interpretable spatial adjustments can improve the explainability of genotype response. These advantages extend to study of component traits in seed yield and other physiological endtrait predictions [ 22 , 67 – 69 ], including for ML based predictions and prescriptive breeding approaches [ 33 , 70 ]. As previously explained, soil mapping and interpretable spatial adjustments can be useful to study abiotic and biotic stress responses particularly in conjunction with HTP and ML [ 71 – 73 ]. Furthermore, advances in root trait studies [ 23 , 24 , 74 – 76 ] can be complemented with the use of soil maps and spatial adjustments for a more holistic interpretation of plant response accounting for both above– and below ground traits. There are four main avenues to further improve this work. (a) The setup of this experiment does not allow us to make causal inferences about the true effect of various soil parameters, but it does open the door for future work where researchers can start to investigate nutrient use efficiency for different genotypes [ 77 ], and also the complex interactions of soil, genotypes, weather, and management for optimal plant growth and development. (b) Future work can investigate the use of soil, weather, genotype, and remote sensing data to develop a generalized model for yield prediction. While we included 282 unique populations, it still does not fully represent the entire genetic diversity of the USDA germplasm collection as our testbed program aims to develop elite soybean varieties. We controlled for variations in genetic background by centering populations around their mean values, but more work looking into more crops should be done to determine the effectiveness of this methodology. More recent work that studied the effect of including genetic relationships in spatial adjustment methods has shown that including SNP and pedigree information can help to improve spatial adjustments; however, this was based on simulated data [ 78 ]. As breeders continue to have more tools at their disposal modeling the genetic and environment factors could help to make more informed selections. This can remove non-genetic effects when evaluating lines, and also start to investigate the genetic components that effect soil nutrient use efficiency and help to create more prescriptive cultivars across regions [ 70 ]. (c) The analysis methods presented in our study were not used to compare performance in the following years, to see how different selection methods performed. Future work should investigate the repeatability of methods across years and locations. (d) Variable soil fertility was observed across all testing sites, but the locations that were tested are generally considered to be high yielding environments. Future work investigating the utility of soil mapping in lower productive soil is needed to see the utility of this method in a wider range of environments. 5 Conclusion and Future work We demonstrate the usefulness of spatial adjustment models to improve the selection efficiency in plant breeding programs by removing environmental trends that can bias selection decision making. While spatial adjustments do not give any advantage over unadjusted plot seed yield based analysis when no field variation exist, such a scenario is extremely rare. Therefore, unreplicated trials will benefit from the use of spatial adjustment models. In this paper, we demonstrate the usefulness of a ML method for spatial adjustment in plot experiments. This method models soil fertility trends within a field, and allows breeders to increase their knowledge about the quality of the growing conditions of each plot within a field trial. Compared to unadjusted, Moving Mean and Spline methods, the XGB Local model had the greatest increase in the mean relative efficiency, the lowest median CZ coefficient across all tests and selection thresholds, and showed a consistent ability to reduce field gradient effects on the yield of plots. The XGBoost method we demonstrate is based on soil features, which becomes more important when breeders want to grow genetically diverse and phenologically dissimilar genotypes in the same test where the soil map indicates variability in nutrients, soil texture properties, pH, CEC, and OM. XGBoost provides interpretability of the adjustments that can increase breeders’ confidence in the application of spatial adjustment methods. Additionally, soil-based adjustments provide opportunities to select for genotypes that respond to soil features, for example, nutrient deficiency response. Author contributions A.K.S and M.E.C conceived the research; M.E.C. and A.K.S. performed experiments and data collection; B.A.M. contributed to the interpretation of soil data; P.M.D. contributed to the interpretation of statistical results; M.E.C. and L.G.R. built machine learning models and interpreted the results with inputs from S.S., B.G., and A.K.S.; M.E.C. and L.G.R. prepared the first draft with A.K.S. and S.S. All authors contributed in the development of the manuscript, and reviewed the manuscript. M.E.C. and L.G.R., as co-first authors, made equal contributions to the paper. Funding The authors sincerely appreciate the funding support from the North Central Soybean Research Program (A.K.S.), USDA CRIS project IOW04714 (A.K.S.), AI Institute for Resilient Agriculture (USDA-NIFA #2021-67021-35329) (B.G., S.S., A.K.S.), COALESCE: COntext Aware LEarning for Sustainable CybEr-Agricultural Systems (CPS Frontier # 1954556) (S.S., B.G., A.K.S.), Smart Integrated Farm Network for Rural Agricultural Communities (SIRAC) (NSF S&CC #1952045) (A.K.S., S.S.), RF Baker Center for Plant Breeding (A.K.S.), and Plant Sciences Institute (A.K.S., S.S., B.G.). M.E.C. was partly supported by a graduate assistantship through NSF NRT Predictive Plant Phenomics project. Competing interests None. Data Availability Data and Codes will be made available publicly after acceptance of the paper through corresponding authors’ GitHub pages. Supplementary Material Download figure Open in new tab Figure 6: An example of a trimmed XGBoost tree model stopping at a depth of eight. The circles represent other branches no shown here, the ellipses are the leaf, and boxes are the splitting decision nodes. Download figure Open in new tab Figure 7: XGBoost feature importance plots. The mean(—SHAP Value—) is a weighted measurement based on the number of times a feature was used by the model to split the data across all trees. The splitting is done as function of information gain. The higher the score the more important the the feature is for the given model. The dendrogram on the right of the plots represent how the model clusters the different features using average euclidean distance and threshold of one. Download figure Open in new tab Figure 8: Explanatory Shapley Values summary plots. Download figure Open in new tab Figure 9: Explanatory Waterfall charts for genotype IA2102 across the nine experiments. There are two Waterfall per subplot, each showing how the feature values contributed to the final predicted yield at the given range and row field coordinates. Feature measurements are to the left of the feature names, while the Shapley values are within the arrow bar. The predicted means for the model are on the x-axis, and the predictions for the plot, shown on top, are obtained by adding and subtracting the respective Shapley values. Acknowledgments General We thank staff and student members of Singh Soybean group at ISU, particularly Brian Scott, Will Doepke, Jennifer Hicks, and Ryan Dunn for their assistance with field experiments and phenotyping. We thank Sarah Jones and Ashlyn Rairdin for help in editing the manuscript. We thank Yones Khaledian and Caner Ferhatoglu for generating the soil maps, and members of the Geospatial Laboratory for Soil Informatics group for their assistance in soil core sampling. References [1]. ↵ M. Cooper et al. , “ Modelling selection response in plant-breeding programs using crop models as mechanistic gene-to-phenotype (cgm-g2p) multi-trait link functions ,” in silico Plants , vol. 3 , no. 1 , diaa016 , 2021 . OpenUrl [2]. ↵ D. P. Singh , A. K. Singh , and A. Singh , Plant breeding and cultivar development . Academic Press , 2021 . [3]. ↵ W. T. Federer , “ Augmented designs with one-way elimination of heterogeneity ,” Biometrics , vol. 17 , no. 3 , pp. 447 – 473 , 1961 . OpenUrl CrossRef [4]. ↵ R. Kempton , “ The design and analysis of unreplicated field trials ,” Vortraege fuer Pflanzenzuechtung (Germany ) , 1984 . [5]. ↵ C. Gardner , “ An evaluation of effects of mass selection and seed irradiation with thermal neutrons on yield of corn ,” Crop Science , vol. 1 , no. 4 , 1961 . [6]. ↵ F. D. Richey , “ Adjusting yields to their regression on a moving average, as a means of correcting for soil heterogeneity ,” J. Agric. Res ., vol. 27 , no. 1-13 , pp. 79 – 90 , 1924 . OpenUrl [7]. ↵ C. Qiao , K. Basford , I. DeLacy , and M. Cooper , “ Evaluation of experimental designs and spatial analyses in wheat breeding trials ,” Theoretical and Applied Genetics , vol. 100 , no. 1 , pp. 9 – 16 , 2000 . OpenUrl CrossRef [8]. ↵ C. McGrath , D. Wright , A. P. Mallarino , and A. W. Lenssen , “ Soybean nutrient needs ,” Agriculture and Environment Extension Publications , vol. 189 , no. 7 , 2013 . [9]. ↵ A. Mallarino , J. Webb , and A. Blackmer , “ Soil test values and grain yields during 14 years of potassium fertilization of corn and soybean ,” Journal of Production Agriculture , vol. 4 , no. 4 , pp. 560 – 567 , 1991 . OpenUrl [10]. ↵ J. R. Dodd and A. P. Mallarino , “ Soil-test phosphorus and crop grain yield responses to long-term phosphorus fertilization for corn-soybean rotations ,” Soil Science Society of America Journal , vol. 69 , no. 4 , pp. 1118 – 1128 , 2005 . OpenUrl Web of Science [11]. ↵ K. Nagasubramanian , S. Jones , A. K. Singh , S. Sarkar , A. Singh , and B. Ganapathysubramanian , “ Plant disease identification using explainable 3d deep learning on hyperspectral images ,” Plant methods , vol. 15 , no. 1 , pp. 1 – 10 , 2019 . OpenUrl CrossRef [12]. A. Singh et al. , “ Challenges and opportunities in machine-augmented plant stress phenotyping ,” Trends in Plant Science , vol. 26 , no. 1 , pp. 53 – 69 , 2021 . OpenUrl [13]. E. C. Tetila et al. , “ Automatic recognition of soybean leaf diseases using uav images and deep convolutional neural networks ,” IEEE geoscience and remote sensing letters , vol. 17 , no. 5 , pp. 903 – 907 , 2019 . OpenUrl [14]. A. K. Singh , B. Ganapathysubramanian , S. Sarkar , and A. Singh , “ Deep learning for plant stress phenotyping: Trends and future perspectives ,” Trends in plant science , vol. 23 , no. 10 , pp. 883 – 898 , 2018 . OpenUrl CrossRef [15]. S. Ghosal , D. Blystone , A. K. Singh , B. Ganapathysubramanian , A. Singh , and S. Sarkar , “ An explainable deep machine vision framework for plant stress phenotyping ,” Proceedings of the National Academy of Sciences , vol. 115 , no. 18 , pp. 4613 – 4618 , 2018 . OpenUrl Abstract / FREE Full Text [16]. A. Rairdin et al. , “ Deep learning-based phenotyping for genome wide association studies of sudden death syndrome in soybean ,” Frontiers in plant science , vol. 13 , 2022 . [17]. K. Nagasubramanian , S. Jones , S. Sarkar , A. K. Singh , A. Singh , and B. Ganapathysubramanian , “ Hyperspectral band selection using genetic algorithm and support vector machines for early identification of charcoal rot disease in soybean stems ,” Plant methods , vol. 14 , no. 1 , p. 86, 2018 . [18]. A. Singh , B. Ganapathysubramanian , A. K. Singh , and S. Sarkar , “ Machine learning for high-throughput stress phenotyping in plants ,” Trends in plant science , vol. 21 , no. 2 , pp. 110 – 124 , 2016 . OpenUrl CrossRef [19]. K. Nagasubramanian , A. K. Singh , A. Singh , S. Sarkar , and B. Ganapathysubramanian , “ Usefulness of interpretability methods to explain deep learning based plant stress phenotyping ,” arXiv preprint arXiv:2007.05729 , 2020 . [20]. K. Nagasubramanian , A. Singh , A. Singh , S. Sarkar , and B. Ganapathysubramanian , “ Plant phenotyping with limited annotation: Doing more with less ,” The Plant Phenome Journal , vol. 5 , no. 1 , e20051 , 2022 . OpenUrl [21]. ↵ S. Kar et al. , “ Self-supervised learning improves agricultural pest classification ,” in AI for Agriculture and Food Systems , 2021 . [22]. ↵ L. G. Riera et al. , “ Deep multiview image fusion for soybean yield estimation in breeding applications ,” Plant Phenomics , vol. 2021 , 2021 . [23]. ↵ K. G. Falk et al. , “ Soybean root system architecture trait study through genotypic, phenotypic, and shape-based clusters ,” Plant Phenomics , vol. 2020 , 2020 . [24]. ↵ T. Z. Jubery , C. N. Carley , A. Singh , S. Sarkar , B. Ganapathysubramanian , and A. K. Singh , “ Using machine learning to develop a fully automated soybean nodule acquisition pipeline (snap) ,” Plant Phenomics , vol. 2021 , 2021 . [25]. C. Miao , Z. Xu , E. Rodene , J. Yang , J. C. Schnable , et al. , “ Semantic segmentation of sorghum using hyperspectral data identifies genetic associations ,” Plant Phenomics , vol. 2020 , 2020 . [26]. ↵ S. Ghosal et al. , “ A weakly supervised deep learning framework for sorghum head detection and counting ,” Plant Phenomics , vol. 2019 , p. 1 525 874 , 2019 . OpenUrl [27]. ↵ A. Akintayo , G. L. Tylka , A. K. Singh , B. Ganapathysubramanian , A. Singh , and S. Sarkar , “ A deep learning framework to discern and count microscopic nematode eggs ,” Scientific reports , vol. 8 , no. 1 , pp. 1 – 11 , 2018 . OpenUrl [28]. ↵ S. Fudickar , E. J. Nustede , E. Dreyer , and J. Bornhorst , “ Mask r-cnn based c. elegans detection with a diy microscope ,” Biosensors , vol. 11 , no. 8 , p. 257, 2021 . [29]. ↵ A. A. Dobbels and A. J. Lorenz , “ Soybean iron deficiency chlorosis high-throughput phenotyping using an unmanned aircraft system ,” Plant methods , vol. 15 , no. 1 , pp. 1 – 9 , 2019 . OpenUrl CrossRef [30]. J. Zhang et al. , “ Computer vision and machine learning for robust phenotyping in genome-wide studies ,” Scientific Reports , vol. 7 , no. 1 , pp. 1 – 11 , 2017 . OpenUrl [31]. H. S. Naik et al. , “ A real-time phenotyping framework using machine learning for plant stress severity rating in soybean ,” Plant methods , vol. 13 , no. 1 , pp. 1 – 12 , 2017 . OpenUrl CrossRef [32]. ↵ S. Chiranjeevi et al. , “ Exploring the use of 3d point cloud data for improved plant stress rating ,” in AI for Agriculture and Food Systems , 2021 . [33]. ↵ K. Parmley , K. Nagasubramanian , S. Sarkar , B. Ganapathysubramanian , and A. K. Singh , “ Development of optimized phenomic predictors for efficient plant breeding decisions using phenomic-assisted selection in soybean ,” Plant Phenomics , vol. 2019 , 2019 . [34]. ↵ J. Shook , T. Gangopadhyay , L. Wu , B. Ganapathysubramanian , S. Sarkar , and A. K. Singh , “ Crop yield prediction integrating genotype and weather variables using deep learning ,” Plos one , vol. 16 , no. 6 , e0252402 , 2021 . OpenUrl [35]. ↵ V. Sagan et al. , “ Field-scale crop yield prediction using multi-temporal worldview-3 and planetscope satellite data and deep learning ,” ISPRS journal of photogrammetry and remote sensing , vol. 174 , pp. 265 – 281 , 2021 . OpenUrl [36]. ↵ T. Chen and C. Guestrin , “ XGBoost: A scalable tree boosting system ,” in Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining , ser. KDD’16, San Francisco, California, USA: ACM, 2016 , pp. 785 – 794 , ISBN: 978-1-4503-4232-2. DOI: 10.1145/2939672.2939785 . [Online]. Available: http://doi.acm.org/10.1145/2939672.2939785 . OpenUrl CrossRef [37]. ↵ M. Herrero-Huerta , P. Rodriguez-Gonzalvez , and K. M. Rainey , “ Yield prediction by machine learning from uas-based multi-sensor data fusion in soybean ,” Plant Methods , vol. 16 , no. 1 , pp. 1 – 16 , 2020 . OpenUrl CrossRef [38]. ↵ Y. Khaledian and B. A. Miller , “ Selecting appropriate machine learning methods for digital soil mapping ,” Applied Mathematical Modelling , vol. 81 , pp. 401 – 418 , 2020 . OpenUrl [39]. ↵ F. Technow , “ R package mvnggrad: Moving grid adjustment in plant breeding field trials ,” R package version 0.1 , vol. 5 , 2015 . [40]. ↵ I. Bos and P. Caligari , Selection methods in plant breeding . Springer Science & Business Media , 2007 . [41]. ↵ B.-J. van Rossum , F. van Eeuwijk , and M. Boer , “ Package ‘statgensta’ ,” 2021 . [42]. ↵ M. X. Rodriguez-Alvarez , M. P. Boer , F. A. van Eeuwijk , and P. H. Eilers , “ Correcting for spatial heterogeneity in plant breeding experiments with p-splines ,” Spatial Statistics , vol. 23 , pp. 52 – 71 , 2018 . OpenUrl CrossRef [43]. ↵ L. Buitinck et al. , “ API design for machine learning software: Experiences from the scikit-learn project ,” in ECML PKDD Workshop: Languages for Data Mining and Machine Learning , 2013 , pp. 108 – 122 . [44]. ↵ S. M. Lundberg and S.-I. Lee , “ A unified approach to interpreting model predictions ,” in Advances in Neural Information Processing Systems , I. Guyon et al. , Eds., vol. 30 , Curran Associates, Inc ., 2017 . [45]. ↵ E. Štrumbelj and I. Kononenko , “ Explaining prediction models and individual predictions with feature contributions ,” Knowledge and information systems , vol. 41 , no. 3 , pp. 647 – 665 , 2014 . OpenUrl [46]. ↵ A. Shrikumar , P. Greenside , and A. Kundaje , “ Learning important features through propagating activation differences ,” in International conference on machine learning , PMLR, 2017 , pp. 3145 – 3153 . [47]. ↵ S. Magnussen , “ Application and comparison of spatial models in analyzing tree-genetics field trials ,” Canadian Journal of Forest Research , vol. 20 , no. 5 , pp. 536 – 546 , 1990 . OpenUrl [48]. ↵ A. R. Gilmour , B. R. Cullis , and A. P. Verbyla , “ Accounting for natural and extraneous variation in the analysis of field experiments ,” Journal of Agricultural, Biological, and Environmental Statistics , pp. 269 – 293 , 1997 . [49]. ↵ R. S. Bivand and D. W. Wong , “ Comparing implementations of global and local indicators of spatial association ,” Test , vol. 27 , no. 3 , pp. 716 – 748 , 2018 . OpenUrl [50]. ↵ R. Bivand et al. , “ Package ‘spdep’ ,” The Comprehensive R Archive Network , 2015 . [51]. ↵ D. Stelling , M. Ismail , E. Ebmeyer , M. Fraukn , and G. Robbelen , “ Selection in early generations of dried peas, pisum sativum l. iii. plot size and plot type ,” Plant breeding , vol. 105 , no. 3 , pp. 238 – 247 , 1990 . OpenUrl [52]. ↵ B. Minasny , B. I. Setiawan , S. K. Saptomo , A. B. McBratney , et al. , “ Open digital mapping as a costeffective method for mapping peat thickness and assessing the carbon stock of tropical peatlands ,” Geoderma , vol. 313 , pp. 25 – 40 , 2018 . OpenUrl [53]. C. Ferhatoglu and B. A. Miller , “ Choosing feature selection methods for spatial modeling of soil fertility properties at the field scale ,” Agronomy , vol. 12 , no. 8 , p. 1786 , 2022 . OpenUrl [54]. ↵ B. Minasny and A. B. McBratney , “ Digital soil mapping: A brief history and some lessons ,” Geoderma , vol. 264 , pp. 301 – 311 , 2016 . OpenUrl GeoRef [55]. ↵ L. Hazel and J. L. Lush , “ The efficiency of three methods of selection ,” Journal of Heredity , vol. 33 , no. 11 , pp. 393 – 399 , 1942 . OpenUrl CrossRef Web of Science [56]. ↵ J. E. Rutkoski , “ A practical guide to genetic gain ,” Advances in agronomy , vol. 157 , pp. 217 – 249 , 2019 . OpenUrl [57]. ↵ D. Akdemir , W. Beavis , R. Fritsche-Neto , A. K. Singh , and J. Isidro-Sánchez , “ Multi-objective optimized genomic breeding strategies for sustainable food improvement ,” Heredity , vol. 122 , no. 5 , pp. 672 – 683 , 2019 . OpenUrl [58]. ↵ P. Antonio and E. John , “ Interpretation of soil test results ,” Iowa State University Extension Outreach. October. PM , vol. 1310 , 2013 . [59]. ↵ A. P. Brooker , L. E. Lindsey , S. W. Culman , S. K. Subburayalu , and P. R. Thomison , “Low soil phosphorus and potassium limit soybean grain yield in ohio,” Crop , Forage & Turfgrass Management , vol. 3 , no. 1 , pp. 1 – 5 , 2017 . OpenUrl [60]. ↵ E. R. Smidt , S. P. Conley , J. Zhu , and F. J. Arriaga , “ Identifying field attributes that predict soybean yield using random forest analysis ,” Agronomy Journal , vol. 108 , no. 2 , pp. 637 – 646 , 2016 . OpenUrl [61]. ↵ J. J. Ojeda et al. , “ Quantifying the effects of varietal types × management on the spatial variability of sorghum biomass across us environments ,” GCB Bioenergy , vol. 14 , no. 3 , pp. 411 – 433 , 2022 . OpenUrl [62]. ↵ M. D. Krause , K. O. das Gracas Dias, A. K. Singh, and W. D. Beavis, “Using large soybean historical data to study genotype by environment variation and identify mega-environments with the integration of genetic and non-genetic factors , ” bioRxiv , 2022 . [63]. ↵ B. Lado et al. , “ Increased genomic prediction accuracy in wheat breeding through spatial adjustment of field trial data ,” G3: Genes, Genomes, Genetics , vol. 3 , no. 12 , pp. 2105 – 2114 , 2013 . OpenUrl [64]. ↵ X. Mao , S. Dutta , R. K. Wong , and D. Nettleton , “Adjusting for spatial effects in genomic prediction,” Journal of Agricultural , Biological and Environmental Statistics , vol. 25 , pp. 699 – 718 , 2020 . OpenUrl [65]. ↵ X. Li , T. Guo , Q. Mu , X. Li , and J. Yu , “ Genomic and environmental determinants and their interplay underlying phenotypic plasticity ,” Proceedings of the National Academy of Sciences , vol. 115 , no. 26 , pp. 6679 – 6684 , 2018 . OpenUrl Abstract / FREE Full Text [66]. ↵ J. M. Shook , D. Lourenco , and A. K. Singh , “ Patriot: A pipeline for tracing identity-by-descent for chromosome segments to improve genomic prediction in self-pollinating crop species ,” Frontiers in plant science , p. 2095, 2021 . [67]. ↵ M. V. Chiozza , K. A. Parmley , R. H. Higgins , A. K. Singh , and F. E. Miguez , “ Comparative prediction accuracy of hyperspectral bands for different soybean crop variables: From leaf area to seed composition ,” Field Crops Research , vol. 271 , p. 108 260, 2021 . OpenUrl [68]. A. Xavier , B. Hall , A. A. Hearst , K. A. Cherkauer , and K. M. Rainey , “ Genetic architecture of phenomic-enabled canopy coverage in glycine max ,” Genetics , vol. 206 , no. 2 , pp. 1081 – 1089 , 2017 . OpenUrl Abstract / FREE Full Text [69]. ↵ F. F. Moreira , A. A. Hearst , K. A. Cherkauer , and K. M. Rainey , “ Improving the efficiency of soybean breeding with high-throughput canopy phenotyping ,” Plant Methods , vol. 15 , no. 1 , pp. 1 – 9 , 2019 . OpenUrl CrossRef [70]. ↵ K. A. Parmley , R. H. Higgins , B. Ganapathysubramanian , S. Sarkar , and A. K. Singh , “ Machine learning approach for prescriptive plant breeding ,” Scientific reports , vol. 9 , no. 1 , pp. 1 – 12 , 2019 . OpenUrl [71]. ↵ A. K. Singh et al. , “ High-throughput phenotyping in soybean ,” in High-Throughput Crop Phenotyping , Springer , 2021 , pp. 129 – 163 . [72]. M. Reynolds et al. , “ Breeder friendly phenotyping ,” Plant Science , vol. 295 , p. 110 396 , 2020 . [73]. ↵ W. Guo et al. , “ Uas-based plant phenotyping for research and breeding applications ,” Plant Phenomics , vol. 2021 , 2021 . [74]. ↵ J. P. Lynch and K. M. Brown , “ Topsoil foraging–an architectural adaptation of plants to low phosphorus availability ,” Plant and Soil , vol. 237 , no. 2 , pp. 225 – 237 , 2001 . OpenUrl CrossRef Web of Science [75]. K. G. Falk et al. , “ Computer vision and machine learning enabled soybean root phenotyping pipeline ,” Plant methods , vol. 16 , no. 1 , p. 5 , 2020 . OpenUrl [76]. ↵ C. Carley , M. Zubrod , S. Dutta , and A. K. Singh , “ Using machine learning enabled phenotyping to characterize nodulation in three early vegetative stages in soybean ,” bioRxiv , 2022 . DOI: 10.1101/2022.09.28.509969 . eprint: https://www.biorxiv.org/content/early/2022/09/30/2022.09.28.509969.full.pdf . [Online]. Available: https://www.biorxiv.org/content/early/2022/09/30/2022.09.28.509969 . [77]. ↵ V. Baligar , N. Fageria , and Z. He , “ Nutrient use efficiency in plants ,” Communications in soil science and plant analysis , vol. 32 , no. 7-8 , pp. 921 – 950 , 2001 . OpenUrl CrossRef Web of Science [78]. ↵ É. D. Borges da Silva , A. Xavier , and M. V. Faria , “ Joint modeling of genetics and field variation in plant breeding trials using relationship and different spatial methods: A simulation study of accuracy and bias ,” Agronomy , vol. 11 , no. 7 , p. 1397 , 2021 . OpenUrl View the discussion thread. Back to top Previous Next Posted January 04, 2024. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Leveraging Soil Mapping and Machine Learning to Improve Spatial Adjustments in Plant Breeding Trials Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Leveraging Soil Mapping and Machine Learning to Improve Spatial Adjustments in Plant Breeding Trials Matthew E. Carroll , Luis G. Riera , Bradley A. Miller , Philip M. Dixon , Baskar Ganapathysubramanian , Soumik Sarkar , Asheesh K. Singh bioRxiv 2024.01.03.574114; doi: https://doi.org/10.1101/2024.01.03.574114 Share This Article: Copy Citation Tools Leveraging Soil Mapping and Machine Learning to Improve Spatial Adjustments in Plant Breeding Trials Matthew E. Carroll , Luis G. Riera , Bradley A. Miller , Philip M. Dixon , Baskar Ganapathysubramanian , Soumik Sarkar , Asheesh K. Singh bioRxiv 2024.01.03.574114; doi: https://doi.org/10.1101/2024.01.03.574114 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Plant Biology Subject Areas All Articles Animal Behavior and Cognition (7644) Biochemistry (17726) Bioengineering (13916) Bioinformatics (42033) Biophysics (21486) Cancer Biology (18635) Cell Biology (25549) Clinical Trials (138) Developmental Biology (13397) Ecology (19940) Epidemiology (2067) Evolutionary Biology (24361) Genetics (15620) Genomics (22541) Immunology (17763) Microbiology (40468) Molecular Biology (17207) Neuroscience (88739) Paleontology (667) Pathology (2842) Pharmacology and Toxicology (4834) Physiology (7659) Plant Biology (15175) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9834) Zoology (2272)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.