Full text
23,262 characters
· extracted from
preprint-html
· click to expand
Open source software for tube vocal tract modeling, resonance prediction, illustration, and 3D printing | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Confirmatory Results Open source software for tube vocal tract modeling, resonance prediction, illustration, and 3D printing Runhui Song , Jonas Beskow , Jens Edlund , Mechtild Tronnier , Rui Tu , Kecheng Zhang , Axel Ekström doi: https://doi.org/10.1101/2025.10.15.682256 Runhui Song 1 Department of Linguistics and Philology, Uppsala University 2 Speech, Music & Hearing, KTH Royal Institute of Technology Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: runhuisong{at}163.com axeleks{at}kth.se Jonas Beskow 2 Speech, Music & Hearing, KTH Royal Institute of Technology Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jens Edlund 2 Speech, Music & Hearing, KTH Royal Institute of Technology Find this author on Google Scholar Find this author on PubMed Search for this author on this site Mechtild Tronnier 3 Centre for Languages and Literature, Lund University Find this author on Google Scholar Find this author on PubMed Search for this author on this site Rui Tu 1 Department of Linguistics and Philology, Uppsala University Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kecheng Zhang 2 Speech, Music & Hearing, KTH Royal Institute of Technology Find this author on Google Scholar Find this author on PubMed Search for this author on this site Axel Ekström 2 Speech, Music & Hearing, KTH Royal Institute of Technology 4 Centre for Cultural Evolution, Department of Psychology, Stockholm University Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: runhuisong{at}163.com axeleks{at}kth.se Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract We present accessible code-free tube vocal tract modeling software. The software implements a transfer function. Applications involve exploratory speech acoustics-based basic research and education in phonetic sciences. The program has been made publicly available and presents researchers and students in speech-centric sciences with easily accessible vocal tract modelling and vowel synthesis. Introduction Tube models of the vocal tract are a historically influential method for research in speech production and acoustics ( 1 – 6 ). The modern iteration of this concept, with foundations in simulation through electrical engineering traces back to central work conducted by Chiba and Kajiyama ( 7 ) in Japan, Stevens ( 1 , 8 ) and Fant ( 2 , 3 ) in the United States and Europe. However, while the assumption is largely intuitive, its implementation has often been by engineers alone. This work is concerned with teaching the elemental aspects of speech acoustics to an audience largely unfamiliar with computational work. Generally, computational phonetics is performed by engineers ( 2 , 3 , 9 ) and as such place significant restrictions on participation by students in general linguistics. Here, we introduce TubeN – an interactable graphical user interface allowing easy simulation of tube vocal tract models and their acoustic properties. The ultimate goal of this endavour is to invite a greater number of students from general linguistics to participate in the design and execution of computational speech–centric work, and to make available to broader audiences the intricasies of speech acoustics. Simulating the behavior of an acoustic tube TubeN implements an algorithm developed by Liljencrants and Fant ( 3 ). 1 The algorithm calculates the formant frequencies for an acoustic tube with M cylindrical segments, characterized by input parameters length L n and cross-sectional area A n . It then recursively computes a determinant through tube segments. 2 the value of which is called the transfer determinant Δ n , reflecting the impedance transformation up to the n th tube segment. The angular frequency ω , measured in radians per second ( rad/s ) is defined as: where F represents the frequency of the sound wave measured in Hertz ( Hz ), which corresponds to the specific acoustic frequency being simulated. For example, if the response of the tube to a 500 Hz sound wave is studied, then F = 500 Hz . The normalized phase angle of the n th tube segment is: where c = 35300 cm/s is the speed of sound at 35 °C , and L n is the length of the n th segment. The ratio of the area of two connected tube segments ( A n +1 and A n ) is represented as: The recursive formula for the transfer determinant is: where: After obtaining the determinant of the final tube segment Δ M , a quasi-spectral function is constructed: Download figure Open in new tab Fig. 1. Note that where two subsequent segments are of the same area, they are rendered as a single joined segment. Functionalities of the graphical user interface The software will predict formants for any sequence, and the GUI will automatically update the displayed predicted formants F1-F4 upon any change to the sequence. Adding, removing and altering segments The Add function adds n segments by entering their length l and area a in the following format: The order of input is lips to glottis, such that l 1 corresponds to the anteriormost section (i.e., lip opening). The Remove function lets the user remove any section in the tube sequence. It is operated by first double-clicking the relevant segment, and pressing the Remove button on the GUI. The Alter function lets a user edit the length and area of any one segment after initial input. It is operated similarly to the Remove function, by first double-clicking the relevant segment, and pressing the Alter button on the GUI. Importantly, however, segments can also be edited manually, by clicking a segment, and changing its length (by pressing the Left arrow key ← (to shorten) and Right arrow key →, (to elongate) respectively. The area a of the n th segment can be similarly altered using the Up arrow key ↑ (to expand) and Down arrow key ↓ (to contract). These “steps” are executed in increments of 0.1 cm (for length l ) and 0.1 cm 2 (for area a ). For each such increment, the formant prediction window is updated accordingly, allowing for up-close inspection to changes. Finally, pressing the [TAB] key selects the segment to the right. If not segment it selected, pressing [TAB] will select the left-most segment. The Obliviate function resets the model, and intializes a blank sheet. Download figure Open in new tab Fig. 2. TubeN-generated 3D vocal tract. Shape is identical to that in Figure 1 . Reference tube models corresponding to the corner vowels [a i u], are available with single-button presses. These shapes are as reported by Fant (1971) for an adult male Russian speaker. Synthesis A simplistic vowel synthesis function is implemented, allowing quick-and-easy user evaluation of predicted vowel qualities. A small triangular button is placed to the left of the tube model, and users can press it to play the synthesized sound. At present, the synthesized vowel is kept at a length of 1 second, with a constant fundamental frequency of 100 Hz. More variable vowel synthesis methods are slated for implementation in future iterations. Illustration The Illustrate function allows for quick generation of illustrations for tube models, transfer and peak functions. These illustrations are intended to clearly illustrate predicted formant characteristics, and can be saved to the user’s hard drive, and may potentially be used in relevant publications or teaching materials. 3D Modeling and 3D Printing The 3D File function allows for simplistic 3D modeling and printing of any arbitrary tube sequence. This button creates a 3D-printable.stl file, modeled after the currently implemented tube sequence. Because different 3D printers may require additional customization, we note that users may also slice the resultant file, and customize the relevant G-code prior to printing the tube models. We primarily recommend methods made available through the Trimesh Python library 3 for this purpose. In our example 3D printing, we used a Prusa i3 MK3.9S 3D printer, and printed tube sequences as Polyactic Acid (PLA). We found that when a stand-in voice source (e.g., a duck call) was introduced, the intended vowel quality was reliably reproduced, provided that sufficient closure was achieved at the “closed” (i.e., the segment corresponding to the “glottis”) end of the sequence. If closure was incomplete, vowel quality was predictably degraded due to leakage. Our experience is that printing a tube model normally approx. 2.5 hours. Note that these observations are based on only the settings described. 3D printers may utilize a range of different materials, or require unique considerations. For this reason, we cannot categorically claim that our settings are universally applicable. For example, different materials may be more or less appropriate for recreating intended vowel qualities with physical models. Finally, we do not expound on additional challenges associated with the 3D printing process; however, for the sake of increasing usability, we have made our reflections available to the reader. These can be accessed in the same GitHub folder as the TubeN program itself. 4 View this table: View inline View popup Download powerpoint Table 1. Starting lengths and area sections in Exercise 2. Segments are arranged in order from front to back (i.e., lips to glottis). Use Case: Teaching of source/filter theory As tube vocal tract modeling is known to facilitate teaching the principles of speech acoustics to broader audiences ( 6 ), a key reason for developing the TubeN software was to make tube modeling more widely available to students, teachers, and researchers outside of engineering. Below, we briefly describe our experiences conducting two workshops, offered to phonetics students, with this aim. In–person workshops took place at Lund University, Lund, Sweden, as part of two courses – one first-cycle course, and one second–cycle course (both worth 7.5 credits each in the European Credit Transfer and Accumulation System). Students in the basic-level course took part in the workshop as part of their introductions to general phonetic science. Students in the advanced-level course were course were expected to have backgrounds in linguistics and general linguistic and phonetic theory, but little to no empirical experience of speech production analysis or modeling. Student reception was generally mixed to positive. However, in particular at the undergraduate levels, students gave positive evaluations. Exercise 1 In a first exercise, students were asked to convert an MRI image of a speaker’s vocal tract into two-dimensional models. 5 Participants were informed about the age and sex of the speaker (an adult male), but not of the vowel spoken (long close back rounded vowel [u:]). Students were instructed to trace the effective vocal tract, estimate its length given provided knowledge. They were then to segment the total estimated length into equal-length segments, and roughly estimate the area function of each. Note that this procedure does not reflect mathematical considerations typically applied by researchers in executing such conversions. 6 Rather, it was implemented as a minimalist interpretation of such a procedure, requiring minimal preparation, training, or mathematical background, while still producing appropriate matches to the intended vowel quality. The assignment was designed to simplistically emulate a common procedure in articulatory-acoustic phonetics, where in vocal tract models are explicitly modeled on real-life data ( 2 , 11 ). Exercise 2 In a second exercise, students explored the impact on predicted formants resulting from changing the dimensions of simplistic and arbitrary 2-tube and 4-tube sequences ( 1 ). Participants were asked to explore the effects of stricture on simplistic vocal tract models. From uniform starting positions, they were to use the “screw” function, 7 to explore the impact of changes on formants and vowel quality. This was to demonstrate that even rough imitations of realistic vocal tract configurations are sufficient to recreate relevant vowel qualities. As such, the goal of the exercise was to introduce participants to simplified versions of “distinctive regions” theories of speech acoustics – i.e., theories of speech production that assumes disproportionate divisions of articulatory–acoustic space. Such theories include the “Distinctive regions” theory proposed by Mrayati and colleagues ( 12 ), which presupposes eight such regions. In theory, the proposed workshop paradigm may be extended to such a narrow design; however, at present stage, given the relative lack of experience of the participants, as well as the historical historical success of two-, three- and four-tube models, we opted for this more simplistic set of tasks. Reflections and future iterations A future iteration of our “workshop” setup may seek to let participants design and 3D-print their own vocal tract models, essentially reverse engineering a vowel production phenomenon from the ground up. As the printing process is lengthy, limitations on time precluded the inclusion of such an element in our original workshop design. Nonetheless, with many universities providing so-called maker spaces, such an exercise could be designed as a take-home assignment. We have argued these are promising avenues for future work on phonetics teaching. Concluding thoughts Phonetics serves to bridge the sciences of acoustics and linguistics. However, students’ expectations about speech acoustics often conform to preconceptions acquired in linguistics coursework, and rarely from direct experience with the nuances of speech production, leading to particular acoustic consequences. One main reason for the lack of relevant knowledge in acoustics is related to the lack of accessibility of appropriate pedagogical tools. Here, we presented the TubeN GUI – software that is both publicly available, and which shows promise as a teaching tool of applied phonetics. With its introduction, we hope to encourage the use of speech acoustics tools in related sciences and educational programs. Acknowledgements The results of this work and the tools used will be made more widely accessible through the national infrastructure Språkbanken Tal under funding from the Swedish Research Council (2023-00161_VR). AE received additional support by the Swedish Research Council (2025–00209_VR) Funder Information Declared Swedish Research Council, https://ror.org/03zttf063 , 2023-00161_VR , 2025–00209_VR Footnotes https://github.com/jbeskow/tuben ↵ 1 For full modeling considerations, we refer to the original publications ( 2 , 3 ). ↵ 2 The original algorithm works by considering the tube behavior for one frequency at a time. However, in the current implementation, calculations are parallelized. ↵ 3 https://trimesh.org/ ↵ 4 https://github.com/jbeskow/tuben ↵ 5 The data was collected during a separate data collection session – a magnetic resonance imaging (MRI) single-subject case study at the Stockholm University Brain Imaging Centre, housed at Stockholm University, Stockholm, Sweden. The scanner was a Siemens Prisma 3 Tesla whole-body MRI. ↵ 6 The reader will find such conversions described elsewhere ( 10 ). ↵ 7 Note again that this function not represented by a clickable button, but is an implicit function of the GUI. Bibliography 1. ↵ J. M. Heinz and K. N. Stevens . On the derivation of area functions and acoustic spectra from cineradiographic films of speech . The Journal of the Acoustical Society of America , 36 : 1037 – 1038 , 1964 . doi: 10.1121/1.2143313 . OpenUrl CrossRef 2. ↵ G. Fant . The acoustic theory of speech production: With Calculations Based on X-Ray Studies of Russian Articulations . Mouton, The Hague , 1971 . 3. ↵ J. Liljencrants and G. Fant . Computer program for VT-resonance frequency calculations . STL-QPSR , 16 : 15 – 21 , 1975 . OpenUrl 4. R. Carré , P. Divenyi , and M. Mrayati . Speech: A dynamic process. De Gruyter , 2017 . doi: 10.1515/9781501502019 . OpenUrl CrossRef 5. B. H Story . History of speech synthesis . In The Routledge handbook of phonetics , pages 9 – 33 . Routledge , 2019 . 6. ↵ T. Arai . Education in basic acoustics for acoustic phonetics and speech science . The Journal of the Acoustical Society of America , 152 ( 5 ): 2746 – 2757 , 2022 . doi: 10.1121/10.0015050 . OpenUrl CrossRef PubMed 7. ↵ T. Chiba and M. Kajiyama . The Vowel: Its Nature and Structure . Tokyo-Kaiseikan , Tokyo , 1942 . 8. ↵ K. N. Stevens , S. Kasowski , and C. G. M. Fant . An electrical analog of the vocal tract . The Journal of the Acoustical Society of America , 25 ( 4 ): 734 – 742 , 1953 . OpenUrl CrossRef Web of Science 9. ↵ G. Fant . Vocal tract wall effects, losses, and resonance bandwidths . STL–QPSR , 2 : 28 – 52 , 1972 . OpenUrl 10. ↵ G. Fant . Vocal tract area functions of swedish vowels and a new three-parameter model . In Proceedings of the 2nd International Conference on Spoken Language Processing (ICSLP 1992) , pages 807 – 810 . ISCA , 1992 . doi: 10.21437/ICSLP.1992-262 . OpenUrl CrossRef 11. ↵ B. E. Lindblom and J. E. Sundberg . Acoustical consequences of lip, tongue, jaw, and larynx movement . The Journal of the Acoustical Society of America , 4B ( 50 ): 1166 – 1179 , 1971 . OpenUrl 12. ↵ M. Mrayati , R. Carré , and B. Guérin . Distinctive regions and modes: a new theory of speech production . Speech Communication , 7 : 257 – 286 , 1988 . OpenUrl View the discussion thread. Back to top Previous Next Posted October 16, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Open source software for tube vocal tract modeling, resonance prediction, illustration, and 3D printing Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Open source software for tube vocal tract modeling, resonance prediction, illustration, and 3D printing Runhui Song , Jonas Beskow , Jens Edlund , Mechtild Tronnier , Rui Tu , Kecheng Zhang , Axel Ekström bioRxiv 2025.10.15.682256; doi: https://doi.org/10.1101/2025.10.15.682256 Share This Article: Copy Citation Tools Open source software for tube vocal tract modeling, resonance prediction, illustration, and 3D printing Runhui Song , Jonas Beskow , Jens Edlund , Mechtild Tronnier , Rui Tu , Kecheng Zhang , Axel Ekström bioRxiv 2025.10.15.682256; doi: https://doi.org/10.1101/2025.10.15.682256 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Scientific Communication and Education Subject Areas All Articles Animal Behavior and Cognition (7619) Biochemistry (17642) Bioengineering (13865) Bioinformatics (41862) Biophysics (21409) Cancer Biology (18547) Cell Biology (25436) Clinical Trials (138) Developmental Biology (13358) Ecology (19863) Epidemiology (2067) Evolutionary Biology (24288) Genetics (15587) Genomics (22467) Immunology (17703) Microbiology (40301) Molecular Biology (17142) Neuroscience (88445) Paleontology (666) Pathology (2825) Pharmacology and Toxicology (4815) Physiology (7634) Plant Biology (15109) Scientific Communication and Education (2042) Synthetic Biology (4285) Systems Biology (9812) Zoology (2268)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.