<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v3.0 20080202//EN" "http://dtd.nlm.nih.gov/publishing/3.0/journalpublishing3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="3.0" xml:lang="en" article-type="research article">
 <front>
  <journal-meta>
   <journal-id journal-id-type="publisher-id">
    jmp
   </journal-id>
   <journal-title-group>
    <journal-title>
     Journal of Modern Physics
    </journal-title>
   </journal-title-group>
   <issn pub-type="epub">
    2153-1196
   </issn>
   <issn publication-format="print">
    2153-120X
   </issn>
   <publisher>
    <publisher-name>
     Scientific Research Publishing
    </publisher-name>
   </publisher>
  </journal-meta>
  <article-meta>
   <article-id pub-id-type="doi">
    10.4236/jmp.2024.1512087
   </article-id>
   <article-id pub-id-type="publisher-id">
    jmp-137287
   </article-id>
   <article-categories>
    <subj-group subj-group-type="heading">
     <subject>
      Articles
     </subject>
    </subj-group>
    <subj-group subj-group-type="Discipline-v2">
     <subject>
      Physics 
     </subject>
     <subject>
       Mathematics
     </subject>
    </subj-group>
   </article-categories>
   <title-group>
    Classifying Vibration Modes Generated by The Michelson Interferometer Using Machine Learning Methods
   </title-group>
   <contrib-group>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Xin-Han
      </surname>
      <given-names>
       Tsai
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff1"> 
      <sup>1</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Anthony An-Chih
      </surname>
      <given-names>
       Yeh
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff1"> 
      <sup>1</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Chen-Hsin
      </surname>
      <given-names>
       Lu
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff2"> 
      <sup>2</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Shang-Yu
      </surname>
      <given-names>
       Chou
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff3"> 
      <sup>3</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Shih-Wei
      </surname>
      <given-names>
       Wang
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff4"> 
      <sup>4</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Chi-Wei
      </surname>
      <given-names>
       Lee
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff5"> 
      <sup>5</sup>
     </xref>
    </contrib>
    <contrib contrib-type="author" xlink:type="simple">
     <name name-style="western">
      <surname>
       Po-Han
      </surname>
      <given-names>
       Lee
      </given-names>
     </name> 
     <xref ref-type="aff" rid="aff1"> 
      <sup>1</sup>
     </xref> 
     <xref ref-type="aff" rid="aff6"> 
      <sup>6</sup>
     </xref>
    </contrib>
   </contrib-group> 
   <aff id="aff1">
    <addr-line>
     aThe Affiliated Senior High School of National Taiwan Normal University, Taipei City
    </addr-line> 
   </aff> 
   <aff id="aff2">
    <addr-line>
     aDepartment of Materials Science and Engineering, National Tsing Hua University, Hsinchu City
    </addr-line> 
   </aff> 
   <aff id="aff3">
    <addr-line>
     aThe Department of Dentistry, National Defense Medical Center, Taipei City
    </addr-line> 
   </aff> 
   <aff id="aff4">
    <addr-line>
     aMechanical Science&amp;Engineering Department, University of Illinois at Urbana-Champaign, Urbana, US
    </addr-line> 
   </aff> 
   <aff id="aff5">
    <addr-line>
     aDepartment of Physics, National Tsing Hua University, Hsinchu City
    </addr-line> 
   </aff> 
   <aff id="aff6">
    <addr-line>
     aDepartment of Electro-Optical Engineering, National Taipei University of Technology, Taipei City
    </addr-line> 
   </aff> 
   <pub-date pub-type="epub">
    <day>
     01
    </day> 
    <month>
     11
    </month>
    <year>
     2024
    </year>
   </pub-date> 
   <volume>
    15
   </volume> 
   <issue>
    12
   </issue>
   <fpage>
    2169
   </fpage>
   <lpage>
    2192
   </lpage>
   <history>
    <date date-type="received">
     <day>
      25,
     </day>
     <month>
      September
     </month>
     <year>
      2024
     </year>
    </date>
    <date date-type="published">
     <day>
      8,
     </day>
     <month>
      September
     </month>
     <year>
      2024
     </year> 
    </date> 
    <date date-type="accepted">
     <day>
      8,
     </day>
     <month>
      November
     </month>
     <year>
      2024
     </year> 
    </date>
   </history>
   <permissions>
    <copyright-statement>
     © Copyright 2014 by authors and Scientific Research Publishing Inc. 
    </copyright-statement>
    <copyright-year>
     2014
    </copyright-year>
    <license>
     <license-p>
      This work is licensed under the Creative Commons Attribution International License (CC BY). http://creativecommons.org/licenses/by/4.0/
     </license-p>
    </license>
   </permissions>
   <abstract>
    In this paper, we explore the classification of vibration modes generated by handwriting on an optical desk using deep learning architectures. Three deep learning models—Long Short-Term Memory (LSTM) networks with attention mechanism, Video Vision Transformer (ViViT), and Long-term Recurrent Convolutional Network (LRCN)—were evaluated to determine the most effective method for analyzing time series patterns generated by a Michelson interferometer. The interferometer was used to detect vibration modes created by handwriting, capturing time-series data from the diffraction patterns. Among these models, the LSTM-Attention network achieved the highest validation accuracy, reaching up to 92%, outperforming both ViViT and LRCN. These findings highlight the potential of deep learning in material science for detecting and classifying vibration patterns. The powerful performance of the LSTM-Attention model suggests that it could be applied to similar classification tasks in related fields.
   </abstract>
   <kwd-group> 
    <kwd>
     Michelson Interferometer
    </kwd> 
    <kwd>
      Machine Learning
    </kwd> 
    <kwd>
      Vibration Modes
    </kwd> 
    <kwd>
      Long Short-Term Memory (LSTM)
    </kwd>
   </kwd-group>
  </article-meta>
 </front>
 <body>
  <sec id="s1">
   <title>1. Introduction</title>
   <p>The study of vibration modes generated by materials is essential for applications in defect detection, material science, and earth science. Various techniques, including optical methods like the Michelson interferometer, have been developed to detect these modes, which produce time-series patterns. However, predicting vibration modes resulting from specific behaviors, such as handwriting, continues to pose a challenging task. The Long Short-Term Memory with Attention mechanism (hereafter referred to as LSTM-Attention) enhances the traditional LSTM model by integrating an attention mechanism, which selectively focuses on and assigns weights to different elements within a sequence. This enhancement allows for better feature extraction and more accurate predictions, particularly in complex and extended sequential data.</p>
   <p>LSTM-Attention has proven effective in various applications, such as financial time-series prediction by Xuan Zhang et al. <xref ref-type="bibr" rid="scirp.137287-1">
     [1]
    </xref>, speech emotion recognition by Yeonguk Yu and Yoon-Joong Kim <xref ref-type="bibr" rid="scirp.137287-2">
     [2]
    </xref>, and crude oil price forecasting by Hu <xref ref-type="bibr" rid="scirp.137287-3">
     [3]
    </xref>. Additionally, incorporating the attention mechanism in LSTM models has improved text classification performance <xref ref-type="bibr" rid="scirp.137287-4">
     [4]
    </xref>, and Zhou et al. have demonstrated that this model can be applied to cross-lingual sentiment classification <xref ref-type="bibr" rid="scirp.137287-5">
     [5]
    </xref>.</p>
   <p>ViViT represents a recent advancement in computer vision, merging the transformer architecture with vision-specific inductive biases. This approach has set new benchmarks in image recognition tasks, positioning it as a promising tool for the classification of vibration patterns. Since the model’s release, many researchers have focused on improving its performance, leading to variants such as the Video Swin Transformer and MViTv2 <xref ref-type="bibr" rid="scirp.137287-6">
     [6]
    </xref> <xref ref-type="bibr" rid="scirp.137287-7">
     [7]
    </xref>. ViViT’s versatility is evident from its applications in areas such as video anomaly detection <xref ref-type="bibr" rid="scirp.137287-8">
     [8]
    </xref>, among others.</p>
   <p>The Long-term Recurrent Convolutional Network (LRCN) architecture integrates the feature extraction strengths of Convolutional Neural Networks (CNNs) with the sequential modeling abilities of Long Short-Term Memory (LSTM) networks. This hybrid approach has proven successful in various spatiotemporal tasks, and we explore its potential in classifying vibration modes in this study. LRCN has been applied to a range of challenges. For example, Wei et al. demonstrated its effectiveness in early prediction of epileptic seizures <xref ref-type="bibr" rid="scirp.137287-9">
     [9]
    </xref>, and it has been utilized for cyberbullying detection in social media comments <xref ref-type="bibr" rid="scirp.137287-10">
     [10]
    </xref>. Furthermore, LRCN has been applied to the complex task of handwritten Urdu text recognition, which presents challenges due to its intricacy <xref ref-type="bibr" rid="scirp.137287-11">
     [11]
    </xref>.</p>
   <p>The Michelson interferometer is a widely used optical instrument in physics and engineering, known for its precision in detecting minor variations in path lengths by splitting and recombining light beams. This sensitivity allows it to measure various physical phenomena, such as testing general relativity and detecting gravitational waves. In this study, we employ the Michelson interferometer as a detector to generate time series patterns for deep learning-based classification. The Michelson interferometer has proven its utility in applications such as displacement measurements with 10 picometer accuracy <xref ref-type="bibr" rid="scirp.137287-12">
     [12]
    </xref>, detection of third-generation gravitational waves <xref ref-type="bibr" rid="scirp.137287-13">
     [13]
    </xref>, and laser wavelength measurements, as demonstrated by Monchalin et al. <xref ref-type="bibr" rid="scirp.137287-14">
     [14]
    </xref>.</p>
   <p>Several studies have utilized Michelson interferometers for various purposes, including measuring small displacements, detecting gravitational waves, and material characterization. For instance, Park et al. demonstrated its application in vibration measurement <xref ref-type="bibr" rid="scirp.137287-15">
     [15]
    </xref>, while Cheng et al. proposed using a Michelson-Sagnac Interferometer to digitize vibration signals, which were then fed into a VGG16 algorithm. Their model achieved an impressive accuracy of 98.44% in a six-class classification task <xref ref-type="bibr" rid="scirp.137287-16">
     [16]
    </xref>.</p>
   <p>In addition to the Michelson interferometer, accelerometers are also commonly used to collect vibration data. For example, Medina et al. used LSTM to classify signals generated by an accelerometer during gearset operations to detect faults, achieving an accuracy of up to 99.4% across 10 classes <xref ref-type="bibr" rid="scirp.137287-17">
     [17]
    </xref>. In recent years, machine learning techniques have been increasingly applied to various fields, including physics and engineering. Deep learning methods such as Convolutional Neural Networks (CNNs) and Recurrent Neural Networks (RNNs) have proven effective in classifying patterns and time series data. For instance, Abdelmaksoud et al. used a CNN to classify diverse types of fault signals in induction motors <xref ref-type="bibr" rid="scirp.137287-18">
     [18]
    </xref>, and Yang et al. utilized an RNN to predict the remaining useful life of bearings <xref ref-type="bibr" rid="scirp.137287-19">
     [19]
    </xref>.</p>
   <p>To the best of our knowledge, no prior research has employed machine learning to classify patterns produced by vibration modes generated by a Michelson interferometer. In this paper, we propose a novel method utilizing neural network architectures to classify vibration patterns produced by handwritten behavior on an optical desk using a Michelson interferometer.</p>
  </sec><sec id="s2">
   <title>2. Materials and Methods</title>
   <p>We propose using three deep learning architectures—LSTM-Attention model, ViViT, and LRCN—to classify vibration modes generated by handwritten behavior on an optical desk. A Michelson interferometer generates a time-series pattern, which is fed into these neural networks, which are all built using TensorFlow and Keras. After training and evaluation, we compare their performances to determine the most effective model. This process is depicted by the Michelson interferometer setup shown in <xref ref-type="fig" rid="fig1">
     Figure 1
    </xref>. This study aims to identify the most accurate model for</p>
   <fig id="fig1" position="float">
    <label>Figure 1</label>
    <caption>
     <title>Figure 1. The setup for Michelson Interferometer shows the arrangement of components where the interference pattern (green circle) is observed.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId14.jpeg?20241111023811" />
   </fig>
   <p>this specific task, contributing to fields such as defect detection and material science by providing advanced neural network solutions for classifying vibration patterns.</p>
   <p>The computing resources were provided by Taiwan Computing Cloud (TWCC), an AI model training platform developed by the National Center for High-Performance Computing (NCHC). We used TWCC’s container computing resources for model training and design, leveraging Nvidia V100 32GB GPUs and adjusting the number of GPU nodes as required.</p>
   <sec id="s2_1">
    <title>2.1. Michelson Interferometer Principles</title>
    <p>The input beam is split into two by the beam splitter: 50% of the light is reflected toward mirror M<sub>1</sub>, while the remaining 50% is transmitted toward mirror M<sub>2</sub>. After being reflected from M<sub>1</sub> and M<sub>2</sub>, half of the light from each mirror is redirected by the beam splitter toward the viewing screen as shown in <xref ref-type="fig" rid="fig1">
      Figure 1
     </xref> and <xref ref-type="fig" rid="fig2">
      Figure 2
     </xref>.</p>
    <fig id="fig2" position="float">
     <label>Figure 2</label>
     <caption>
      <title>Figure 2. A diagram of a Michelson interferometer showing the light source, beam splitter, mirrors (M<sub>1</sub> and M<sub>2</sub>), and the viewing screen. The beam splitter divides the light beam into two paths, which reflect off the mirrors and recombine to form an interference pattern on the viewing screen.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId15.jpeg?20241111023812" />
    </fig>
    <p>Thus, the original light beam splits, and the resulting beams are recombined. Since the beams originate from the same source, their phases are highly correlated. When a focusing lens is placed between the laser source and the beam-splitter, the light ray spreads out after the focal point, producing an interference pattern of dark and bright rings on the viewing screen. For theoretical analysis, we first define the electromagnetic field of the input laser as E, with its strength denoted as E<sub>0</sub>.</p>
    <p>After passing through the beam splitter, the incident electromagnetic field E<sub>0</sub> is divided into two components: E<sub>1</sub> = rE<sub>0</sub>, the reflected part, and E<sub>2</sub> = tE<sub>0</sub>, the transmitted part, where r and t are the reflection and transmission coefficients, respectively. The output field E<sub>out</sub> is then the superposition of these two components as they recombine after reflecting from the mirrors. The resulting interference pattern depends on the phase difference between E<sub>1</sub> and E<sub>2</sub>, which leads to constructive or destructive interference on the viewing screen.</p>
    <p>Mathematically, this can be expressed as:</p>
    <p>
     <math display="inline" xmlns="http://www.w3.org/1998/Math/MathML"> <mrow> 
       <msub> 
        <mi>
          E 
        </mi> 
        <mrow> 
         <mtext>
           out 
         </mtext> 
        </mrow> 
       </msub> 
       <mo>
         = 
       </mo> 
       <mi>
         r 
       </mi> 
       <msub> 
        <mi>
          E 
        </mi> 
        <mn>
          1 
        </mn> 
       </msub> 
       <mo>
         + 
       </mo> 
       <mi>
         t 
       </mi> 
       <msub> 
        <mi>
          E 
        </mi> 
        <mn>
          2 
        </mn> 
       </msub> 
      </mrow> 
     </math> (1)</p>
    <p>
     <math display="inline" xmlns="http://www.w3.org/1998/Math/MathML"> <mrow> 
       <msup> 
        <mi>
          r 
        </mi> 
        <mn>
          2 
        </mn> 
       </msup> 
       <mo>
         + 
       </mo> 
       <msup> 
        <mi>
          t 
        </mi> 
        <mn>
          2 
        </mn> 
       </msup> 
       <mo>
         = 
       </mo> 
       <mn>
         1 
       </mn> 
      </mrow> 
     </math></p>
    <p>If we introduce the phase difference 
     <math display="inline" xmlns="http://www.w3.org/1998/Math/MathML"> <mrow> 
       <mi>
         Δ 
       </mi> 
       <mi>
         φ 
       </mi> 
       <mo>
         = 
       </mo> 
       <msub> 
        <mi>
          φ 
        </mi> 
        <mn>
          2 
        </mn> 
       </msub> 
       <mo>
         − 
       </mo> 
       <msub> 
        <mi>
          φ 
        </mi> 
        <mn>
          1 
        </mn> 
       </msub> 
      </mrow> 
     </math> between the two beams due to vibrations or other disturbances, the intensity I<sub>out</sub> of the output will be affected by the interference of the two beams.</p>
    <p>The general equation for the intensity of the output is:</p>
    <p>
     <math display="inline" xmlns="http://www.w3.org/1998/Math/MathML"> <mrow> 
       <msub> 
        <mi>
          I 
        </mi> 
        <mrow> 
         <mtext>
           out 
         </mtext> 
        </mrow> 
       </msub> 
       <mo>
         = 
       </mo> 
       <msup> 
        <mrow> 
         <mrow> 
          <mo>
            | 
          </mo> 
          <mrow> 
           <msub> 
            <mi>
              E 
            </mi> 
            <mrow> 
             <mtext>
               out 
             </mtext> 
            </mrow> 
           </msub> 
          </mrow> 
          <mo>
            | 
          </mo> 
         </mrow> 
        </mrow> 
        <mn>
          2 
        </mn> 
       </msup> 
      </mrow> 
     </math> (2)</p>
    <p>Since the electric field components are now influenced by the phase difference Δφ, the expression for the combined electric field becomes:</p>
    <p>
     <math display="inline" xmlns="http://www.w3.org/1998/Math/MathML"> <mrow> 
       <msub> 
        <mi>
          E 
        </mi> 
        <mrow> 
         <mtext>
           out 
         </mtext> 
        </mrow> 
       </msub> 
       <mo>
         = 
       </mo> 
       <msub> 
        <mi>
          E 
        </mi> 
        <mtext>
          1 
        </mtext> 
       </msub> 
       <mo>
         + 
       </mo> 
       <msub> 
        <mi>
          E 
        </mi> 
        <mtext>
          2 
        </mtext> 
       </msub> 
       <msup> 
        <mtext>
          e 
        </mtext> 
        <mrow> 
         <mi>
           i 
         </mi> 
         <mi>
           Δ 
         </mi> 
         <mi>
           ϕ 
         </mi> 
        </mrow> 
       </msup> 
      </mrow> 
     </math> (3)</p>
    <p>Thus, the intensity becomes:</p>
    <p>
     <math display="inline" xmlns="http://www.w3.org/1998/Math/MathML"> <mrow> 
       <msub> 
        <mi>
          I 
        </mi> 
        <mrow> 
         <mtext>
           out 
         </mtext> 
        </mrow> 
       </msub> 
       <mo>
         = 
       </mo> 
       <msup> 
        <mrow> 
         <mrow> 
          <mo>
            | 
          </mo> 
          <mrow> 
           <msub> 
            <mi>
              E 
            </mi> 
            <mtext>
              1 
            </mtext> 
           </msub> 
           <mo>
             + 
           </mo> 
           <msub> 
            <mi>
              E 
            </mi> 
            <mtext>
              2 
            </mtext> 
           </msub> 
           <msup> 
            <mtext>
              e 
            </mtext> 
            <mrow> 
             <mi>
               i 
             </mi> 
             <mi>
               Δ 
             </mi> 
             <mi>
               ϕ 
             </mi> 
            </mrow> 
           </msup> 
          </mrow> 
          <mo>
            | 
          </mo> 
         </mrow> 
        </mrow> 
        <mn>
          2 
        </mn> 
       </msup> 
      </mrow> 
     </math> (4)</p>
    <p>Expanding this:</p>
    <p>
     <math display="inline" xmlns="http://www.w3.org/1998/Math/MathML"> <mrow> 
       <msub> 
        <mi>
          I 
        </mi> 
        <mrow> 
         <mtext>
           out 
         </mtext> 
        </mrow> 
       </msub> 
       <mo>
         = 
       </mo> 
       <msup> 
        <mrow> 
         <mrow> 
          <mo>
            | 
          </mo> 
          <mrow> 
           <msub> 
            <mi>
              E 
            </mi> 
            <mtext>
              1 
            </mtext> 
           </msub> 
          </mrow> 
          <mo>
            | 
          </mo> 
         </mrow> 
        </mrow> 
        <mn>
          2 
        </mn> 
       </msup> 
       <mo>
         + 
       </mo> 
       <msup> 
        <mrow> 
         <mrow> 
          <mo>
            | 
          </mo> 
          <mrow> 
           <msub> 
            <mi>
              E 
            </mi> 
            <mtext>
              2 
            </mtext> 
           </msub> 
          </mrow> 
          <mo>
            | 
          </mo> 
         </mrow> 
        </mrow> 
        <mn>
          2 
        </mn> 
       </msup> 
       <mo>
         + 
       </mo> 
       <mn>
         2 
       </mn> 
       <mrow> 
        <mo>
          | 
        </mo> 
        <mrow> 
         <msub> 
          <mi>
            E 
          </mi> 
          <mtext>
            1 
          </mtext> 
         </msub> 
        </mrow> 
        <mo>
          | 
        </mo> 
       </mrow> 
       <mrow> 
        <mo>
          | 
        </mo> 
        <mrow> 
         <msub> 
          <mi>
            E 
          </mi> 
          <mtext>
            2 
          </mtext> 
         </msub> 
        </mrow> 
        <mo>
          | 
        </mo> 
       </mrow> 
       <mi>
         cos 
       </mi> 
       <mrow> 
        <mo>
          ( 
        </mo> 
        <mrow> 
         <mi>
           Δ 
         </mi> 
         <mi>
           ϕ 
         </mi> 
        </mrow> 
        <mo>
          ) 
        </mo> 
       </mrow> 
      </mrow> 
     </math> (5)</p>
    <p>Assuming E<sub>1</sub> = rE<sub>0</sub> and E<sub>2</sub> = tE<sub>0</sub>, where r and t are the reflection and transmission coefficients mentioned before, the intensity simplifies to:</p>
    <p>
     <math display="inline" xmlns="http://www.w3.org/1998/Math/MathML"> <mrow> 
       <msub> 
        <mi>
          I 
        </mi> 
        <mrow> 
         <mtext>
           out 
         </mtext> 
        </mrow> 
       </msub> 
       <mo>
         = 
       </mo> 
       <msup> 
        <mrow> 
         <mrow> 
          <mo>
            | 
          </mo> 
          <mrow> 
           <msub> 
            <mi>
              E 
            </mi> 
            <mtext>
              0 
            </mtext> 
           </msub> 
          </mrow> 
          <mo>
            | 
          </mo> 
         </mrow> 
        </mrow> 
        <mn>
          2 
        </mn> 
       </msup> 
       <mrow> 
        <mo>
          ( 
        </mo> 
        <mrow> 
         <msup> 
          <mi>
            r 
          </mi> 
          <mn>
            2 
          </mn> 
         </msup> 
         <mo>
           + 
         </mo> 
         <msup> 
          <mi>
            t 
          </mi> 
          <mn>
            2 
          </mn> 
         </msup> 
         <mo>
           + 
         </mo> 
         <mn>
           2 
         </mn> 
         <mi>
           r 
         </mi> 
         <mi>
           t 
         </mi> 
         <mi>
           cos 
         </mi> 
         <mrow> 
          <mo>
            ( 
          </mo> 
          <mrow> 
           <mi>
             Δ 
           </mi> 
           <mi>
             ϕ 
           </mi> 
          </mrow> 
          <mo>
            ) 
          </mo> 
         </mrow> 
        </mrow> 
        <mo>
          ) 
        </mo> 
       </mrow> 
      </mrow> 
     </math> (6)</p>
    <p>This equation shows how the output intensity depends on the phase difference Δφ, which is affected by any vibrations introduced into the system. The constructive or destructive interference leads to variations in the intensity depending on Δφ.</p>
    <p>If a small signal δ(t), this experimental test, is input by handwriting on the optical desk, the final output signal I<sub>out_signal</sub> will be:</p>
    <p>
     <math display="inline" xmlns="http://www.w3.org/1998/Math/MathML"> <mrow> 
       <msub> 
        <mi>
          I 
        </mi> 
        <mrow> 
         <mtext>
           out_signal 
         </mtext> 
        </mrow> 
       </msub> 
       <mo>
         = 
       </mo> 
       <msub> 
        <mi>
          I 
        </mi> 
        <mtext>
          0 
        </mtext> 
       </msub> 
       <msup> 
        <mrow> 
         <mi>
           cos 
         </mi> 
        </mrow> 
        <mn>
          2 
        </mn> 
       </msup> 
       <mrow> 
        <mo>
          ( 
        </mo> 
        <mrow> 
         <mfrac> 
          <mrow> 
           <mi>
             Δ 
           </mi> 
           <msub> 
            <mi>
              ϕ 
            </mi> 
            <mn>
              0 
            </mn> 
           </msub> 
           <mo>
             + 
           </mo> 
           <mi>
             Δ 
           </mi> 
           <mrow> 
            <mo>
              ( 
            </mo> 
            <mi>
              t 
            </mi> 
            <mo>
              ) 
            </mo> 
           </mrow> 
          </mrow> 
          <mn>
            2 
          </mn> 
         </mfrac> 
        </mrow> 
        <mo>
          ) 
        </mo> 
       </mrow> 
      </mrow> 
     </math> (7)</p>
    <p>where Δ(t) represents the phase vibration generated by the small signal δ(t), and Δϕ<sub>0</sub> is the initial phase difference.</p>
   </sec>
   <sec id="s2_2">
    <title>2.2. Data Collection</title>
    <p>Vibration mode data is collected using a camera recording at 1080p resolution and 60 frames per second. A Michelson interferometer with a 532 nm laser beam is aligned parallel to the optical desk. The laser beam is expanded to form a larger parallel beam, and a plane mirror is placed perpendicular to the beam. A 50/50 beam splitter cube is positioned between the plane mirror and the light source. The interference pattern is generated by combining the two light beams reflected off the plane mirror as they pass through the beam splitter cube, the complete setup is shown in <xref ref-type="table" rid="table1">
      Table 1
     </xref>.</p>
    <table-wrap id="table1">
     <label>
      <xref ref-type="table" rid="table1">
       Table 1
      </xref></label>
     <caption>
      <title>
       <xref ref-type="bibr" rid="scirp.137287-"></xref>Table 1. The parts of our Michelson Interferometer setup.</title>
     </caption>
     <table class="MsoTableGrid custom-table" border="0" cellspacing="0" cellpadding="0"> 
      <tr> 
       <td class="custom-bottom-td acenter" width="37.64%"><p style="text-align:center">equipment name</p></td> 
       <td class="custom-bottom-td acenter" width="62.36%"><p style="text-align:center">spec</p></td> 
      </tr> 
      <tr> 
       <td class="custom-top-td acenter" width="37.64%"><p style="text-align:center">Optical Breadboards</p></td> 
       <td class="custom-top-td acenter" width="62.36%"><p style="text-align:center">aluminum alloy, 700 × 400 × 10 mm, M6-25</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="37.64%"><p style="text-align:center">solid state green laser</p></td> 
       <td class="acenter" width="62.36%"><p style="text-align:center">Wavelength: 532nm, Power 10 mW, CW Power Stability: &lt;3% @ 25˚C</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="37.64%"><p style="text-align:center">laser adjuster</p></td> 
       <td class="acenter" width="62.36%"><p style="text-align:center">2-axis adjustment</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="37.64%"><p style="text-align:center">mirror adjuster</p></td> 
       <td class="acenter" width="62.36%"><p style="text-align:center">ψ1", 2-axis adjustment</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="37.64%"><p style="text-align:center">beam splitter cube adjuster</p></td> 
       <td class="acenter" width="62.36%"><p style="text-align:center">2-axis adjustment</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="37.64%"><p style="text-align:center">Aluminized reflector</p></td> 
       <td class="acenter" width="62.36%"><p style="text-align:center">O1" × 3 mm Thick, R &gt; 95% for 400 - 700 nm</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="37.64%"><p style="text-align:center">plano-convex lens</p></td> 
       <td class="acenter" width="62.36%"><p style="text-align:center">ψ2", EFL = 150 mm</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="37.64%"><p style="text-align:center">plano-convex lens</p></td> 
       <td class="acenter" width="62.36%"><p style="text-align:center">ψ1", EFL = 50 mm</p></td> 
      </tr> 
      <tr> 
       <td class="acenter" width="37.64%"><p style="text-align:center">Nonpolarized beam splitter</p></td> 
       <td class="acenter" width="62.36%"><p style="text-align:center">50/50, 1" cube</p></td> 
      </tr> 
     </table>
    </table-wrap>
    <p>To generate vibration modes, lowercase English letters (a-z) were handwritten on the optical desk, as shown in <xref ref-type="fig" rid="fig3">
      Figure 3
     </xref>. Each writing instance was recorded as a video, with each video serving as a data sample for training and testing. Additionally, during intervals between handwritten characters, a pattern with no writing behavior was recorded and labeled as “other” or “_”. This occurs during the idle time between each letter. Fifty samples were captured for each letter, all written by the same author.</p>
    <p>This results in a total of 50 × 26 = 1300 samples for the letters “a-z” and an additional 1300 samples for the “_” label. To prevent overfitting, we selected 50 “other” samples from the 1300 available for training and evaluation. The data was shuffled and split, with 20% reserved for testing, 16% for validation, and the remaining 64% for training.</p>
    <fig id="fig3" position="float">
     <label>Figure 3</label>
     <caption>
      <title>Figure 3. The illustration of behavior of hand-written character “a” on the optical desk.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId34.jpeg?20241111023814" />
    </fig>
   </sec>
   <sec id="s2_3">
    <title>2.3. Deep Learning Models</title>
    <p>In this experiment, the vibration signals are a type of time series data, where the focus is on identifying patterns over time and monitoring continuous steps throughout the entire sequence. Therefore, certain well-known machine learning models that primarily focus on image classification and recognition, such as Support Vector Machines (SVM) and K-Nearest Neighbors (KNN), as well as deep learning models like Autoencoder + CNN and Autoencoder + Transformer, are not considered in this paper. Instead, we will focus on models such as LSTM (Long Short-Term Memory) with Attention, ViViT (Video Vision Transformer), and LRCN (Long-term Recurrent Convolutional Network), which will be described in detail in the following sections.</p>
    <p>The LRCN (Long-term Recurrent Convolutional Networks), as proposed by Donahue et al. <xref ref-type="bibr" rid="scirp.137287-20">
      [20]
     </xref>, is a model designed for processing 2D video data by combining the strengths of convolutional neural networks (CNNs) and recurrent neural networks (RNNs). It efficiently processes interferogram frames, with the 2D CNN serving as a feature extractor that captures spatial patterns and relevant information. This network employs convolutional layers with batch normalization to stabilize training, Leaky ReLU activation to enhance feature learning, and max-pooling to downsample the feature maps. Additionally, dropout is applied during training to prevent overfitting.</p>
    <p>The spatial features extracted by the CNN are then passed through a flattening layer, converting the 2D representation into a 1D format suitable for the LSTM layer, which manages the temporal patterns in the data. Additionally, batch normalization is applied to assist with optimization, further enhancing the model’s structure and performance, as described in <xref ref-type="bibr" rid="scirp.137287-21">
      [21]
     </xref>.</p>
    <p>The unique strength of the LRCN model lies in its ability to capture both spatial and temporal dependencies in the interference patterns. This hybrid approach, which combines CNNs for spatial feature extraction and LSTMs for sequential analysis, makes it particularly well-suited for analyzing video data. The model excels at classifying vibration modes by effectively capturing both spatial and temporal patterns, resulting in competitive performance on the interferogram classification task.</p>
    <p>The ViViT model, developed by Arnab et al. <xref ref-type="bibr" rid="scirp.137287-22">
      [22]
     </xref>, is a state-of-the-art architecture designed for sequence classification tasks using the Transformer’s self-attention mechanism. In this study, the ViViT model is applied to 2D video data from interferograms. It leverages multiple attention heads and layers, allowing the model to capture long-range dependencies and interactions within the data. By using self-attention, the ViViT model dynamically focuses on different regions of the interferogram frames, emphasizing the most salient features that are crucial for accurate classification.</p>
    <p>The key strength of the ViViT model lies in its ability to capture complex spatial relationships across entire interferogram sequences. By incorporating attention mechanisms, the model outperforms traditional CNN-based methods in capturing global contextual information, which is essential for accurately classifying vibration modes with intricate spatial patterns. Additionally, the inclusion of label smoothing during training enhances model generalization, resulting in improved accuracy and robustness in the interferogram classification task.</p>
    <p>On the other hand, the LRCN model, designed for 1D video data, takes a different approach to handling the concentric circular nature of the diffraction pattern. Instead of processing the entire pattern, the model extracts a single diameter from the interferogram. This simplification preserves essential spatial information while reducing the input data’s complexity. The 1D CNN then captures spatial features through layers of 1D convolutions, max-pooling, and dropout, effectively analyzing the one-dimensional time-series patterns of the interferogram.</p>
    <p>The main advantage of the LRCN model is its efficiency in handling 1D video data, providing a compact representation while retaining key spatial information. Although it achieves moderate accuracy on the interferogram classification task, the model may face challenges when dealing with more complex spatial patterns due to the simplified input representation.</p>
    <p>The LSTM-Attention model <xref ref-type="bibr" rid="scirp.137287-23">
      [23]
     </xref>, on the other hand, is designed to capture temporal dependencies and subtle variations within the 1D time series patterns of the interferogram. Equipped with 128 LSTM units, this model excels at learning long-term dependencies, which is crucial for accurately identifying vibration modes generated by handwritten behavior. The attention mechanism further enhances the model by dynamically focusing on relevant parts of the input sequence, improving its ability to recognize intricate temporal patterns.</p>
    <p>Self-attention allows the LSTM-Attention model to dynamically focus on specific parts of the time series patterns, enhancing classification accuracy. This attention mechanism enables the model to adaptively assign weights to different time steps, enhancing its ability to capture and prioritize relevant features critical for accurate classification. Among the models tested, the LSTM-Attention model demonstrates superior performance in classifying vibration modes, achieving higher accuracy. The interpretability analysis of the model further reveals insights into its decision-making process, providing a clearer understanding of how it identifies crucial temporal features for precise classification.</p>
   </sec>
   <sec id="s2_4">
    <title>2.4. Utilization of the Models</title>
    <p>Each of the three models—LSTM-Attention for 1D, LRCN for 1D/2D, and ViViT for 2D—was trained on preprocessed data using TensorFlow and Keras. During training, categorical crossentropy was used as the loss function, while Adam and RMSProp served as optimization algorithms.</p>
    <p>After training, the models were evaluated on a separate testing set. Metrics such as accuracy were calculated to compare the models’ effectiveness. A confusion matrix was also generated to visually assess the classification performance of each model.</p>
    <p>Comprehensive analysis of the models’ results revealed each model’s strengths and weaknesses in classifying the vibration modes generated by handwritten behavior. Additionally, hyperparameter tuning and optimization were applied to improve classification accuracy on the interferogram data.</p>
    <p>The findings from this study help identify the most suitable deep learning architecture for classifying vibration modes generated by handwritten movements on an optical desk. The research also contributes valuable insights into applying deep learning for material science and vibration pattern recognition.</p>
   </sec>
   <sec id="s2_5">
    <title>2.5. Data Preprocessing</title>
    <p>We begin with a 1920 × 1080 RGB video recorded at 60 fps. The video is then converted to grayscale. After that, we crop the video to focus on the concentric circle pattern produced by the interferometer. The cropped video is resized to 175 × 175 pixels and normalize the pixel values by dividing each pixel by 255. This normalization ensures that the data values remain fall the range of 0 to 1, represented as float32. An example of this preprocessing steps is shown in <xref ref-type="fig" rid="fig4">
      Figure 4
     </xref>.</p>
    <fig id="fig4" position="float">
     <label>Figure 4</label>
     <caption>
      <title>Figure 4. A brief example of the process.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId35.jpeg?20241111023818" />
    </fig>
    <p>The above-mentioned preprocessing steps were used while training the LRCN 2D model and ViViT. However, since the pattern consists of concentric circles, we realized we could simplify the input data by extracting a single line of pixels from each video frame, passing through the center of the pattern. This extracted line effectively represents the pattern’s information. A period of one data sample is shown in <xref ref-type="fig" rid="fig5">
      Figure 5
     </xref>, where each line corresponds to a single frame of the video.</p>
    <p>We can summarize our preprocessing steps as follows, and the workflow is illustrated in <xref ref-type="fig" rid="fig6">
      Figure 6
     </xref>.</p>
    <p>1) Convert 1920 × 1080 @ 60 fps RGB video data to 1920 × 1080 @ 60 fps grayscale.</p>
    <p>2) Crop the video to match the concentric circle (700 × 700).</p>
    <p>3) Resize it to 175 × 175 using the cv2 default resize algorithm (inter-linear interpolation).</p>
    <p>4) Normalize the pixel values to be between 0 and 1 by dividing by 255.</p>
    <p>5) Extract the horizontal centerline intensity of the image if using LRCN 1D or LSTM-Attention.</p>
    <fig id="fig5" position="float">
     <label>Figure 5</label>
     <caption>
      <title>Figure 5. First 40 frames light intensity of an example data.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId36.jpeg?20241111023818" />
    </fig>
    <fig id="fig6" position="float">
     <label>Figure 6</label>
     <caption>
      <title>Figure 6. Preprocessing workflow.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId37.jpeg?20241111023818" />
    </fig>
    <p>The data labeling process utilizes one-hot encoding for 27 classes, which includes 26 English lowercase letters and one additional class for patterns generated during inactivity on the optical desk. Each class contains 50 data points. To prevent overfitting, label smoothing <xref ref-type="bibr" rid="scirp.137287-24">
      [24]
     </xref> with a factor of 0.1 was applied.</p>
   </sec>
   <sec id="s2_6">
    <title>2.6. Model Training</title>
    <p>Initially, we trained our model using 2D video data. Since the LRCN model is known for its ability to handle this type of data, we applied it to our dataset. The architecture of the model is illustrated in <xref ref-type="fig" rid="fig7">
      Figure 7
     </xref>, and the detailed workflow is depicted in <xref ref-type="fig" rid="fig8">
      Figure 8
     </xref>.</p>
    <fig id="fig7" position="float">
     <label>Figure 7</label>
     <caption>
      <title>Figure 7. LRCN model architecture.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId38.jpeg?20241111023819" />
    </fig>
    <p>After applying the LRCN model, we also implemented the ViViT model for comparison. Given ViViT’s strength in handling video data through its transformer-based architecture, we explored its performance on our dataset. This allowed us to evaluate the differences in classification accuracy and model efficiency between the two approaches.</p>
    <p>There are four variations of the ViViT model proposed by Arnab et al. <xref ref-type="bibr" rid="scirp.137287-22">
      [22]
     </xref>. For simplicity, we focus on evaluating the Spatio-temporal attention variation. The ViViT workflow of our study is shown in <xref ref-type="fig" rid="fig9">
      Figure 9
     </xref>. The first step involves performing Tubelet Embedding using a conv3D layer. Since conv3D requires a fixed input length, we limit the data length to 175 frames. If the data contains fewer than 175 frames, it is zero-padded, while longer data is cropped.</p>
    <p>To improve model performance through data simplification, we applied a feature extraction method that leverages the concentric nature of the diffraction pattern. By extracting only one diameter of the pattern, we preserve the necessary information while reducing complexity.</p>
    <p>After completing the feature extraction, we applied the LRCN 1D model, which is simpler than the LRCN 2D model previously used. The architecture of the LRCN 1D model is shown in <xref ref-type="fig" rid="fig10">
      Figure 10
     </xref>.</p>
    <p>The LRCN 1D model shares the same architecture as the LRCN 2D model, with</p>
    <fig id="fig8" position="float">
     <label>Figure 8</label>
     <caption>
      <title>Figure 8. LRCN workflow.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId39.jpeg?20241111023819" />
    </fig>
    <fig id="fig9" position="float">
     <label>Figure 9</label>
     <caption>
      <title>Figure 9. ViViT model workflow.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId40.jpeg?20241111023819" />
    </fig>
    <fig id="fig10" position="float">
     <label>Figure 10</label>
     <caption>
      <title>Figure 10. LRCN 1D model architecture.</title>
     </caption>
     <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId41.jpeg?20241111023819" />
    </fig>
    <p>the key difference being the use of 1D CNN layers for feature extraction instead of 2D CNN layers. In addition, we also explored LSTM-Attention models, which were adapted from the framework proposed by Wang et al. <xref ref-type="bibr" rid="scirp.137287-23">
      [23]
     </xref>. The architecture of the LSTM-Attention model is depicted in <xref ref-type="fig" rid="fig11">
      Figure 11
     </xref>, with the detailed workflow outlined in <xref ref-type="fig" rid="fig12">
      Figure 12
     </xref>. This model utilizes global average pooling, originally proposed by Lin et al. <xref ref-type="bibr" rid="scirp.137287-25">
      [25]
     </xref>, which functions similarly to a fully connected layer but with fewer parameters, thereby reducing the likelihood of overfitting. Additional research related to the comparative analysis of speaker identification performance using deep learning can be referenced <xref ref-type="bibr" rid="scirp.137287-26">
      [26]
     </xref>, which provides a strong foundation for the use of multiple classifiers in complex classification tasks.</p>
   </sec>
  </sec><sec id="s3">
   <title>3. Results and Discussion</title>
   <p>Based on our test results, the LRCN model demonstrates optimal performance when configured with four layers of convolutional blocks (Conv -&gt; Pooling -&gt; Batch Norm) and L1 regularization, with the regularization parameter λ set to</p>
   <fig id="fig11" position="float">
    <label>Figure 11</label>
    <caption>
     <title>Figure 11. LSTM-Attention model architecture.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId42.jpeg?20241111023820" />
   </fig>
   <p>0.01. We also used categorical cross-entropy as the loss function. <xref ref-type="fig" rid="fig13">
     Figure 13
    </xref> and <xref ref-type="fig" rid="fig14">
     Figure 14
    </xref> illustrate the training and validation loss, as well as the accuracy, respectively. These figures provide a clear depiction of the model’s performance throughout the training process, reflecting its convergence and ability to generalize to the validation data.</p>
   <p>As we can see, the model’s validation accuracy is only about 76%, indicating that the model is insufficient for achieving higher precision. The confusion matrix, shown in <xref ref-type="fig" rid="fig15">
     Figure 15
    </xref>, further highlights the areas where the model struggles, particularly in distinguishing between certain classes.</p>
   <p>The confusion matrix reveals that the LRCN model struggles not only with noise but also with distinguishing between similar letters, such as “O” and “N”.</p>
   <fig id="fig12" position="float">
    <label>Figure 12</label>
    <caption>
     <title>Figure 12. LSTM-Attention workflow.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId43.jpeg?20241111023820" />
   </fig>
   <fig id="fig13" position="float">
    <label>Figure 13</label>
    <caption>
     <title>Figure 13. The accuracy graph over epochs by LRCN 2D model training.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId44.jpeg?20241111023820" />
   </fig>
   <fig id="fig14" position="float">
    <label>Figure 14</label>
    <caption>
     <title>Figure 14. The loss graph over epochs by LRCN 2D model.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId45.jpeg?20241111023820" />
   </fig>
   <fig id="fig15" position="float">
    <label>Figure 15</label>
    <caption>
     <title>Figure 15. Confusion matrix of LRCN 2D from label a-z and “_”.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId46.jpeg?20241111023820" />
   </fig>
   <p>Additionally, the label “_” remains unidentifiable, possibly due to LSTM’s limitations in distinguishing subtle differences in patterns. To improve this, we experimented with the ViViT model, which utilizes the Transformer architecture for sequence classification. Based on our tests, the most optimized configuration for ViViT includes 8 layers of attention (L = 8, as shown in <xref ref-type="fig" rid="fig9">
     Figure 9
    </xref>), with each layer using 8 heads. The patch size is set to (16, 16, 16), and the projection dimension is configured to 27. The training/validation accuracy and loss are depicted in <xref ref-type="fig" rid="fig16">
     Figure 16
    </xref> and <xref ref-type="fig" rid="fig17">
     Figure 17
    </xref>.</p>
   <p>As observed, the model achieved a validation accuracy of approximately 77%, but signs of overfitting became evident as the validation loss began increasing around epoch 100. To mitigate this issue, we applied label smoothing, following the recommendation of Arnab et al. <xref ref-type="bibr" rid="scirp.137287-22">
     [22]
    </xref>, with a label smoothing factor set to 0.1. The updated training/validation accuracy and validation loss are displayed in <xref ref-type="fig" rid="fig18">
     Figure 18
    </xref> and <xref ref-type="fig" rid="fig19">
     Figure 19
    </xref>.</p>
   <p>The best result with ViViT reached a validation accuracy of 86%, which represents a substantial improvement. The loss consistently decreased while the accuracy continued to rise. The confusion matrix, which further illustrates the model’s performance, is displayed in <xref ref-type="fig" rid="fig20">
     Figure 20
    </xref>.</p>
   <p>The confusion matrix reveals that while the ViViT model outperforms LRCN</p>
   <fig id="fig16" position="float">
    <label>Figure 16</label>
    <caption>
     <title>Figure 16. The accuracy graph over epochs by ViViT.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId47.jpeg?20241111023820" />
   </fig>
   <fig id="fig17" position="float">
    <label>Figure 17</label>
    <caption>
     <title>Figure 17. The loss graph over epochs by ViViT.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId48.jpeg?20241111023820" />
   </fig>
   <fig id="fig18" position="float">
    <label>Figure 18</label>
    <caption>
     <title>Figure 18. The accuracy graph over epochs by ViViT with label smoothing.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId49.jpeg?20241111023820" />
   </fig>
   <fig id="fig19" position="float">
    <label>Figure 19</label>
    <caption>
     <title>Figure 19. The loss graph over epochs by ViViT with label smoothing.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId50.jpeg?20241111023821" />
   </fig>
   <fig id="fig20" position="float">
    <label>Figure 20</label>
    <caption>
     <title>Figure 20. The confusion matrix of ViViT from label a-z and _.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId51.jpeg?20241111023821" />
   </fig>
   <p>overall, it is not yet delivering optimal performance. The label “_” still shows an accuracy of 0%. We believe there is potential for further improvement. Recognizing that data simplification can often enhance model performance, we explored an innovative approach: since the diffraction pattern consists of concentric circles, we extracted a single diameter from the pattern, which retains the same amount of information. After applying this feature extraction method, we tried the LRCN 1D model due to its simplicity, having previously tested the LRCN 2D. However, the model only achieved 74% validation accuracy, which led us to move forward and try the LSTM-Attention model instead.</p>
   <p>The LSTM-Attention model, adapted from Wang et al. <xref ref-type="bibr" rid="scirp.137287-23">
     [23]
    </xref>, utilizes global average pooling (first proposed by Lin et al., 2013) to function like a fully connected layer, but with fewer parameters, thereby reducing the risk of overfitting. As anticipated, the model performs well in classifying the dataset, with a progressive decrease in loss value and an increase in accuracy, as demonstrated in <xref ref-type="fig" rid="fig21">
     Figure 21
    </xref> and <xref ref-type="fig" rid="fig22">
     Figure 22
    </xref>.</p>
   <fig id="fig21" position="float">
    <label>Figure 21</label>
    <caption>
     <title>Figure 21. The accuracy graph over epochs by LSTM-Attention model.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId52.jpeg?20241111023820" />
   </fig>
   <fig id="fig22" position="float">
    <label>Figure 22</label>
    <caption>
     <title>Figure 22. The loss graph over epochs by LSTM-Attention model.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId53.jpeg?20241111023820" />
   </fig>
   <p>Next, we validated our model using the test dataset, and the resulting confusion matrix is shown in <xref ref-type="fig" rid="fig23">
     Figure 23
    </xref>. While the model struggles to classify the “_” label and shows inaccuracy in classifying the “X” label, overall, it performs effectively in classifying most of the data.</p>
   <fig id="fig23" position="float">
    <label>Figure 23</label>
    <caption>
     <title>Figure 23. The confusion matrix of LSTM-Attention from label a-z and _.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId54.jpeg?20241111023820" />
   </fig>
   <p>The best result we achieved was 92% validation accuracy using the LSTM-Attention model. However, the accuracy for the label “_” remains 0%, indicating that the model still struggles with classifying this particular label.</p>
   <p>
    <xref ref-type="fig" rid="figFigures 24">
     Figures 24
    </xref>-<xref ref-type="bibr" rid="scirp.137287-#f26">
     26
    </xref> illustrate the prediction accuracy for characters a-z and “_” as predicted by the three models: LRCN 2D, ViViT, and LSTM-Attention. Among these, the LSTM-Attention model produced the best results. For the LRCN 2D model, the characters with prediction accuracy below 0.7 include “I, N, O, P, V, S, X”. For the ViViT model, the less accurately predicted characters are “E, R, V, X”. Meanwhile, for the LSTM-Attention model, the characters “K, Q, X” have lower prediction accuracy. The character “X” proves particularly difficult to recognize across all three models, likely due to the variations in handwriting style and order.</p>
   <p>As evident from the confusion matrices of all four models, the label “_” (indicating no writing on the optical desk) proved particularly challenging to classify across all models. This difficulty stems from the fact that the label represents various patterns in each sample, making it hard to categorize consistently. The 0% accuracy for the “_” label may be attributed to the absence of a clear, distinct</p>
   <fig id="fig24" position="float">
    <label>Figure 24</label>
    <caption>
     <title>Figure 24. The correctness of character a-z and _ by LRCN 2D model.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId55.jpeg?20241111023820" />
   </fig>
   <fig id="fig25" position="float">
    <label>Figure 25</label>
    <caption>
     <title>Figure 25. The correctness of character a-z and _ by ViViT model.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId56.jpeg?20241111023821" />
   </fig>
   <fig id="fig26" position="float">
    <label>Figure 26</label>
    <caption>
     <title>Figure 26. The correctness of character a-z and _ by LSTM-Attention model.</title>
    </caption>
    <graphic mimetype="image" position="float" xlink:type="simple" xlink:href="https://html.scirp.org/file/7505439-rId57.jpeg?20241111023821" />
   </fig>
   <p>interference pattern, as each sample could contain distinct levels of ambient noise or random signals, resulting in no recognizable vibration pattern for the models.</p>
   <p>To address this issue, exploring alternative approaches or further refining data preprocessing techniques is crucial. One potential solution involves augmenting the data to generate more representative patterns for this category. However, when we attempted to train the models with all 1300 samples for the “_” label, we encountered overfitting issues due to the large dataset size, and fine-tuning this extensive data proved too time-consuming to achieve optimal results.</p>
  </sec><sec id="s4">
   <title>4. Conclusion</title>
   <p>Our experimental results demonstrate the successful development of a method that accurately classifies vibration modes using a Michelson interferometer and machine learning techniques. For the vibration signals generated by handwriting, we ranked the performance of three models: LRCN 2D, ViViT, and LSTM-Attention. The LSTM-Attention model excelled, effectively recognizing video data of vibration patterns and achieving an impressive accuracy of 92% in identifying the characters a-z. The system showed strong performance in distinguishing different vibration patterns generated by handwriting on the optical desk, highlighting its potential for precise classification in this context. Beyond handwriting pattern classification, we believe this system has broader potential applications. The ability to recognize various vibration modes suggests that this method could be valuable in fields like earthquake prediction. This research significantly contributes to the understanding of material science and the application of machine learning in vibration pattern recognition. With further research in predictive maintenance, structural health monitoring, and other fields where vibration analysis plays a critical role, these advancements could greatly benefit society.</p>
  </sec><sec id="s5">
   <title>Acknowledgements</title>
   <p>We acknowledge the support provided by the Ministry of Education under Grant MOE A-112-01 (AITC: Promoting AI Education in Elementary and Middle Schools). We also extend our gratitude to the National Center for High-performance Computing (NCHC) for offering computational and storage resources.</p>
  </sec>
 </body><back>
  <ref-list>
   <title>References</title>
   <ref id="scirp.137287-ref1">
    <label>1</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Zhang, X., Liang, X., Zhiyuli, A., Zhang, S., Xu, R. and Wu, B. (2019) AT-LSTM: An Attention-Based LSTM Model for Financial Time Series Prediction. IOP Conference Series: Materials Science and Engineering, 569, Article 052037. &gt;https://doi.org/10.1088/1757-899x/569/5/052037 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref2">
    <label>2</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Yu, Y. and Kim, Y. (2020) Attention-LSTM-Attention Model for Speech Emotion Recognition and Analysis of IEMOCAP Database. Electronics, 9, Article 713. &gt;https://doi.org/10.3390/electronics9050713 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref3">
    <label>3</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Hu, Z. (2021) Crude Oil Price Prediction Using CEEMDAN and LSTM-Attention with News Sentiment Index. Oil&amp;Gas Science and Technology. Revue d’IFP Energies nouvelles, 76, Article No. 28. &gt;https://doi.org/10.2516/ogst/2021010 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref4">
    <label>4</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Bai, X. (2018) Text Classification Based on LSTM and Attention. 2018 Thirteenth International Conference on Digital Information Management, Berlin, 24-26 September 2018, 29-32. 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref5">
    <label>5</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Zhou, X., Wan, X. and Xiao, J. (2016) Attention-Based LSTM Network for Cross-Lingual Sentiment Classification. Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing, Austin, 1-5 November 2016, 247-256. &gt;https://doi.org/10.18653/v1/d16-1024 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref6">
    <label>6</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Liu, Z., Ning, J., Cao, Y., Wei, Y., Zhang, Z., Lin, S., et al. (2022) Video Swin Transformer. 2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition, New Orleans, 18-24 June 2022, 3192-3201. &gt;https://doi.org/10.1109/cvpr52688.2022.00320 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref7">
    <label>7</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Li, Y., Wu, C., Fan, H., Mangalam, K., Xiong, B., Malik, J., et al. (2022) MViTv2: Improved Multiscale Vision Transformers for Classification and Detection. 2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition, New Orleans, 18-24 June 2022, 4794-4804. &gt;https://doi.org/10.1109/cvpr52688.2022.00476 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref8">
    <label>8</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Yuan, H., Cai, Z., Zhou, H., Wang, Y. and Chen, X. (2021) Transanomaly: Video Anomaly Detection Using Video Vision Transformer. IEEE Access, 9, 123977-123986. &gt;https://doi.org/10.1109/access.2021.3109102 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref9">
    <label>9</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Wei, X., Zhou, L., Zhang, Z., Chen, Z. and Zhou, Y. (2019) Early Prediction of Epileptic Seizures Using a Long-Term Recurrent Convolutional Network. Journal of Neuroscience Methods, 327, Article 108395. &gt;https://doi.org/10.1016/j.jneumeth.2019.108395 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref10">
    <label>10</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Bu, S. and Cho, S. (2018) A Hybrid Deep Learning System of CNN and LRCN to Detect Cyberbullying from SNS Comments. In: Lecture Notes in Computer Science, Springer, 561-572. &gt;https://doi.org/10.1007/978-3-319-92639-1_47 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref11">
    <label>11</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Ganai, A.F. and Khursheed, F. (2023) Computationally Efficient Holistic Approach for Handwritten Urdu Recognition Using LRCN Model. International Journal of Intelligent Systems and Applications in Engineering, 11, 536-551. &gt;https://ijisae.org/index.php/IJISAE/article/view/2724 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref12">
    <label>12</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Lawall, J. and Kessler, E. (2000) Michelson Interferometry with 10 pm Accuracy. Review of Scientific Instruments, 71, 2669-2676. &gt;https://doi.org/10.1063/1.1150715 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref13">
    <label>13</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Freise, A., Chelkowski, S., Hild, S., Pozzo, W.D., Perreca, A. and Vecchio, A. (2009) Triple Michelson Interferometer for a Third-Generation Gravitational Wave Detector. Classical and Quantum Gravity, 26, Article 085012. &gt;https://doi.org/10.1088/0264-9381/26/8/085012 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref14">
    <label>14</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Monchalin, J.-P., Kelly, M.J., Thomas, J.E., Kurnit, N.A., Szöke, A., Zernike, F., et al. (1981) Accurate Laser Wavelength Measurement with a Precision Two-Beam Scanning Michelson Interferometer. Applied Optics, 20, 736-757. &gt;https://doi.org/10.1364/ao.20.000736 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref15">
    <label>15</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Park, S., Lee, J., Kim, Y. and Lee, B.H. (2020) Nanometer-Scale Vibration Measurement Using an Optical Quadrature Interferometer Based on 3×3 Fiber-Optic Coupler. Sensors, 20, Article 2665. &gt;https://doi.org/10.3390/s20092665 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref16">
    <label>16</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Cheng, J., Song, Q., Peng, H., Huang, J., Wu, H. and Jia, B. (2022) Optimization of VGG16 Algorithm Pattern Recognition for Signals of Michelson-Sagnac Interference Vibration Sensing System. Photonics, 9, Article 535. &gt;https://doi.org/10.3390/photonics9080535 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref17">
    <label>17</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Medina, R., Macancela, J., Lucero, P., Cabrera, D., Li, C., Cerrada, M., et al. (2019) A LSTM Neural Network Approach Using Vibration Signals for Classifying Faults in a Gearbox. 2019 International Conference on Sensing, Diagnostics, Prognostics, and Control, Beijing, 15-17 August 2019, 208-214. &gt;https://doi.org/10.1109/sdpc.2019.00045 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref18">
    <label>18</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Abdelmaksoud, M., Torki, M., El-Habrouk, M. and Elgeneidy, M. (2023) Convolutional-Neural-Network-Based Multi-Signals Fault Diagnosis of Induction Motor Using Single and Multi-Channels Datasets. Alexandria Engineering Journal, 73, 231-248. &gt;https://doi.org/10.1016/j.aej.2023.04.053 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref19">
    <label>19</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Yang, J., Peng, Y., Xie, J. and Wang, P. (2022) Remaining Useful Life Prediction Method for Bearings Based on LSTM with Uncertainty Quantification. Sensors, 22, Article 4549. &gt;https://doi.org/10.3390/s22124549 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref20">
    <label>20</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Donahue, J., Hendricks, L.A., Rohrbach, M., Venugopalan, S., Guadarrama, S., Saenko, K., et al. (2017) Long-Term Recurrent Convolutional Networks for Visual Recognition and Description. IEEE Transactions on Pattern Analysis and Machine Intelligence, 39, 677-691. &gt;https://doi.org/10.1109/tpami.2016.2599174 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref21">
    <label>21</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Santurkar, S., Tsipras, D., Ilyas, A. and Madry, A. (2018) How Does Batch Normalization Help Optimization? Advances in Neural Information Processing Systems, 31, 2483-2493.
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref22">
    <label>22</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lucic, M. and Schmid, C. (2021) Vivit: A Video Vision Transformer. 2021 IEEE/CVF International Conference on Computer Vision, Montreal, 10-17 October 2021, 6816-6826. &gt;https://doi.org/10.1109/iccv48922.2021.00676 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref23">
    <label>23</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Wang, Y., Huang, M., Zhu, x. and Zhao, L. (2016) Attention-Based LSTM for Aspect-Level Sentiment Classification. Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing, Austin, 1-5 November 2016, 606-615. &gt;https://doi.org/10.18653/v1/d16-1058 
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref24">
    <label>24</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Müller, R., Kornblith, S. and Hinton, G.E. (2019) When Does Label Smoothing Help? Advances in Neural Information Processing Systems, 32, 4694-4703.
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref25">
    <label>25</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Lin, M., Chen, Q. and Yan, S. (2013) Network in Network. &gt;https://doi.org/10.48550/arXiv.1312.4400
    </mixed-citation>
   </ref>
   <ref id="scirp.137287-ref26">
    <label>26</label>
    <mixed-citation publication-type="other" xlink:type="simple">
     Keser, S. and Gezer, E. (2025) Comparative Analysis of Speaker Identification Performance Using Deep Learning, Machine Learning, and Novel Subspace Classifiers with Multiple Feature Extraction Techniques. Digital Signal Processing, 156, Article 104811. &gt;https://doi.org/10.1016/j.dsp.2024.104811
    </mixed-citation>
   </ref>
  </ref-list>
 </back>
</article>